mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 00:00:17 +02:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e9e88968d7 |
No files matched your search
@@ -14,7 +14,6 @@ env:
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_FORCE32BITALLOCATOR: 1
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
@@ -25,7 +24,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/checkout@v2
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -73,11 +72,6 @@ jobs:
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: Install
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
- name: ASM Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
@@ -172,17 +166,6 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: ARMEmitter tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target emitter_tests
|
||||
|
||||
- name: ARMEmitter Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ARMEmitterTests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
@@ -205,6 +188,12 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
|
||||
|
||||
- name: Install
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
- name: Test GL No-Thunks
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -241,19 +230,13 @@ jobs:
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -1,196 +0,0 @@
|
||||
name: GLIBC fault test
|
||||
# This workflow file is the same as the `Build + Test` with some key differences
|
||||
# - Runs on any x86 and ARM64 runner
|
||||
# - Disables the glibc jemalloc compile option
|
||||
# - Enables the glibc allocator fault option
|
||||
# - Disables gvisor tests to reduce stress on CI machines (tmp/shm tests overwhelm them)
|
||||
# - Disables thunk tests since they are incompatible with glibc fault allocator
|
||||
# - Disables ARMEmitter tests (We don't want to fault test vixl's disassembler)
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
# Customize the CMake build type here (Release, Debug, RelWithDebInfo, etc.)
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_FORCE32BITALLOCATOR: 1
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
# Run on an x86 device and any ARM runner.
|
||||
arch: [[self-hosted, x64], [self-hosted, ARM64]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: Install
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
- name: ASM Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
|
||||
- name: IR Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target ir_tests
|
||||
|
||||
- name: IR Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
|
||||
|
||||
- name: Posix Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the posixtest
|
||||
run: cmake --build . --config $BUILD_TYPE --target posix_tests
|
||||
|
||||
- name: Posix Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_Posix.log || true
|
||||
|
||||
- name: gcc target tests 64
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_64
|
||||
|
||||
- name: GCC64 Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
|
||||
|
||||
- name: gcc target tests 32
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the gvisor tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
|
||||
|
||||
- name: GCC32 Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
|
||||
|
||||
- name: APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target api_tests
|
||||
|
||||
- name: APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -13,7 +13,6 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
@@ -25,7 +24,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/checkout@v2
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -111,7 +110,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
+2
-5
@@ -3,7 +3,7 @@
|
||||
path = External/vixl
|
||||
url = https://github.com/FEX-Emu/vixl.git
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = Source/Common/cpp-optparse
|
||||
path = External/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
[submodule "External/imgui"]
|
||||
path = External/imgui
|
||||
@@ -48,11 +48,8 @@
|
||||
[submodule "External/robin-map"]
|
||||
shallow = true
|
||||
path = External/robin-map
|
||||
url = https://github.com/FEX-Emu/robin-map.git
|
||||
url = https://github.com/Tessil/robin-map.git
|
||||
[submodule "External/Vulkan-Headers"]
|
||||
shallow = true
|
||||
path = External/Vulkan-Headers
|
||||
url = https://github.com/KhronosGroup/Vulkan-Headers.git
|
||||
[submodule "External/jemalloc_glibc"]
|
||||
path = External/jemalloc_glibc
|
||||
url = https://github.com/FEX-Emu/jemalloc.git
|
||||
+4
-44
@@ -7,7 +7,6 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
@@ -19,10 +18,10 @@ option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
option(ENABLE_JEMALLOC_GLIBC_ALLOC "Enables jemalloc glibc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
@@ -31,22 +30,13 @@ option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
|
||||
if (NOT CONTAINS_MINGW EQUAL -1)
|
||||
message (STATUS "Mingw build")
|
||||
set (MINGW_BUILD TRUE)
|
||||
set (ENABLE_JEMALLOC FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
@@ -58,14 +48,6 @@ if (ENABLE_FEXCORE_PROFILER)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC AND ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
|
||||
message(FATAL_ERROR "Can't have both glibc fault allocator and jemalloc glibc allocator enabled at the same time")
|
||||
endif()
|
||||
|
||||
if (ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
|
||||
add_definitions(-DGLIBC_ALLOCATOR_FAULT=1)
|
||||
endif()
|
||||
|
||||
# uninstall target
|
||||
if(NOT TARGET uninstall)
|
||||
configure_file(
|
||||
@@ -180,13 +162,8 @@ endif()
|
||||
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
add_definitions(-DTERMUX_BUILD=1)
|
||||
set(TERMUX_BUILD 1)
|
||||
|
||||
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
|
||||
# Termux builds can't rely on X11 packages
|
||||
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
|
||||
set(BUILD_FEXCONFIG FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
@@ -200,22 +177,7 @@ if (ENABLE_TSAN)
|
||||
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
# The glibc jemalloc subproject which hooks the glibc allocator.
|
||||
# Required for thunks to work.
|
||||
# All host native libraries will use this allocator, while *most* other FEX internal allocations will use the other jemalloc allocator.
|
||||
add_definitions(-DENABLE_JEMALLOC_GLIBC=1)
|
||||
add_subdirectory(External/jemalloc_glibc/)
|
||||
else()
|
||||
message (STATUS
|
||||
" jemalloc glibc allocator disabled!\n"
|
||||
" This is not a recommended configuration!\n"
|
||||
" This will very explicitly break thunk execution!\n"
|
||||
" Use at your own risk!")
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
# The jemalloc subproject that all FEXCore fextl objects allocate through.
|
||||
add_definitions(-DENABLE_JEMALLOC=1)
|
||||
add_subdirectory(External/jemalloc/)
|
||||
include_directories(External/jemalloc/pregen/include/)
|
||||
@@ -235,11 +197,6 @@ set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-poin
|
||||
|
||||
include_directories(External/robin-map/include/)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
# Enable vixl disassembler if tests are enabled.
|
||||
set(COMPILE_VIXL_DISASSEMBLER TRUE)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(External/vixl/src/)
|
||||
|
||||
@@ -267,6 +224,9 @@ if (BUILD_TESTS)
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/cpp-optparse/)
|
||||
include_directories(External/cpp-optparse/)
|
||||
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"Config": {
|
||||
"HideHypervisorBit": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -173,14 +173,6 @@
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"WaylandClient": {
|
||||
"Library" : "libwayland-client-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
}
|
||||
}
|
||||
Vendored
-1
@@ -77,7 +77,6 @@ configure_file(
|
||||
|
||||
include_directories(${CMAKE_BINARY_DIR}/generated)
|
||||
|
||||
add_compile_options(-fno-exceptions)
|
||||
add_subdirectory(Source/)
|
||||
|
||||
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
|
||||
+7
-10
@@ -22,10 +22,10 @@ def print_header():
|
||||
#define OPT_UINT64(group, enum, json, default) OPT_BASE(uint64_t, group, enum, json, default)
|
||||
#endif
|
||||
#ifndef OPT_STR
|
||||
#define OPT_STR(group, enum, json, default) OPT_BASE(fextl::string, group, enum, json, default)
|
||||
#define OPT_STR(group, enum, json, default) OPT_BASE(std::string, group, enum, json, default)
|
||||
#endif
|
||||
#ifndef OPT_STRARRAY
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_BASE(fextl::string, group, enum, json, default)
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_BASE(std::string, group, enum, json, default)
|
||||
#endif
|
||||
|
||||
'''
|
||||
@@ -371,16 +371,13 @@ def print_parse_argloader_options(options):
|
||||
|
||||
value_type = op_vals["Type"]
|
||||
NeedsString = False
|
||||
conversion_func = "fextl::fmt::format(\"{}\", "
|
||||
conversion_func = "std::to_string"
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
NeedsString = True
|
||||
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = "("
|
||||
if (value_type == "bool"):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
conversion_func = ""
|
||||
|
||||
if (value_type == "strarray"):
|
||||
# these need a bit more help
|
||||
@@ -390,11 +387,11 @@ def print_parse_argloader_options(options):
|
||||
output_argloader.write("\t}\n")
|
||||
else:
|
||||
if (NeedsString):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
output_argloader.write("\tstd::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
else:
|
||||
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
|
||||
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}(UserValue));\n".format(op_key.upper(), conversion_func))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
+11
-40
@@ -281,8 +281,9 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tvoid* Data[0];\n")
|
||||
output_file.write("\tIROps Op;\n\n")
|
||||
output_file.write("\tuint8_t Size;\n")
|
||||
output_file.write("\tuint8_t ElementSize;\n")
|
||||
output_file.write("\tuint8_t _pad;\n")
|
||||
output_file.write("\tuint8_t NumArgs;\n")
|
||||
output_file.write("\tuint8_t ElementSize : 7;\n")
|
||||
output_file.write("\tbool HasDest : 1;\n")
|
||||
|
||||
output_file.write("\ttemplate<typename T>\n")
|
||||
output_file.write("\tT const* C() const { return reinterpret_cast<T const*>(Data); }\n")
|
||||
@@ -292,7 +293,6 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tOrderedNodeWrapper Args[0];\n")
|
||||
|
||||
output_file.write("};\n\n");
|
||||
output_file.write("static_assert(sizeof(IROp_Header) == sizeof(uint32_t), \"IROp_Header should be 32-bits in size\");\n\n");
|
||||
|
||||
# Now the user defined types
|
||||
output_file.write("// User defined IR Op structs\n")
|
||||
@@ -358,10 +358,8 @@ def print_ir_sizes():
|
||||
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] std::string_view const& GetName(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetArgs(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
|
||||
|
||||
output_file.write("#undef IROP_SIZES\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -419,7 +417,7 @@ def print_ir_getname():
|
||||
def print_ir_getraargs():
|
||||
output_file.write("#ifdef IROP_GETRAARGS_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> IRRAArgs = {\n")
|
||||
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> IRArgs = {\n")
|
||||
for op in IROps:
|
||||
SSAArgs = op.SSAArgNum
|
||||
|
||||
@@ -432,18 +430,6 @@ def print_ir_getraargs():
|
||||
|
||||
output_file.write("};\n\n")
|
||||
|
||||
|
||||
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> IRArgs = {\n")
|
||||
for op in IROps:
|
||||
SSAArgs = op.SSAArgNum
|
||||
output_file.write("\t{},\n".format(SSAArgs))
|
||||
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("uint8_t GetRAArgs(IROps Op) {\n")
|
||||
output_file.write(" return IRRAArgs[Op];\n")
|
||||
output_file.write("}\n")
|
||||
|
||||
output_file.write("uint8_t GetArgs(IROps Op) {\n")
|
||||
output_file.write(" return IRArgs[Op];\n")
|
||||
output_file.write("}\n")
|
||||
@@ -467,25 +453,6 @@ def print_ir_hassideeffects():
|
||||
output_file.write("#undef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
|
||||
def print_ir_gethasdest():
|
||||
output_file.write("#ifdef IROP_GETHASDEST_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<bool, OP_LAST + 1> IRDest = {\n")
|
||||
for op in IROps:
|
||||
if op.HasDest:
|
||||
output_file.write("\ttrue,\n")
|
||||
else:
|
||||
output_file.write("\tfalse,\n")
|
||||
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("bool GetHasDest(IROps Op) {\n")
|
||||
output_file.write(" return IRDest[Op];\n")
|
||||
output_file.write("}\n")
|
||||
|
||||
output_file.write("#undef IROP_GETHASDEST_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
|
||||
# Print out IR argument printing
|
||||
def print_ir_arg_printer():
|
||||
output_file.write("#ifdef IROP_ARGPRINTER_HELPER\n")
|
||||
@@ -580,13 +547,13 @@ def print_ir_allocator_helpers():
|
||||
|
||||
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetName(HeaderOp->Op));\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT(HeaderOp->HasDest, \"Op {} has no dest\\n\", GetName(HeaderOp->Op));\n")
|
||||
output_file.write("\t\treturn HeaderOp->Size / HeaderOp->ElementSize;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn GetHasDest(HeaderOp->Op);\n")
|
||||
output_file.write("\t\treturn HeaderOp->HasDest;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tIROps GetOpType(const OrderedNode *Op) const {\n")
|
||||
@@ -664,6 +631,8 @@ def print_ir_allocator_helpers():
|
||||
|
||||
output_file.write("\t\tOp.first->Header.Size = InferSize;\n")
|
||||
|
||||
output_file.write("\t\tOp.first->Header.NumArgs = {};\n".format(op.SSAArgNum))
|
||||
|
||||
# Some ops without a destination still need an operating size
|
||||
# Effectively reusing the destination size value for operation size
|
||||
if op.DestSize != None:
|
||||
@@ -674,6 +643,9 @@ def print_ir_allocator_helpers():
|
||||
else:
|
||||
output_file.write("\t\tOp.first->Header.ElementSize = Op.first->Header.Size / ({});\n".format(op.NumElements))
|
||||
|
||||
if (op.HasDest):
|
||||
output_file.write("\t\tOp.first->Header.HasDest = true;\n")
|
||||
|
||||
# Insert validation here
|
||||
if op.EmitValidation != None:
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
@@ -761,7 +733,6 @@ print_ir_reg_classes()
|
||||
print_ir_getname()
|
||||
print_ir_getraargs()
|
||||
print_ir_hassideeffects()
|
||||
print_ir_gethasdest()
|
||||
print_ir_arg_printer()
|
||||
print_ir_allocator_helpers()
|
||||
print_ir_parser_switch_helper()
|
||||
|
||||
+20
-51
@@ -1,19 +1,13 @@
|
||||
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (FEXCORE_BASE_SRCS
|
||||
Common/Paths.cpp
|
||||
Interface/Config/Config.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/CPUInfo.cpp
|
||||
Utils/FileLoading.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list(APPEND FEXCORE_BASE_SRCS
|
||||
Utils/Allocator/64BitAllocator.cpp)
|
||||
endif()
|
||||
|
||||
set (SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Common/SoftFloat-3e/extF80_add.c
|
||||
@@ -105,11 +99,12 @@ set (SRCS
|
||||
Interface/Core/X86Tables.cpp
|
||||
Interface/Core/X86DebugInfo.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64_stubs.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/X86Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
Interface/Core/Interpreter/InterpreterFallbacks.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
@@ -142,23 +137,14 @@ set (SRCS
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/Allocator/64BitAllocator.cpp
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
)
|
||||
|
||||
if (_M_ARM_64)
|
||||
list(APPEND SRCS Utils/ArchHelpers/Arm64.cpp)
|
||||
else()
|
||||
list(APPEND SRCS Utils/ArchHelpers/Arm64_stubs.cpp)
|
||||
endif()
|
||||
|
||||
if (ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT)
|
||||
list(APPEND FEXCORE_BASE_SRCS
|
||||
Utils/AllocatorOverride.cpp)
|
||||
endif()
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
@@ -176,6 +162,11 @@ if (ENABLE_INTERPRETER)
|
||||
Interface/Core/Interpreter/VectorOps.cpp)
|
||||
endif()
|
||||
|
||||
if(_M_ARM_64)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/ArchHelpers/Arm64.cpp)
|
||||
endif()
|
||||
|
||||
set(DEFINES -DTHREAD_LOCAL=_Thread_local)
|
||||
|
||||
if (_M_X86_64)
|
||||
@@ -231,22 +222,11 @@ if (ENABLE_JIT_ARM64)
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
else()
|
||||
list (APPEND LIBS synchronization)
|
||||
endif()
|
||||
|
||||
set (LIBS fmt::fmt vixl dl xxhash tiny-json FEXHeaderUtils)
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC_GLIBC_ALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc_glibc)
|
||||
endif()
|
||||
|
||||
# Generate config
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
@@ -351,7 +331,7 @@ function(AddDefaultOptionsToTarget Name)
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
|
||||
|
||||
target_compile_definitions(${Name} PRIVATE ${DEFINES})
|
||||
add_dependencies(${Name} CONFIG_INC IR_INC)
|
||||
add_dependencies(${Name} CONFIG_INC)
|
||||
|
||||
target_compile_options(${Name}
|
||||
PRIVATE
|
||||
@@ -393,6 +373,7 @@ AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
@@ -404,16 +385,6 @@ function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
if (MINGW_BUILD)
|
||||
# Mingw build isn't building a linux shared library, so it can't have a SONAME.
|
||||
set_target_properties(${Name} PROPERTIES NO_SONAME ON)
|
||||
# Change the suffixes otherwise cmake continues using .a and .so
|
||||
if (${Type} STREQUAL SHARED)
|
||||
set_target_properties(${Name} PROPERTIES SUFFIX ".dll")
|
||||
elseif(${Type} STREQUAL STATIC)
|
||||
set_target_properties(${Name} PROPERTIES SUFFIX ".lib")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
endfunction()
|
||||
@@ -422,12 +393,10 @@ AddObject(${PROJECT_NAME}_object OBJECT)
|
||||
AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
install(TARGETS ${PROJECT_NAME} ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries
|
||||
ARCHIVE
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries)
|
||||
endif()
|
||||
install(TARGETS ${PROJECT_NAME} ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries
|
||||
ARCHIVE
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries)
|
||||
+22
-47
@@ -1,89 +1,64 @@
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <string>
|
||||
#include <unistd.h>
|
||||
|
||||
#include <fmt/format.h>
|
||||
|
||||
namespace FEXCore {
|
||||
JITSymbols::JITSymbols() {
|
||||
JITSymbols::JITSymbols() : fp{nullptr, std::fclose} {
|
||||
}
|
||||
|
||||
JITSymbols::~JITSymbols() {
|
||||
if (fd != -1) {
|
||||
close(fd);
|
||||
}
|
||||
}
|
||||
JITSymbols::~JITSymbols() = default;
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
#ifdef __ANDROID__
|
||||
// Android simpleperf looks in /data/local/tmp instead of /tmp
|
||||
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
|
||||
#else
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
#endif
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
fp.reset(fopen(PerfMap.c_str(), "wb"));
|
||||
if (fp) {
|
||||
// Disable buffering on this file
|
||||
setvbuf(fp.get(), nullptr, _IONBF, 0);
|
||||
}
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (fd == -1) return;
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
fmt::print(fp.get(), "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (fd == -1) return;
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
fmt::print(fp.get(), "{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (fd == -1) return;
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
fmt::print(fp.get(), "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (fd == -1) return;
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
fmt::print(fp.get(), "{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
|
||||
if (fd == -1) return;
|
||||
if (!fp) return;
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
fmt::print(fp.get(), "{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
+3
-1
@@ -19,6 +19,8 @@ public:
|
||||
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
|
||||
|
||||
private:
|
||||
int fd{-1};
|
||||
using FILEPtr = std::unique_ptr<FILE, decltype(&std::fclose)>;
|
||||
|
||||
FILEPtr fp;
|
||||
};
|
||||
}
|
||||
+91
@@ -0,0 +1,91 @@
|
||||
#include "Common/Paths.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstdlib>
|
||||
#include <filesystem>
|
||||
#include <memory>
|
||||
#include <pwd.h>
|
||||
#include <system_error>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Paths {
|
||||
std::unique_ptr<std::string> CachePath;
|
||||
std::unique_ptr<std::string> EntryCache;
|
||||
|
||||
char const* FindUserHomeThroughUID() {
|
||||
auto passwd = getpwuid(geteuid());
|
||||
if (passwd) {
|
||||
return passwd->pw_dir;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const char *GetHomeDirectory() {
|
||||
char const *HomeDir = getenv("HOME");
|
||||
|
||||
// Try to get home directory from uid
|
||||
if (!HomeDir) {
|
||||
HomeDir = FindUserHomeThroughUID();
|
||||
}
|
||||
|
||||
// try the PWD
|
||||
if (!HomeDir) {
|
||||
HomeDir = getenv("PWD");
|
||||
}
|
||||
|
||||
// Still doesn't exit? You get local
|
||||
if (!HomeDir) {
|
||||
HomeDir = ".";
|
||||
}
|
||||
|
||||
return HomeDir;
|
||||
}
|
||||
|
||||
void InitializePaths() {
|
||||
CachePath = std::make_unique<std::string>();
|
||||
EntryCache = std::make_unique<std::string>();
|
||||
|
||||
char const *HomeDir = getenv("HOME");
|
||||
|
||||
if (!HomeDir) {
|
||||
HomeDir = getenv("PWD");
|
||||
}
|
||||
|
||||
if (!HomeDir) {
|
||||
HomeDir = ".";
|
||||
}
|
||||
|
||||
char *XDGDataDir = getenv("XDG_DATA_DIR");
|
||||
if (XDGDataDir) {
|
||||
*CachePath = XDGDataDir;
|
||||
}
|
||||
else {
|
||||
if (HomeDir) {
|
||||
*CachePath = HomeDir;
|
||||
}
|
||||
}
|
||||
|
||||
*CachePath += "/.fex-emu/";
|
||||
*EntryCache = *CachePath + "/EntryCache/";
|
||||
|
||||
std::error_code ec{};
|
||||
// Ensure the folder structure is created for our Data
|
||||
if (!std::filesystem::exists(*EntryCache, ec) &&
|
||||
!std::filesystem::create_directories(*EntryCache, ec)) {
|
||||
LogMan::Msg::DFmt("Couldn't create EntryCache directory: '{}'", *EntryCache);
|
||||
}
|
||||
}
|
||||
|
||||
void ShutdownPaths() {
|
||||
CachePath.reset();
|
||||
EntryCache.reset();
|
||||
}
|
||||
|
||||
std::string GetCachePath() {
|
||||
return *CachePath;
|
||||
}
|
||||
|
||||
std::string GetEntryCachePath() {
|
||||
return *EntryCache;
|
||||
}
|
||||
}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
#pragma once
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore::Paths {
|
||||
void InitializePaths();
|
||||
void ShutdownPaths();
|
||||
|
||||
const char *GetHomeDirectory();
|
||||
|
||||
std::string GetCachePath();
|
||||
std::string GetEntryCachePath();
|
||||
}
|
||||
+23
-38
@@ -2,19 +2,19 @@
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
#include <sstream>
|
||||
|
||||
extern "C" {
|
||||
#include "SoftFloat-3e/platform.h"
|
||||
#include "SoftFloat-3e/softfloat.h"
|
||||
}
|
||||
|
||||
struct FEX_PACKED X80SoftFloat {
|
||||
struct X80SoftFloat {
|
||||
#ifdef _M_X86_64
|
||||
// Define this to push some operations to x87
|
||||
// Only useful to see if precision loss is killing something
|
||||
@@ -32,28 +32,22 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#else
|
||||
#error No 128bit float for this target!
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
#define LIBRARY_PRECISION BIGFLOAT
|
||||
#else
|
||||
// Mingw Win32 libraries don't have `__float128` helpers. Needs to use a lower precision.
|
||||
#define LIBRARY_PRECISION double
|
||||
#endif
|
||||
|
||||
uint64_t Significand : 64;
|
||||
uint16_t Exponent : 15;
|
||||
uint16_t Sign : 1;
|
||||
struct __attribute__((packed)) {
|
||||
uint64_t Significand : 64;
|
||||
uint16_t Exponent : 15;
|
||||
unsigned Sign : 1;
|
||||
};
|
||||
|
||||
X80SoftFloat() { memset(this, 0, sizeof(*this)); }
|
||||
X80SoftFloat(uint16_t _Sign, uint16_t _Exponent, uint64_t _Significand)
|
||||
X80SoftFloat(unsigned _Sign, uint16_t _Exponent, uint64_t _Significand)
|
||||
: Significand {_Significand}
|
||||
, Exponent {_Exponent}
|
||||
, Sign {_Sign}
|
||||
{
|
||||
}
|
||||
|
||||
fextl::string str() const {
|
||||
fextl::ostringstream string;
|
||||
std::string str() const {
|
||||
std::ostringstream string;
|
||||
string << std::hex << Sign;
|
||||
string << "_" << Exponent;
|
||||
string << "_" << (Significand >> 63);
|
||||
@@ -268,7 +262,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Int = FRNDINT(rhs, softfloat_round_minMag);
|
||||
LIBRARY_PRECISION Src2_d = Int;
|
||||
BIGFLOAT Src2_d = Int;
|
||||
Src2_d = exp2l(Src2_d);
|
||||
X80SoftFloat Src2_X80 = Src2_d;
|
||||
X80SoftFloat Result = extF80_mul(lhs, Src2_X80);
|
||||
@@ -292,8 +286,8 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src1_d = lhs;
|
||||
LIBRARY_PRECISION Result = exp2l(Src1_d);
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Result = exp2l(Src1_d);
|
||||
Result -= 1.0;
|
||||
return Result;
|
||||
#endif
|
||||
@@ -317,9 +311,9 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src1_d = lhs;
|
||||
LIBRARY_PRECISION Src2_d = rhs;
|
||||
LIBRARY_PRECISION Tmp = Src2_d * log2l(Src1_d);
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = Src2_d * log2l(Src1_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
@@ -342,9 +336,9 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src1_d = lhs;
|
||||
LIBRARY_PRECISION Src2_d = rhs;
|
||||
LIBRARY_PRECISION Tmp = atan2l(Src1_d, Src2_d);
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = atan2l(Src1_d, Src2_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
@@ -366,7 +360,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src_d = lhs;
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = tanl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
@@ -388,7 +382,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src_d = lhs;
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = sinl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
@@ -410,7 +404,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
LIBRARY_PRECISION Src_d = lhs;
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = cosl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
@@ -445,7 +439,6 @@ struct FEX_PACKED X80SoftFloat {
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
operator BIGFLOAT() const {
|
||||
#if BIGFLOATSIZE == 16
|
||||
const float128_t Result = extF80_to_f128(*this);
|
||||
@@ -456,7 +449,6 @@ struct FEX_PACKED X80SoftFloat {
|
||||
return result;
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
operator int16_t() const {
|
||||
auto rv = extF80_to_i32(*this, softfloat_roundingMode, false);
|
||||
@@ -525,7 +517,6 @@ struct FEX_PACKED X80SoftFloat {
|
||||
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
X80SoftFloat(BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(FEXCore::BitCast<float128_t>(rhs));
|
||||
@@ -533,7 +524,6 @@ struct FEX_PACKED X80SoftFloat {
|
||||
*this = FEXCore::BitCast<long double>(rhs);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
X80SoftFloat(const int16_t rhs) {
|
||||
*this = i32_to_extF80(rhs);
|
||||
@@ -572,9 +562,4 @@ private:
|
||||
static constexpr uint32_t ExponentBias = 16383;
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
static_assert(sizeof(X80SoftFloat) == 10, "tword must be 10bytes in size");
|
||||
#else
|
||||
// Padding on this extends to 16-bytes rather than 10-bytes on WIN32.
|
||||
static_assert(sizeof(X80SoftFloat) == 16, "tword must be 16bytes in size");
|
||||
#endif
|
||||
+12
-13
@@ -1,49 +1,48 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::StrConv {
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, bool *Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint8_t *Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint16_t *Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint32_t *Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, int32_t *Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
*Result = std::stoi(std::string(Value), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint64_t *Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
*Result = std::stoull(std::string(Value), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, std::string *Result) {
|
||||
*Result = Value;
|
||||
return true;
|
||||
}
|
||||
template <typename T,
|
||||
typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, T *Result) {
|
||||
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
|
||||
*Result = static_cast<T>(std::stoull(std::string(Value), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, fextl::string *Result) {
|
||||
*Result = Value;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
+9
-10
@@ -1,11 +1,11 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore::StringUtils {
|
||||
// Trim the left side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static fextl::string LeftTrim(fextl::string String, std::string_view TrimTokens = " \t\n\r") {
|
||||
size_t pos = fextl::string::npos;
|
||||
if ((pos = String.find_first_not_of(TrimTokens)) != fextl::string::npos) {
|
||||
[[maybe_unused]] static std::string LeftTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_first_not_of(TrimTokens)) != std::string::npos) {
|
||||
String.erase(0, pos);
|
||||
}
|
||||
|
||||
@@ -13,9 +13,9 @@ namespace FEXCore::StringUtils {
|
||||
}
|
||||
|
||||
// Trim the right side of the string of whitespace and new lines
|
||||
[[maybe_unused]] static fextl::string RightTrim(fextl::string String, std::string_view TrimTokens = " \t\n\r") {
|
||||
size_t pos = fextl::string::npos;
|
||||
if ((pos = String.find_last_not_of(TrimTokens)) != fextl::string::npos) {
|
||||
[[maybe_unused]] static std::string RightTrim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
size_t pos = std::string::npos;
|
||||
if ((pos = String.find_last_not_of(TrimTokens)) != std::string::npos) {
|
||||
String.erase(String.begin() + pos + 1, String.end());
|
||||
}
|
||||
|
||||
@@ -23,8 +23,7 @@ namespace FEXCore::StringUtils {
|
||||
}
|
||||
|
||||
// Trim both the left and right of the string of whitespace and new lines
|
||||
[[maybe_unused]] static fextl::string Trim(fextl::string String, std::string_view TrimTokens = " \t\n\r") {
|
||||
return RightTrim(LeftTrim(std::move(String), TrimTokens), TrimTokens);
|
||||
[[maybe_unused]] static std::string Trim(std::string String, std::string TrimTokens = " \t\n\r") {
|
||||
return RightTrim(LeftTrim(String, TrimTokens), TrimTokens);
|
||||
}
|
||||
|
||||
}
|
||||
+338
-89
@@ -1,34 +1,36 @@
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "Common/Paths.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <cstdlib>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <list>
|
||||
#include <optional>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <sys/sysinfo.h>
|
||||
#include <system_error>
|
||||
#include <type_traits>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class Context;
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Config {
|
||||
@@ -40,75 +42,173 @@ namespace DefaultValues {
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
PATH_CONFIG_DIR_GLOBAL,
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
PATH_CONFIG_FILE_GLOBAL,
|
||||
PATH_LAST,
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
std::unique_ptr<std::list<json_t>> json_objects;
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
void SetDataDirectory(const std::string_view Path) {
|
||||
Paths[PATH_DATA_DIR] = Path;
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = std::make_unique<std::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
void SetConfigDirectory(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_DIR_LOCAL + Global] = Path;
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
void SetConfigFileLocation(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
|
||||
static void LoadJSonConfig(const std::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
std::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
|
||||
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
}
|
||||
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
}
|
||||
|
||||
for (json_t const* ConfigItem = json_getChild(ConfigList);
|
||||
ConfigItem != nullptr;
|
||||
ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::string GetDataDirectory() {
|
||||
std::string DataDir{};
|
||||
|
||||
char const *HomeDir = Paths::GetHomeDirectory();
|
||||
char const *DataXDG = getenv("XDG_DATA_HOME");
|
||||
char const *DataOverride = getenv("FEX_APP_DATA_LOCATION");
|
||||
if (DataOverride) {
|
||||
// Data override will override the complete directory
|
||||
DataDir = DataOverride;
|
||||
}
|
||||
else {
|
||||
DataDir = DataXDG ?: HomeDir;
|
||||
DataDir += "/.fex-emu/";
|
||||
}
|
||||
return DataDir;
|
||||
}
|
||||
|
||||
fextl::string const& GetDataDirectory() {
|
||||
return Paths[PATH_DATA_DIR];
|
||||
std::string GetConfigDirectory(bool Global) {
|
||||
std::string ConfigDir;
|
||||
if (Global) {
|
||||
ConfigDir = GLOBAL_DATA_DIRECTORY;
|
||||
}
|
||||
else {
|
||||
char const *HomeDir = Paths::GetHomeDirectory();
|
||||
char const *ConfigXDG = getenv("XDG_CONFIG_HOME");
|
||||
char const *ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (ConfigOverride) {
|
||||
// Config override completely overrides the config directory
|
||||
ConfigDir = ConfigOverride;
|
||||
}
|
||||
else {
|
||||
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
|
||||
ConfigDir += "/.fex-emu/";
|
||||
}
|
||||
|
||||
// Ensure the folder structure is created for our configuration
|
||||
std::error_code ec{};
|
||||
if (!std::filesystem::exists(ConfigDir, ec) &&
|
||||
!std::filesystem::create_directories(ConfigDir, ec)) {
|
||||
// Let's go local in this case
|
||||
return "./";
|
||||
}
|
||||
}
|
||||
|
||||
return ConfigDir;
|
||||
}
|
||||
|
||||
fextl::string const& GetConfigDirectory(bool Global) {
|
||||
return Paths[PATH_CONFIG_DIR_LOCAL + Global];
|
||||
std::string GetConfigFileLocation(bool Global) {
|
||||
std::string ConfigFile{};
|
||||
if (Global) {
|
||||
ConfigFile = GetConfigDirectory(true) + "Config.json";
|
||||
}
|
||||
else {
|
||||
const char *AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig) {
|
||||
// App config environment variable overwrites only the config file
|
||||
ConfigFile = AppConfig;
|
||||
}
|
||||
else {
|
||||
ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
}
|
||||
}
|
||||
return ConfigFile;
|
||||
}
|
||||
|
||||
fextl::string const& GetConfigFileLocation(bool Global) {
|
||||
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
|
||||
}
|
||||
|
||||
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
|
||||
fextl::string ConfigFile = GetConfigDirectory(Global);
|
||||
std::string GetApplicationConfig(const std::string &Filename, bool Global) {
|
||||
std::string ConfigFile = GetConfigDirectory(Global);
|
||||
|
||||
std::error_code ec{};
|
||||
if (!Global &&
|
||||
!FHU::Filesystem::Exists(ConfigFile) &&
|
||||
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
|
||||
!std::filesystem::exists(ConfigFile, ec) &&
|
||||
!std::filesystem::create_directories(ConfigFile, ec)) {
|
||||
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
|
||||
// Let's go local in this case
|
||||
return fextl::fmt::format("./{}.json", Program);
|
||||
return "./" + Filename + ".json";
|
||||
}
|
||||
|
||||
ConfigFile += "AppConfig/";
|
||||
|
||||
// Attempt to create the local folder if it doesn't exist
|
||||
if (!Global &&
|
||||
!FHU::Filesystem::Exists(ConfigFile) &&
|
||||
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
|
||||
!std::filesystem::exists(ConfigFile, ec) &&
|
||||
!std::filesystem::create_directories(ConfigFile, ec)) {
|
||||
// Let's go local in this case
|
||||
return fextl::fmt::format("./{}.json", Program);
|
||||
return "./" + Filename + ".json";
|
||||
}
|
||||
|
||||
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
|
||||
ConfigFile += Filename + ".json";
|
||||
return ConfigFile;
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, fextl::string const &Config) {
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, std::string const &Config) {
|
||||
}
|
||||
|
||||
uint64_t GetConfig(FEXCore::Context::Context *CTX, ConfigOption Option) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
|
||||
@@ -168,17 +268,17 @@ namespace DefaultValues {
|
||||
}
|
||||
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
std::unordered_map<std::string, std::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](FEXCore::Config::LayerValue const &Value) {
|
||||
for (const auto &EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == fextl::string::npos) {
|
||||
if (ItEq == std::string::npos) {
|
||||
// Broken environment variable
|
||||
// Skip
|
||||
continue;
|
||||
}
|
||||
auto Key = fextl::string(EnvVar.begin(), EnvVar.begin() + ItEq);
|
||||
auto Value = fextl::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
|
||||
auto Key = std::string(EnvVar.begin(), EnvVar.begin() + ItEq);
|
||||
auto Value = std::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
|
||||
|
||||
// Add the key to the map, overwriting whatever previous value was there
|
||||
LookupMap.insert_or_assign(std::move(Key), std::move(Value));
|
||||
@@ -211,7 +311,7 @@ namespace DefaultValues {
|
||||
}
|
||||
|
||||
void Initialize() {
|
||||
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
|
||||
AddLayer(std::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
|
||||
Meta = ConfigLayers.begin()->second.get();
|
||||
}
|
||||
|
||||
@@ -229,15 +329,16 @@ namespace DefaultValues {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::string ExpandPath(fextl::string const &ContainerPrefix, fextl::string PathName) {
|
||||
std::string ExpandPath(std::string const &ContainerPrefix, std::string PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
std::filesystem::path Path{PathName};
|
||||
|
||||
// Expand home if it exists
|
||||
if (FHU::Filesystem::IsRelative(PathName)) {
|
||||
fextl::string Home = getenv("HOME") ?: "";
|
||||
if (Path.is_relative()) {
|
||||
std::string Home = getenv("HOME") ?: "";
|
||||
// Home expansion only works if it is the first character
|
||||
// This matches bash behaviour
|
||||
if (PathName.at(0) == '~') {
|
||||
@@ -246,15 +347,12 @@ namespace DefaultValues {
|
||||
}
|
||||
|
||||
// Expand relative path to absolute
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
char *RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
|
||||
if (RealPath) {
|
||||
PathName = RealPath;
|
||||
}
|
||||
Path = std::filesystem::absolute(Path);
|
||||
|
||||
// Only return if it exists
|
||||
if (FHU::Filesystem::Exists(PathName)) {
|
||||
return PathName;
|
||||
std::error_code ec{};
|
||||
if (std::filesystem::exists(Path, ec)) {
|
||||
return Path;
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -270,9 +368,9 @@ namespace DefaultValues {
|
||||
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
|
||||
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
|
||||
if (!ContainerPrefix.empty() && !PathName.empty()) {
|
||||
if (!FHU::Filesystem::Exists(PathName)) {
|
||||
if (!std::filesystem::exists(PathName)) {
|
||||
auto ContainerPath = ContainerPrefix + PathName;
|
||||
if (FHU::Filesystem::Exists(ContainerPath)) {
|
||||
if (std::filesystem::exists(ContainerPath)) {
|
||||
return ContainerPath;
|
||||
}
|
||||
}
|
||||
@@ -281,15 +379,15 @@ namespace DefaultValues {
|
||||
return {};
|
||||
}
|
||||
|
||||
constexpr char ContainerManager[] = "/run/host/container-manager";
|
||||
|
||||
fextl::string FindContainer() {
|
||||
std::string FindContainer() {
|
||||
// We only support pressure-vessel at the moment
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::vector<char> Manager{};
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
return ManagerStr;
|
||||
}
|
||||
@@ -297,13 +395,14 @@ namespace DefaultValues {
|
||||
return {};
|
||||
}
|
||||
|
||||
fextl::string FindContainerPrefix() {
|
||||
std::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::vector<char> Manager{};
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
|
||||
// We are running inside of pressure vessel
|
||||
@@ -325,7 +424,7 @@ namespace DefaultValues {
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
if (Cores == 0) {
|
||||
// When the number of emulated CPU cores is zero then auto detect
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, fextl::fmt::format("{}", FEXCore::CPUInfo::CalculateNumberOfCPUs()));
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, std::to_string(get_nprocs_conf()));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -344,7 +443,7 @@ namespace DefaultValues {
|
||||
#endif
|
||||
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, std::to_string(FEXCore::Config::CONFIG_IRJIT));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -358,8 +457,8 @@ namespace DefaultValues {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
|
||||
std::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
@@ -368,15 +467,16 @@ namespace DefaultValues {
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
|
||||
FEX_CONFIG_OPT(PathName, ROOTFS);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix,PathName());
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
}
|
||||
else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
std::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
std::error_code ec{};
|
||||
if (std::filesystem::exists(NamedRootFS, ec)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
@@ -398,8 +498,9 @@ namespace DefaultValues {
|
||||
}
|
||||
else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
std::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
std::error_code ec{};
|
||||
if (std::filesystem::exists(NamedConfig, ec)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
@@ -413,11 +514,11 @@ namespace DefaultValues {
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
|
||||
// Single stepping also enforces single instruction size blocks
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, std::to_string(1u));
|
||||
}
|
||||
}
|
||||
|
||||
void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer) {
|
||||
void AddLayer(std::unique_ptr<FEXCore::Config::Layer> _Layer) {
|
||||
ConfigLayers.emplace(_Layer->GetLayerType(), std::move(_Layer));
|
||||
}
|
||||
|
||||
@@ -429,7 +530,7 @@ namespace DefaultValues {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
std::optional<std::string*> Get(ConfigOption Option) {
|
||||
return Meta->Get(Option);
|
||||
}
|
||||
|
||||
@@ -470,7 +571,7 @@ namespace DefaultValues {
|
||||
}
|
||||
|
||||
template<>
|
||||
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, fextl::string Default) {
|
||||
std::string Value<std::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
@@ -481,13 +582,13 @@ namespace DefaultValues {
|
||||
}
|
||||
|
||||
template<>
|
||||
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
std::string Value<std::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
}
|
||||
else {
|
||||
return fextl::string(Default);
|
||||
return std::string(Default);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -502,19 +603,167 @@ namespace DefaultValues {
|
||||
template uint64_t Value<uint64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint64_t Default);
|
||||
|
||||
// Constructor
|
||||
template Value<fextl::string>::Value(FEXCore::Config::ConfigOption _Option, fextl::string Default);
|
||||
template Value<std::string>::Value(FEXCore::Config::ConfigOption _Option, std::string Default);
|
||||
template Value<bool>::Value(FEXCore::Config::ConfigOption _Option, bool Default);
|
||||
template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t Default);
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List) {
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, std::list<std::string> *List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
template void Value<std::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, std::list<std::string> *List);
|
||||
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(std::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
std::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
std::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit EnvLoader(char *const _envp[]);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const std::map<std::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
static const std::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
|
||||
: FEXCore::Config::Layer(Layer) {
|
||||
}
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
auto it = ConfigLookup.find(ConfigName);
|
||||
if (it != ConfigLookup.end()) {
|
||||
Set(it->second, ConfigString);
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(std::string ConfigFile)
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
|
||||
void MainLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const std::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char *const _envp[])
|
||||
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
|
||||
, envp {_envp} {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
std::unordered_map<std::string_view, std::string_view> EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (std::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
std::string_view Value {Var.substr(pos+1)};
|
||||
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key]=Value;
|
||||
}
|
||||
|
||||
std::function GetVar = [=](const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
// If envp[] was empty, search using std::getenv()
|
||||
const char* vs = std::getenv(id.data());
|
||||
if (vs) {
|
||||
return vs;
|
||||
}
|
||||
else {
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(it.first)).has_value()) {
|
||||
Set(it.second, std::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
|
||||
if (File) {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return std::make_unique<FEXCore::Config::EnvLoader>(_envp);
|
||||
}
|
||||
}
|
||||
|
||||
+6
-15
@@ -53,7 +53,7 @@
|
||||
},
|
||||
"EnableAVX": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Determines whether or not we use the expanded register file for AVX or not"
|
||||
]
|
||||
@@ -240,17 +240,6 @@
|
||||
"Also needs x86_64-linux-gnu-objdump in PATH.",
|
||||
"Can be very slow."
|
||||
]
|
||||
},
|
||||
"InjectLibSegFault": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Sets the environment variable LD_PRELOAD=libSegFault.so",
|
||||
"This allows the user to very easily enable libSegFault without dealing with environment variables",
|
||||
"Very useful for applications that have launch scripts that set the variable to nothing at launch",
|
||||
"Set this in an application configuration for injecting in to only specific applications.",
|
||||
"\tNote: If x86/x86_64 libSegFault.so isn't installed then this option won't work."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Logging": {
|
||||
@@ -343,12 +332,14 @@
|
||||
"Useful for a process that keeps restarting and doesn't work"
|
||||
]
|
||||
},
|
||||
"HideHypervisorBit": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Hides the hypervisor CPUID bit when set.",
|
||||
"Should only be used for applications that have issues with this set."
|
||||
"An application that uses try-catch or longjump extensively needs the ability to do context aware state flushing",
|
||||
"In the case of FEX's block-linking, it won't always ensure that RIP is synchronized.",
|
||||
"If an exception occurs and RIP isn't synchronized, then FEX's exception stack restore may not long jump as expected",
|
||||
"Can be useful for Wine applications that rely on stack unwinding"
|
||||
]
|
||||
}
|
||||
},
|
||||
|
||||
+181
-35
@@ -1,3 +1,4 @@
|
||||
#include "Common/Paths.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
@@ -18,76 +19,221 @@ namespace FEXCore::HLE {
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void InitializeStaticTables(OperatingMode Mode) {
|
||||
FEXCore::Paths::InitializePaths();
|
||||
X86Tables::InitializeInfoTables(Mode);
|
||||
IR::InstallOpcodeHandlers(Mode);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext() {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>();
|
||||
void ShutdownStaticTables() {
|
||||
FEXCore::Paths::ShutdownPaths();
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::InitializeContext() {
|
||||
return FEXCore::CPU::CreateCPUCore(this);
|
||||
FEXCore::Context::Context *CreateNewContext() {
|
||||
return new FEXCore::Context::Context{};
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
CustomExitHandler = std::move(handler);
|
||||
bool InitializeContext(FEXCore::Context::Context *CTX) {
|
||||
return FEXCore::CPU::CreateCPUCore(CTX);
|
||||
}
|
||||
|
||||
ExitHandler FEXCore::Context::ContextImpl::GetExitHandler() const {
|
||||
return CustomExitHandler;
|
||||
void DestroyContext(FEXCore::Context::Context *CTX) {
|
||||
if (CTX->ParentThread) {
|
||||
CTX->DestroyThread(CTX->ParentThread);
|
||||
}
|
||||
delete CTX;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::Stop() {
|
||||
Stop(false);
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
return CTX->InitCore(InitialRIP, StackPointer);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler) {
|
||||
CTX->CustomExitHandler = std::move(handler);
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason FEXCore::Context::ContextImpl::GetExitReason() {
|
||||
return ParentThread->ExitReason;
|
||||
ExitHandler GetExitHandler(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->CustomExitHandler;
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::IsDone() const {
|
||||
return IsPaused();
|
||||
void Run(FEXCore::Context::Context *CTX) {
|
||||
CTX->Run();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::GetCPUState(FEXCore::Core::CPUState *State) const {
|
||||
memcpy(State, ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
void Step(FEXCore::Context::Context *CTX) {
|
||||
CTX->Step();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCPUState(const FEXCore::Core::CPUState *State) {
|
||||
memcpy(ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
Thread->CTX->CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
FEXCore::Context::ExitReason RunUntilExit(FEXCore::Context::Context *CTX) {
|
||||
return CTX->RunUntilExit();
|
||||
}
|
||||
|
||||
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
|
||||
return HostFeatures;
|
||||
int GetProgramStatus(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->GetProgramStatus();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator *_SignalDelegation) {
|
||||
SignalDelegation = _SignalDelegation;
|
||||
FEXCore::Context::ExitReason GetExitReason(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->ParentThread->ExitReason;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) {
|
||||
SyscallHandler = Handler;
|
||||
SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
bool IsDone(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->IsPaused();
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
|
||||
return CPUID.RunFunction(Function, Leaf);
|
||||
void GetCPUState(const FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
memcpy(State, CTX->ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
|
||||
return CPUID.RunXCRFunction(Function);
|
||||
void SetCPUState(FEXCore::Context::Context *CTX, const FEXCore::Core::CPUState *State) {
|
||||
memcpy(CTX->ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
void Pause(FEXCore::Context::Context *CTX) {
|
||||
CTX->Pause();
|
||||
}
|
||||
|
||||
void Stop(FEXCore::Context::Context *CTX) {
|
||||
CTX->Stop(false);
|
||||
}
|
||||
|
||||
void SetCustomCPUBackendFactory(FEXCore::Context::Context *CTX, CustomCPUFactoryType Factory) {
|
||||
CTX->CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
bool AddVirtualMemoryMapping([[maybe_unused]] FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t VirtualAddress, [[maybe_unused]] uint64_t PhysicalAddress, [[maybe_unused]] uint64_t Size) {
|
||||
return false;
|
||||
}
|
||||
|
||||
void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t Syscall, [[maybe_unused]] FEXCore::HLE::SyscallVisitor *Visitor) {
|
||||
}
|
||||
|
||||
HostFeatures GetHostFeatures(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->HostFeatures;
|
||||
}
|
||||
|
||||
void HandleCallback(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
CTX->HandleCallback(Thread, RIP);
|
||||
}
|
||||
|
||||
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterHostSignalHandler(Signal, std::move(Func), Required);
|
||||
}
|
||||
|
||||
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterFrontendHostSignalHandler(Signal, std::move(Func), Required);
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
return CTX->CreateThread(NewThreadState, ParentTID);
|
||||
}
|
||||
|
||||
void ExecutionThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return CTX->ExecutionThread(Thread);
|
||||
}
|
||||
|
||||
void InitializeThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return CTX->InitializeThread(Thread);
|
||||
}
|
||||
|
||||
void RunThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
CTX->RunThread(Thread);
|
||||
}
|
||||
|
||||
void StopThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
CTX->StopThread(Thread);
|
||||
}
|
||||
|
||||
void DestroyThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
CTX->DestroyThread(Thread);
|
||||
}
|
||||
|
||||
void CleanupAfterFork(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
|
||||
CTX->CleanupAfterFork(Thread);
|
||||
}
|
||||
|
||||
void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation) {
|
||||
CTX->SignalDelegation = SignalDelegation;
|
||||
}
|
||||
|
||||
void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler) {
|
||||
CTX->SyscallHandler = Handler;
|
||||
CTX->SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf) {
|
||||
return CTX->CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CTX->CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
}
|
||||
|
||||
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader) {
|
||||
CTX->SetAOTIRLoader(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
|
||||
CTX->SetAOTIRWriter(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(FEXCore::Context::Context *CTX, std::function<void(const std::string&)> CacheRenamer) {
|
||||
CTX->SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache(FEXCore::Context::Context *CTX) {
|
||||
CTX->FinalizeAOTIRCache();
|
||||
}
|
||||
|
||||
void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
CTX->WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(FEXCore::Context::Context *CTX, const std::string &Name) {
|
||||
return CTX->LoadAOTIRCacheEntry(Name);
|
||||
}
|
||||
void UnloadAOTIRCacheEntry(FEXCore::Context::Context *CTX, IR::AOTIRCacheEntry *Entry) {
|
||||
return CTX->UnloadAOTIRCacheEntry(Entry);
|
||||
}
|
||||
|
||||
CustomIRResult AddCustomIREntrypoint(FEXCore::Context::Context *CTX, uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
return CTX->AddCustomIREntrypoint(Entrypoint, Handler, Creator, Data);
|
||||
}
|
||||
|
||||
void AppendThunkDefinitions(FEXCore::Context::Context *CTX, std::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
CTX->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
void CompileRIP(FEXCore::Context::Context *CTX, uint64_t RIP) {
|
||||
CTX->CompileRIP(CTX->ParentThread, RIP);
|
||||
}
|
||||
uint64_t GetThreadCount(FEXCore::Context::Context *CTX) {
|
||||
return CTX->GetThreadCount();
|
||||
}
|
||||
|
||||
FEXCore::Core::RuntimeStats *GetRuntimeStatsForThread(FEXCore::Context::Context *CTX, uint64_t Thread) {
|
||||
return CTX->GetRuntimeStatsForThread(Thread);
|
||||
}
|
||||
|
||||
bool GetDebugDataForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
return CTX->GetDebugDataForRIP(RIP, Data);
|
||||
}
|
||||
|
||||
bool FindHostCodeForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, uint8_t **Code) {
|
||||
return CTX->FindHostCodeForRIP(RIP, Code);
|
||||
}
|
||||
|
||||
// XXX:
|
||||
// bool FindIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir) {
|
||||
// return CTX->FindIRForRIP(RIP, ir);
|
||||
// }
|
||||
|
||||
// void SetIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir) {
|
||||
// CTX->SetIRForRIP(RIP, ir);
|
||||
// }
|
||||
}
|
||||
|
||||
}
|
||||
+118
-178
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "FEXHeaderUtils/ScopedSignalMask.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
@@ -13,25 +14,24 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <stdint.h>
|
||||
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <functional>
|
||||
#include <istream>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <stddef.h>
|
||||
#include <string>
|
||||
#include <unordered_map>
|
||||
#include <queue>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
@@ -70,128 +70,7 @@ namespace FEXCore::Context {
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitializeContext() override;
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
|
||||
|
||||
void SetExitHandler(ExitHandler handler) override;
|
||||
ExitHandler GetExitHandler() const override;
|
||||
|
||||
void Pause() override;
|
||||
void Run() override;
|
||||
void Stop() override;
|
||||
void Step() override;
|
||||
|
||||
ExitReason RunUntilExit() override;
|
||||
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
|
||||
|
||||
int GetProgramStatus() const override;
|
||||
|
||||
ExitReason GetExitReason() override;
|
||||
|
||||
bool IsDone() const override;
|
||||
|
||||
void GetCPUState(FEXCore::Core::CPUState *State) const override;
|
||||
void SetCPUState(const FEXCore::Core::CPUState *State) override;
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - InitializeThread(Thread);
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
/**
|
||||
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* The OS thread will wait until RunThread is executed
|
||||
*/
|
||||
void InitializeThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
/**
|
||||
* @brief Starts the OS thread object to start executing guest code
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void RunThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void StopThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override;
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) override;
|
||||
|
||||
void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(CacheReader);
|
||||
}
|
||||
void SetAOTIRWriter(std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)> CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(CacheWriter);
|
||||
}
|
||||
void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache() override {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) override;
|
||||
void MarkMemoryShared() override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator = nullptr, void *Data = nullptr) override;
|
||||
|
||||
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
|
||||
|
||||
public:
|
||||
struct Context {
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
friend class FEXCore::CPU::Arm64JITCore;
|
||||
@@ -237,13 +116,15 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(x86dec_SynchronizeRIPOnAllBlocks, X86DEC_SYNCHRONIZERIPONALLBLOCKS);
|
||||
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
|
||||
} Config;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
|
||||
std::mutex ThreadCreationMutex;
|
||||
FEXCore::Core::InternalThreadState* ParentThread{};
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*> Threads;
|
||||
FEXCore::Core::InternalThreadState* ParentThread;
|
||||
std::vector<FEXCore::Core::InternalThreadState*> Threads;
|
||||
std::atomic_bool CoreShuttingDown{false};
|
||||
bool NeedToCheckXID{true};
|
||||
|
||||
@@ -254,44 +135,53 @@ namespace FEXCore::Context {
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
std::shared_mutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
|
||||
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
#endif
|
||||
|
||||
SignalDelegator *SignalDelegation{};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
ContextImpl();
|
||||
~ContextImpl();
|
||||
Context();
|
||||
~Context();
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer);
|
||||
FEXCore::Context::ExitReason RunUntilExit();
|
||||
int GetProgramStatus() const;
|
||||
bool IsPaused() const { return !Running; }
|
||||
void Pause();
|
||||
void Run();
|
||||
void WaitForThreadsToRun();
|
||||
void Step();
|
||||
void Stop(bool IgnoreCurrentThread);
|
||||
void WaitForIdle();
|
||||
void StopThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
|
||||
|
||||
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
|
||||
void StartGdbServer();
|
||||
void StopGdbServer();
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
FHU::ScopedSignalMaskWithSharedLock lk(Frame->Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
@@ -300,15 +190,26 @@ namespace FEXCore::Context {
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data);
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
// Debugger interface
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
uint64_t GetThreadCount() const;
|
||||
FEXCore::Core::RuntimeStats *GetRuntimeStatsForThread(uint64_t Thread);
|
||||
bool GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data);
|
||||
bool FindHostCodeForRIP(uint64_t RIP, uint8_t **Code);
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
@@ -335,6 +236,29 @@ namespace FEXCore::Context {
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - InitializeThread(Thread);
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
|
||||
|
||||
/**
|
||||
* @brief Initializes TID, PID and TLS data for a thread
|
||||
*
|
||||
@@ -342,57 +266,76 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* The OS thread will wait until RunThread is executed
|
||||
*/
|
||||
void InitializeThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Starts the OS thread object to start executing guest code
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void RunThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
void CleanupAfterFork(FEXCore::Core::InternalThreadState *ExceptForThread);
|
||||
|
||||
std::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const std::string &filename);
|
||||
void UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry);
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
void GetVDSOSigReturn(VDSOSigReturn *VDSOPointers) override {
|
||||
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
|
||||
}
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
|
||||
}
|
||||
void FinalizeAOTIRCache() {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
|
||||
void IncrementIdleRefCount() override {
|
||||
++IdleWaitRefCount;
|
||||
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const { return AtomicTSOEmulationEnabled; }
|
||||
|
||||
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
|
||||
SupportsHardwareTSO = HardwareTSOSupported;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
void SetAOTIRLoader(std::function<int(const std::string&)> CacheReader) {
|
||||
IRCaptureCache.SetAOTIRLoader(CacheReader);
|
||||
}
|
||||
|
||||
void EnableExitOnHLT() override { ExitOnHLT = true; }
|
||||
void SetAOTIRWriter(std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
|
||||
IRCaptureCache.SetAOTIRWriter(CacheWriter);
|
||||
}
|
||||
|
||||
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
|
||||
void SetAOTIRRenamer(std::function<void(const std::string&)> CacheRenamer) {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void AppendThunkDefinitions(std::vector<FEXCore::IR::ThunkDefinition> const& Definitions);
|
||||
|
||||
FEXCore::Utils::PooledAllocatorMMap OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorMMap FrontendAllocator;
|
||||
|
||||
void MarkMemoryShared();
|
||||
|
||||
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
}
|
||||
else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Does some final thread initialization
|
||||
@@ -420,20 +363,17 @@ namespace FEXCore::Context {
|
||||
|
||||
// Entry Cache
|
||||
std::mutex ExitMutex;
|
||||
fextl::unique_ptr<GdbServer> DebugServer;
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
std::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool ExitOnHLT = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
fextl::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
std::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
+262
-165
@@ -1,7 +1,10 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Buffer.h"
|
||||
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/Utils/ArchHelpers/Arm64.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <csignal>
|
||||
@@ -15,10 +18,6 @@ FEXCORE_TELEMETRY_STATIC_INIT(Cas32Tear, TYPE_CAS_32BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas64Tear, TYPE_CAS_64BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas128Tear, TYPE_CAS_128BIT_TEAR);
|
||||
|
||||
static void ClearICache(void* Begin, std::size_t Length) {
|
||||
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
|
||||
}
|
||||
|
||||
static __uint128_t LoadAcquire128(uint64_t Addr) {
|
||||
__uint128_t Result{};
|
||||
uint64_t Lower;
|
||||
@@ -239,16 +238,20 @@ std::pair<uint64_t, uint64_t> DoLoad128(uint64_t Addr) {
|
||||
return {ResultLower, ResultUpper};
|
||||
}
|
||||
|
||||
static bool RunCASPAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg1, uint32_t DesiredReg2, uint32_t ExpectedReg1, uint32_t ExpectedReg2, uint32_t AddressReg) {
|
||||
static bool RunCASPAL(void *_ucontext, void *_info, uint32_t Size, uint32_t DesiredReg1, uint32_t DesiredReg2, uint32_t ExpectedReg1, uint32_t ExpectedReg2, uint32_t AddressReg) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
|
||||
//Bus_ADRALN check happens in HandleCASPAL and HandleCASPAL_ARMv8
|
||||
|
||||
if (Size == 0) {
|
||||
// 32bit
|
||||
uint64_t Addr = GPRs[AddressReg];
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
uint32_t DesiredLower = GPRs[DesiredReg1];
|
||||
uint32_t DesiredUpper = GPRs[DesiredReg2];
|
||||
uint32_t DesiredLower = mcontext->regs[DesiredReg1];
|
||||
uint32_t DesiredUpper = mcontext->regs[DesiredReg2];
|
||||
|
||||
uint32_t ExpectedLower = GPRs[ExpectedReg1];
|
||||
uint32_t ExpectedUpper = GPRs[ExpectedReg2];
|
||||
uint32_t ExpectedLower = mcontext->regs[ExpectedReg1];
|
||||
uint32_t ExpectedUpper = mcontext->regs[ExpectedReg2];
|
||||
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
// It isn't even guaranteed to work on x86
|
||||
@@ -346,16 +349,16 @@ static bool RunCASPAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
GPRs[ExpectedReg1] = FailedResult & ~0U;
|
||||
GPRs[ExpectedReg2] = FailedResult >> 32;
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
GPRs[ExpectedReg1] = FailedResult & ~0U;
|
||||
GPRs[ExpectedReg2] = FailedResult >> 32;
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -409,8 +412,8 @@ static bool RunCASPAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
GPRs[ExpectedReg1] = FailedResult & ~0U;
|
||||
GPRs[ExpectedReg2] = FailedResult >> 32;
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -419,7 +422,14 @@ static bool RunCASPAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleCASPAL(uint32_t Instr, uint64_t *GPRs) {
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
|
||||
uint32_t DesiredReg1 = Instr & 0b11111;
|
||||
@@ -428,10 +438,18 @@ bool HandleCASPAL(uint32_t Instr, uint64_t *GPRs) {
|
||||
uint32_t ExpectedReg2 = ExpectedReg1 + 1;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
return RunCASPAL(GPRs, Size, DesiredReg1, DesiredReg2, ExpectedReg1, ExpectedReg2, AddressReg);
|
||||
return RunCASPAL(_ucontext, _info, Size, DesiredReg1, DesiredReg2, ExpectedReg1, ExpectedReg2, AddressReg);
|
||||
}
|
||||
|
||||
uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uint64_t *GPRs) {
|
||||
uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return 0;
|
||||
}
|
||||
|
||||
// caspair
|
||||
// [1] ldaxp(TMP2.W(), TMP3.W(), MemOperand(MemSrc)); <-- DataReg & AddrReg
|
||||
// [2] cmp(TMP2.W(), Expected.first.W()); <-- ExpectedReg1
|
||||
@@ -446,7 +464,7 @@ uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uint64_t *
|
||||
// [11] mov(Dst.second.W(), TMP3.W());
|
||||
// [12] clrex();
|
||||
|
||||
uint32_t *PC = (uint32_t*)ProgramCounter;
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(_ucontext);
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
@@ -473,16 +491,16 @@ uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uint64_t *
|
||||
}
|
||||
else {
|
||||
uint32_t NextInstr = PC[1];
|
||||
if ((NextInstr & ArchHelpers::Arm64::CLREX_MASK) == ArchHelpers::Arm64::CLREX_INST) {
|
||||
uint64_t Addr = GPRs[AddrReg];
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::CLREX_MASK) == FEXCore::ArchHelpers::Arm64::CLREX_INST) {
|
||||
uint64_t Addr = mcontext->regs[AddrReg];
|
||||
|
||||
auto Res = DoLoad128(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (DataReg != 31) {
|
||||
GPRs[DataReg] = std::get<0>(Res);
|
||||
mcontext->regs[DataReg] = std::get<0>(Res);
|
||||
}
|
||||
if (DataReg2 != 31) {
|
||||
GPRs[DataReg2] = std::get<1>(Res);
|
||||
mcontext->regs[DataReg2] = std::get<1>(Res);
|
||||
}
|
||||
|
||||
// Skip ldaxp and clrex
|
||||
@@ -495,30 +513,37 @@ uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uint64_t *
|
||||
//Only 32-bit pairs
|
||||
for(int i = 1; i < 10; i++) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
ExpectedReg1 = GetRmReg(NextInstr);
|
||||
} else if ((NextInstr & ArchHelpers::Arm64::CCMP_MASK) == ArchHelpers::Arm64::CCMP_INST) {
|
||||
} else if ((NextInstr & FEXCore::ArchHelpers::Arm64::CCMP_MASK) == FEXCore::ArchHelpers::Arm64::CCMP_INST) {
|
||||
ExpectedReg2 = GetRmReg(NextInstr);
|
||||
} else if ((NextInstr & ArchHelpers::Arm64::STLXP_MASK) == ArchHelpers::Arm64::STLXP_INST) {
|
||||
} else if ((NextInstr & FEXCore::ArchHelpers::Arm64::STLXP_MASK) == FEXCore::ArchHelpers::Arm64::STLXP_INST) {
|
||||
DesiredReg1 = (NextInstr & 0x1F);
|
||||
DesiredReg2 = (NextInstr >> 10) & 0x1F;
|
||||
}
|
||||
}
|
||||
|
||||
//mov expected into the temp registers used by JIT
|
||||
GPRs[DataReg] = GPRs[ExpectedReg1];
|
||||
GPRs[DataReg2] = GPRs[ExpectedReg2];
|
||||
mcontext->regs[DataReg] = mcontext->regs[ExpectedReg1];
|
||||
mcontext->regs[DataReg2] = mcontext->regs[ExpectedReg2];
|
||||
|
||||
if(RunCASPAL(GPRs, Size, DesiredReg1, DesiredReg2, DataReg, DataReg2, AddrReg)) {
|
||||
if(RunCASPAL(_ucontext, _info, Size, DesiredReg1, DesiredReg2, DataReg, DataReg2, AddrReg)) {
|
||||
return 9 * sizeof(uint32_t); // skip to mov + clrex
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
static bool HandleAtomicVectorStore(uint32_t Instr, uintptr_t ProgramCounter) {
|
||||
uint32_t *PC = (uint32_t*)ProgramCounter;
|
||||
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(_ucontext);
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
@@ -546,7 +571,7 @@ static bool HandleAtomicVectorStore(uint32_t Instr, uintptr_t ProgramCounter) {
|
||||
PC[1] = STP;
|
||||
PC[2] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ClearICache(&PC[0], 16);
|
||||
FEXCore::ARMEmitter::Buffer::ClearICache(&PC[0], 16);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -1252,8 +1277,10 @@ uint64_t DoCAS64(
|
||||
}
|
||||
}
|
||||
|
||||
static bool RunCASAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg) {
|
||||
uint64_t Addr = GPRs[AddressReg];
|
||||
static bool RunCASAL(void *_ucontext, void *_info, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
// It isn't even guaranteed to work on x86
|
||||
@@ -1267,8 +1294,8 @@ static bool RunCASAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
// Only need to handle 16, 32, 64
|
||||
if (Size == 2) {
|
||||
auto Res = DoCAS16<false>(
|
||||
GPRs[DesiredReg],
|
||||
GPRs[ExpectedReg],
|
||||
mcontext->regs[DesiredReg],
|
||||
mcontext->regs[ExpectedReg],
|
||||
Addr,
|
||||
[](uint16_t, uint16_t Expected) -> uint16_t {
|
||||
// Expected is just Expected
|
||||
@@ -1282,14 +1309,14 @@ static bool RunCASAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = Res;
|
||||
mcontext->regs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
else if (Size == 4) {
|
||||
auto Res = DoCAS32<false>(
|
||||
GPRs[DesiredReg],
|
||||
GPRs[ExpectedReg],
|
||||
mcontext->regs[DesiredReg],
|
||||
mcontext->regs[ExpectedReg],
|
||||
Addr,
|
||||
[](uint32_t, uint32_t Expected) -> uint32_t {
|
||||
// Expected is just Expected
|
||||
@@ -1303,14 +1330,14 @@ static bool RunCASAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = Res;
|
||||
mcontext->regs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
else if (Size == 8) {
|
||||
auto Res = DoCAS64<false>(
|
||||
GPRs[DesiredReg],
|
||||
GPRs[ExpectedReg],
|
||||
mcontext->regs[DesiredReg],
|
||||
mcontext->regs[ExpectedReg],
|
||||
Addr,
|
||||
[](uint64_t, uint64_t Expected) -> uint64_t {
|
||||
// Expected is just Expected
|
||||
@@ -1324,7 +1351,7 @@ static bool RunCASAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = Res;
|
||||
mcontext->regs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1332,22 +1359,37 @@ static bool RunCASAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool HandleCASAL(uint64_t *GPRs, uint32_t Instr) {
|
||||
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t DesiredReg = Instr & 0b11111;
|
||||
uint32_t ExpectedReg = (Instr >> 16) & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
return RunCASAL(GPRs, Size, DesiredReg, ExpectedReg, AddressReg);
|
||||
return RunCASAL(_ucontext, _info, Size, DesiredReg, ExpectedReg, AddressReg);
|
||||
}
|
||||
|
||||
static bool HandleAtomicMemOp(uint32_t Instr, uint64_t *GPRs) {
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
uint32_t ResultReg = Instr & 0b11111;
|
||||
uint32_t SourceReg = (Instr >> 16) & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = GPRs[AddressReg];
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
uint8_t Op = (Instr >> 12) & 0xF;
|
||||
|
||||
@@ -1400,7 +1442,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t *GPRs) {
|
||||
}
|
||||
|
||||
auto Res = DoCAS16<true>(
|
||||
GPRs[SourceReg],
|
||||
mcontext->regs[SourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
@@ -1408,7 +1450,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t *GPRs) {
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1461,7 +1503,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t *GPRs) {
|
||||
}
|
||||
|
||||
auto Res = DoCAS32<true>(
|
||||
GPRs[SourceReg],
|
||||
mcontext->regs[SourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
@@ -1469,7 +1511,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t *GPRs) {
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1522,7 +1564,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t *GPRs) {
|
||||
}
|
||||
|
||||
auto Res = DoCAS64<true>(
|
||||
GPRs[SourceReg],
|
||||
mcontext->regs[SourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
@@ -1530,7 +1572,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t *GPRs) {
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1538,19 +1580,26 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t *GPRs) {
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool HandleAtomicLoad(uint32_t Instr, uint64_t *GPRs, int64_t Offset) {
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t ResultReg = Instr & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = GPRs[AddressReg] + Offset;
|
||||
uint64_t Addr = mcontext->regs[AddressReg] + Offset;
|
||||
|
||||
if (Size == 2) {
|
||||
auto Res = DoLoad16(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1558,7 +1607,7 @@ static bool HandleAtomicLoad(uint32_t Instr, uint64_t *GPRs, int64_t Offset) {
|
||||
auto Res = DoLoad32(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1566,7 +1615,7 @@ static bool HandleAtomicLoad(uint32_t Instr, uint64_t *GPRs, int64_t Offset) {
|
||||
auto Res = DoLoad64(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1574,18 +1623,25 @@ static bool HandleAtomicLoad(uint32_t Instr, uint64_t *GPRs, int64_t Offset) {
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool HandleAtomicStore(uint32_t Instr, uint64_t *GPRs, int64_t Offset) {
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = GPRs[AddressReg] + Offset;
|
||||
uint64_t Addr = mcontext->regs[AddressReg] + Offset;
|
||||
|
||||
constexpr bool DoRetry = false;
|
||||
if (Size == 2) {
|
||||
DoCAS16<DoRetry>(
|
||||
GPRs[DataReg],
|
||||
mcontext->regs[DataReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
[](uint16_t SrcVal, uint16_t) -> uint16_t {
|
||||
@@ -1600,7 +1656,7 @@ static bool HandleAtomicStore(uint32_t Instr, uint64_t *GPRs, int64_t Offset) {
|
||||
}
|
||||
else if (Size == 4) {
|
||||
DoCAS32<DoRetry>(
|
||||
GPRs[DataReg],
|
||||
mcontext->regs[DataReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
[](uint32_t SrcVal, uint32_t) -> uint32_t {
|
||||
@@ -1615,7 +1671,7 @@ static bool HandleAtomicStore(uint32_t Instr, uint64_t *GPRs, int64_t Offset) {
|
||||
}
|
||||
else if (Size == 8) {
|
||||
DoCAS64<DoRetry>(
|
||||
GPRs[DataReg],
|
||||
mcontext->regs[DataReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
[](uint64_t SrcVal, uint64_t) -> uint64_t {
|
||||
@@ -1632,8 +1688,36 @@ static bool HandleAtomicStore(uint32_t Instr, uint64_t *GPRs, int64_t Offset) {
|
||||
return false;
|
||||
}
|
||||
|
||||
static uint64_t HandleCAS_NoAtomics(uintptr_t ProgramCounter, uint64_t *GPRs)
|
||||
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return false;
|
||||
}
|
||||
uint32_t ResultReg = Instr & 0b11111;
|
||||
uint32_t ResultReg2 = (Instr >> 10) & 0x1F;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
auto Res = DoLoad128(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
mcontext->regs[ResultReg] = std::get<0>(Res);
|
||||
}
|
||||
if (ResultReg2 != 31) {
|
||||
mcontext->regs[ResultReg2] = std::get<1>(Res);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
static uint64_t HandleCAS_NoAtomics(void *_ucontext, void *_info)
|
||||
{
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
|
||||
// ARMv8.0 CAS
|
||||
// [1] ldaxrb(TMP2.W(), MemOperand(MemSrc))
|
||||
// [2] cmp (TMP2.W(), Expected.W())
|
||||
@@ -1645,7 +1729,7 @@ static uint64_t HandleCAS_NoAtomics(uintptr_t ProgramCounter, uint64_t *GPRs)
|
||||
// [8] mov (.., TMP2.W());
|
||||
// [9] clrex
|
||||
|
||||
uint32_t *PC = (uint32_t*)ProgramCounter;
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(_ucontext);
|
||||
uint32_t Instr = PC[0];
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
uint32_t AddressReg = GetRnReg(Instr);
|
||||
@@ -1654,7 +1738,7 @@ static uint64_t HandleCAS_NoAtomics(uintptr_t ProgramCounter, uint64_t *GPRs)
|
||||
uint32_t ExpectedReg = 0;
|
||||
for (size_t i = 1; i < 6; ++i) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & ArchHelpers::Arm64::STLXR_MASK) == ArchHelpers::Arm64::STLXR_INST) {
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::STLXR_MASK) == FEXCore::ArchHelpers::Arm64::STLXR_INST) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Just double check that the memory destination matches
|
||||
const uint32_t StoreAddressReg = GetRnReg(NextInstr);
|
||||
@@ -1662,23 +1746,31 @@ static uint64_t HandleCAS_NoAtomics(uintptr_t ProgramCounter, uint64_t *GPRs)
|
||||
#endif
|
||||
DesiredReg = GetRdReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
ExpectedReg = GetRmReg(NextInstr);
|
||||
}
|
||||
}
|
||||
//set up CASAL by doing mov(TMP2, Expected)
|
||||
GPRs[ResultReg] = GPRs[ExpectedReg];
|
||||
mcontext->regs[ResultReg] = mcontext->regs[ExpectedReg];
|
||||
|
||||
if(RunCASAL(GPRs, Size, DesiredReg, ResultReg, AddressReg)) {
|
||||
if(RunCASAL(_ucontext, _info, Size, DesiredReg, ResultReg, AddressReg)) {
|
||||
return 7 * sizeof(uint32_t); //jump to mov to allocated register
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_t *GPRs) {
|
||||
uint32_t *PC = (uint32_t*)ProgramCounter;
|
||||
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
// This only handles alignment problems
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(_ucontext);
|
||||
uint32_t Instr = PC[0];
|
||||
|
||||
// Atomic Add
|
||||
@@ -1720,7 +1812,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
// - The [5]mov instruction source is always the destination register from [1] ldaxr*
|
||||
uint32_t ResultReg = GetRdReg(Instr);
|
||||
uint32_t AddressReg = GetRnReg(Instr);
|
||||
uint64_t Addr = GPRs[AddressReg];
|
||||
uint64_t Addr = mcontext->regs[AddressReg];
|
||||
|
||||
size_t NumInstructionsToSkip = 0;
|
||||
|
||||
@@ -1738,13 +1830,13 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
// Scan forward at most five instructions to find our instructions
|
||||
for (size_t i = 1; i < 6; ++i) {
|
||||
uint32_t NextInstr = PC[i];
|
||||
if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::ADD_INST ||
|
||||
(NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::ADD_SHIFT_INST) {
|
||||
if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ADD_SHIFT_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_ADD;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::SUB_INST ||
|
||||
(NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::SUB_SHIFT_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::SUB_SHIFT_INST) {
|
||||
uint32_t RnReg = GetRnReg(NextInstr);
|
||||
if (RnReg == REGISTER_MASK) {
|
||||
// Zero reg means neg
|
||||
@@ -1755,35 +1847,35 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
}
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_SHIFT_INST ) {
|
||||
return HandleCAS_NoAtomics(ProgramCounter, GPRs); //ARMv8.0 CAS
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::CMP_SHIFT_INST ) {
|
||||
return HandleCAS_NoAtomics(_ucontext, _info); //ARMv8.0 CAS
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::AND_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::AND_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_AND;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::BIC_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::BIC_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_BIC;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::OR_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::OR_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_OR;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::ORN_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::ORN_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_ORN;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::EOR_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::EOR_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_EOR;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::EON_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::ALU_OP_MASK) == FEXCore::ArchHelpers::Arm64::EON_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_EON;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::STLXR_MASK) == ArchHelpers::Arm64::STLXR_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::STLXR_MASK) == FEXCore::ArchHelpers::Arm64::STLXR_INST) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Just double check that the memory destination matches
|
||||
const uint32_t StoreAddressReg = GetRnReg(NextInstr);
|
||||
@@ -1799,7 +1891,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
DataSourceReg = StoreResultReg;
|
||||
}
|
||||
}
|
||||
else if ((NextInstr & ArchHelpers::Arm64::CBNZ_MASK) == ArchHelpers::Arm64::CBNZ_INST) {
|
||||
else if ((NextInstr & FEXCore::ArchHelpers::Arm64::CBNZ_MASK) == FEXCore::ArchHelpers::Arm64::CBNZ_INST) {
|
||||
// Found the CBNZ, we want to skip to just after this instruction when done
|
||||
NumInstructionsToSkip = i + 1;
|
||||
// This is the last instruction we care about. Leave now
|
||||
@@ -1895,12 +1987,12 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
FEXCore::ToUnderlying(AtomicOp));
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Res = DoCAS16<DoRetry>(
|
||||
GPRs[DataSourceReg],
|
||||
mcontext->regs[DataSourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
@@ -1909,7 +2001,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
if (AtomicFetch && ResultReg != 31) {
|
||||
// On atomic fetch then we store the resulting value back in to the loadacquire destination register
|
||||
// We want the memory value BEFORE the ALU op
|
||||
GPRs[ResultReg] = Res;
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
}
|
||||
else if (Size == 4) {
|
||||
@@ -1949,12 +2041,12 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
FEXCore::ToUnderlying(AtomicOp));
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Res = DoCAS32<DoRetry>(
|
||||
GPRs[DataSourceReg],
|
||||
mcontext->regs[DataSourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
@@ -1963,7 +2055,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
if (AtomicFetch && ResultReg != 31) {
|
||||
// On atomic fetch then we store the resulting value back in to the loadacquire destination register
|
||||
// We want the memory value BEFORE the ALU op
|
||||
GPRs[ResultReg] = Res;
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
}
|
||||
else if (Size == 8) {
|
||||
@@ -2003,12 +2095,12 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
FEXCore::ToUnderlying(AtomicOp));
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Res = DoCAS64<DoRetry>(
|
||||
GPRs[DataSourceReg],
|
||||
mcontext->regs[DataSourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
@@ -2016,7 +2108,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
if (AtomicFetch && ResultReg != 31) {
|
||||
// On atomic fetch then we store the resulting value back in to the loadacquire destination register
|
||||
// We want the memory value BEFORE the ALU op
|
||||
GPRs[ResultReg] = Res;
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2024,17 +2116,15 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
return NumInstructionsToSkip * 4;
|
||||
}
|
||||
|
||||
[[nodiscard]] std::pair<bool, int32_t> HandleUnalignedAccess(bool ParanoidTSO, uintptr_t ProgramCounter, uint64_t *GPRs) {
|
||||
bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext) {
|
||||
#ifdef _M_ARM_64
|
||||
constexpr bool is_arm64 = true;
|
||||
#else
|
||||
constexpr bool is_arm64 = false;
|
||||
#endif
|
||||
|
||||
constexpr auto NotHandled = std::make_pair(false, 0);
|
||||
|
||||
if constexpr (is_arm64) {
|
||||
uint32_t *PC = (uint32_t*)ProgramCounter;
|
||||
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(ucontext);
|
||||
uint32_t Instr = PC[0];
|
||||
|
||||
// 1 = 16bit
|
||||
@@ -2043,16 +2133,17 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr, 0)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -2063,20 +2154,20 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ( (Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, 0)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr, 0)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -2087,22 +2178,22 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == LDAPUR_INST) { // LDAPUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, Offset)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicLoad(ucontext, info, Instr, Offset)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAPUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAPUR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -2114,22 +2205,22 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDUR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, Offset)) {
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicStore(ucontext, info, Instr, Offset)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -2141,82 +2232,88 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
PC[-1] = DMB;
|
||||
PC[0] = STUR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
|
||||
}
|
||||
}
|
||||
else if ((Instr & ArchHelpers::Arm64::LDAXP_MASK) == ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXP_MASK) == FEXCore::ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
//Should be compare and swap pair only. LDAXP not used elsewhere
|
||||
uint64_t BytesToSkip = ArchHelpers::Arm64::HandleCASPAL_ARMv8(Instr, ProgramCounter, GPRs);
|
||||
uint64_t BytesToSkip = FEXCore::ArchHelpers::Arm64::HandleCASPAL_ARMv8(ucontext, info, Instr);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, BytesToSkip);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + BytesToSkip);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
if (ArchHelpers::Arm64::HandleAtomicVectorStore(Instr, ProgramCounter)) {
|
||||
return std::make_pair(true, 0);
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicVectorStore(ucontext, info, Instr)) {
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXP: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if ((Instr & ArchHelpers::Arm64::STLXP_MASK) == ArchHelpers::Arm64::STLXP_INST) { // STLXP
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::STLXP_MASK) == FEXCore::ArchHelpers::Arm64::STLXP_INST) { // STLXP
|
||||
//Should not trigger - middle of an LDAXP/STAXP pair.
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLXP: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
else if ((Instr & ArchHelpers::Arm64::CASPAL_MASK) == ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (ArchHelpers::Arm64::HandleCASPAL(Instr, GPRs)) {
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & ArchHelpers::Arm64::CASAL_MASK) == ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (ArchHelpers::Arm64::HandleCASAL(GPRs, Instr)) {
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASAL: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & ArchHelpers::Arm64::ATOMIC_MEM_MASK) == ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (ArchHelpers::Arm64::HandleAtomicMemOp(Instr, GPRs)) {
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}: PC: 0x{:x} Instruction: 0x{:08x}\n", Op, ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}: PC: {} Instruction: 0x{:08x}\n", Op, fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if ((Instr & ArchHelpers::Arm64::LDAXR_MASK) == ArchHelpers::Arm64::LDAXR_INST) { // LDAXR*
|
||||
uint64_t BytesToSkip = ArchHelpers::Arm64::HandleAtomicLoadstoreExclusive(ProgramCounter, GPRs);
|
||||
else if ((Instr & FEXCore::ArchHelpers::Arm64::LDAXR_MASK) == FEXCore::ArchHelpers::Arm64::LDAXR_INST) { // LDAXR*
|
||||
uint64_t BytesToSkip = FEXCore::ArchHelpers::Arm64::HandleAtomicLoadstoreExclusive(ucontext, info);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, BytesToSkip);
|
||||
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + BytesToSkip);
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXR: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXR: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS: PC: {} Instruction: 0x{:08x}\n", fmt::ptr(PC), PC[0]);
|
||||
return false;
|
||||
}
|
||||
|
||||
FEXCore::ARMEmitter::Buffer::ClearICache(&PC[-1], 16);
|
||||
return true;
|
||||
}
|
||||
return NotHandled;
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
+10
-22
@@ -1,9 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t CASPAL_MASK = 0xBF'E0'FC'00;
|
||||
@@ -27,9 +24,6 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
|
||||
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
|
||||
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
|
||||
constexpr uint32_t LDAR_INST = 0x08'DF'FC'00;
|
||||
constexpr uint32_t LDAPR_INST = 0x38'BF'C0'00;
|
||||
constexpr uint32_t STLR_INST = 0x08'9F'FC'00;
|
||||
|
||||
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
|
||||
@@ -102,20 +96,14 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
return (Instr >> RM_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief On ARM64 handles an unaligned memory access that the JIT has done.
|
||||
*
|
||||
* This is an OS agnostic handler where the frontend must provide FEXCore with the information necessary to know if this is safe.
|
||||
* This does not check if the PC is within a JIT code buffer, the frontend must provide that safety with `CPUBackend::IsAddressInCodeBuffer`.
|
||||
*
|
||||
* @param ParanoidTSO If the unaligned fault needs to handled directly or can be backpatched.
|
||||
* @param ProgramCounter The location in memory for the instruction that did the access
|
||||
* @param GPRs The array of GPRs from the signal context. This will be modified and the host context needs to be updated on signal return.
|
||||
*
|
||||
* @return A pair where the first element is if the unaligned access has been handle and the second element is how many bytes to modify the host PC
|
||||
* by. FEXCore will return a positive or negative offset depending on internal handling.
|
||||
*/
|
||||
[[nodiscard]]
|
||||
FEX_DEFAULT_VISIBILITY
|
||||
std::pair<bool, int32_t> HandleUnalignedAccess(bool ParanoidTSO, uintptr_t ProgramCounter, uint64_t *GPRs);
|
||||
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr, int64_t Offset);
|
||||
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr);
|
||||
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info);
|
||||
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr);
|
||||
uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr);
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr);
|
||||
[[nodiscard]] bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext);
|
||||
}
|
||||
+62
-226
@@ -1,12 +1,10 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
@@ -20,162 +18,18 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
namespace x64 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
}
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
|
||||
{FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 20> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
}
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
: Emitter(size ? (uint8_t*)FEXCore::Allocator::VirtualAlloc(size, true) : nullptr, size)
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
: Emitter(size ? (uint8_t*)FEXCore::Allocator::mmap(nullptr, size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0) : nullptr, size)
|
||||
, EmitterCTX {ctx} {
|
||||
CPU.SetUp();
|
||||
|
||||
// Number of register available is dependent on what operating mode the proccess is in.
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
StaticRegisters = x64::SRA;
|
||||
GeneralRegisters = x64::RA;
|
||||
GeneralPairRegisters = x64::RAPair;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
}
|
||||
else {
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
GeneralPairRegisters = x32::RAPair;
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
}
|
||||
}
|
||||
|
||||
Arm64Emitter::~Arm64Emitter() {
|
||||
auto BufferSize = GetBufferSize();
|
||||
if (BufferSize) {
|
||||
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
|
||||
FEXCore::Allocator::munmap(GetBufferBase(), BufferSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -257,11 +111,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We need to save pairs of registers
|
||||
// We save r19-r30
|
||||
const fextl::vector<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>> CalleeSaved = {{
|
||||
#ifdef _WIN32
|
||||
// Platform register, Just save it twice to make logic easy.
|
||||
{ARMEmitter::XReg::x18, ARMEmitter::XReg::x18},
|
||||
#endif
|
||||
const std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
|
||||
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
|
||||
{ARMEmitter::XReg::x21, ARMEmitter::XReg::x22},
|
||||
{ARMEmitter::XReg::x23, ARMEmitter::XReg::x24},
|
||||
@@ -325,17 +175,13 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
32);
|
||||
}
|
||||
|
||||
const fextl::vector<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>> CalleeSaved = {{
|
||||
const std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
|
||||
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
|
||||
{ARMEmitter::XReg::x27, ARMEmitter::XReg::x28},
|
||||
{ARMEmitter::XReg::x25, ARMEmitter::XReg::x26},
|
||||
{ARMEmitter::XReg::x23, ARMEmitter::XReg::x24},
|
||||
{ARMEmitter::XReg::x21, ARMEmitter::XReg::x22},
|
||||
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
|
||||
#ifdef _WIN32
|
||||
// Platform register.
|
||||
{ARMEmitter::XReg::x18, ARMEmitter::XReg::zr},
|
||||
#endif
|
||||
}};
|
||||
|
||||
for (auto &RegPair : CalleeSaved) {
|
||||
@@ -343,14 +189,14 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -365,31 +211,33 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B, STATE.R(), TMP4.R());
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Reg, PRED_TMP_32B, STATE.R(), TMP4.R());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GPRSpillMask && FPRSpillMask == ~0U) {
|
||||
// Optimize the common case where we can spill four registers per instruction
|
||||
auto TmpReg = SRA64[__builtin_ffs(GPRSpillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
@@ -421,33 +269,33 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B.Zeroing(), STATE.R(), TMP4.R());
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg, PRED_TMP_32B, STATE.R(), TMP4.R());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
|
||||
auto TmpReg = SRA64[__builtin_ffs(GPRFillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
@@ -464,9 +312,9 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
for (size_t i = 0; i < SRA64.size(); i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -482,10 +330,10 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto GPRSize = 1 * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const auto FPRSize = RAFPR.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
@@ -494,31 +342,25 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
st4b(Reg1, Reg2, Reg3, Reg4, PRED_TMP_32B, TmpReg, 0);
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
static_assert(RAFPR.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
}
|
||||
|
||||
@@ -526,30 +368,24 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
ld4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
ld4b(Reg1, Reg2, Reg3, Reg4, PRED_TMP_32B, ARMEmitter::Reg::rsp);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
for (size_t i = 0; i < RAFPR.size(); i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
|
||||
@@ -26,9 +26,39 @@
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// All but x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9, FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r18, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r15, FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r13, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
// All are callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23, FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r19
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RA64Pair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19, FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27, FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
|
||||
/*FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1, FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,*/FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, // FEXCore::ARMEmitter::VReg::v0 ~ FEXCore::ARMEmitter::VReg::v3 are used as temps
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9, FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13, FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15
|
||||
};
|
||||
|
||||
// Contains the address to the currently available CPU state
|
||||
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
|
||||
|
||||
@@ -55,45 +85,21 @@ constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PRe
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size);
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
~Arm64Emitter();
|
||||
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters{};
|
||||
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters{};
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
* @{ */
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
constexpr static uint64_t FPRBase = (1ULL << 32);
|
||||
constexpr static uint64_t GPRPairBase = (2ULL << 32);
|
||||
|
||||
/** @} */
|
||||
|
||||
constexpr static uint8_t RA_32 = 0;
|
||||
constexpr static uint8_t RA_64 = 1;
|
||||
constexpr static uint8_t RA_FPR = 2;
|
||||
|
||||
void LoadConstant(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
void SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
|
||||
// Register 0-18 + 29 + 30 are caller saved
|
||||
static constexpr uint32_t CALLER_GPR_MASK = 0b0110'0000'0000'0111'1111'1111'1111'1111U;
|
||||
static constexpr uint32_t CALLER_GPR_MASK = 0b0011'1111'1111'1111'1111;
|
||||
|
||||
// This isn't technically true because the lower 64-bits of v8..v15 are callee saved
|
||||
// We can't guarantee only the lower 64bits are used so flush everything
|
||||
|
||||
+1
-6
@@ -1,6 +1,6 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/ArchHelpers/Arm64.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
@@ -22,11 +22,6 @@ bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
|
||||
}
|
||||
|
||||
std::pair<bool, int32_t> HandleUnalignedAccess(bool ParanoidTSO, uintptr_t ProgramCounter, uint64_t *GPRs) {
|
||||
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
}
|
||||
+18
-173
@@ -10,16 +10,6 @@
|
||||
* FEX-Emu ALU operations usually have a 32-bit or 64-bit operating size encoded in the IR operation,
|
||||
* This allows FEX to use a single helper function which decodes to both handlers.
|
||||
*/
|
||||
private:
|
||||
static bool IsADRRange(int64_t Imm) {
|
||||
return Imm >= -1048576 && Imm <= 1048575;
|
||||
}
|
||||
static bool IsADRPRange(int64_t Imm) {
|
||||
return Imm >= -4294967296 && Imm <= 4294963200;
|
||||
}
|
||||
static bool IsADRPAligned(int64_t Imm) {
|
||||
return (Imm & 0xFFF) == 0;
|
||||
}
|
||||
public:
|
||||
// PC relative
|
||||
void adr(FEXCore::ARMEmitter::Register rd, uint32_t Imm) {
|
||||
@@ -29,7 +19,7 @@ public:
|
||||
|
||||
void adr(FEXCore::ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575, "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
@@ -56,7 +46,7 @@ public:
|
||||
|
||||
void adrp(FEXCore::ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
LOGMAN_THROW_A_FMT(Imm >= -4294967296 && Imm <= 4294963200 && (Imm & 0xFFF) == 0, "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
@@ -76,49 +66,6 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void LongAddressGen(FEXCore::ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
|
||||
if (IsADRRange(Imm)) {
|
||||
// If the range is in ADR range then we can just use ADR.
|
||||
adr(rd, Label);
|
||||
}
|
||||
else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL)
|
||||
- (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
|
||||
// If the range is in the ADRP range then we can use ADRP.
|
||||
bool NeedsOffset = !IsADRPAligned(reinterpret_cast<uint64_t>(Label->Location));
|
||||
uint64_t AlignedOffset = reinterpret_cast<uint64_t>(Label->Location) & 0xFFFULL;
|
||||
|
||||
// First emit ADRP
|
||||
adrp(rd, ADRPImm >> 12);
|
||||
|
||||
if (NeedsOffset) {
|
||||
// Now even an add
|
||||
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
|
||||
}
|
||||
}
|
||||
else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset too large");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
void LongAddressGen(FEXCore::ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN });
|
||||
// Emit a register index and a nop. These will be backpatched.
|
||||
dc32(rd.Idx());
|
||||
nop();
|
||||
}
|
||||
|
||||
void LongAddressGen(FEXCore::ARMEmitter::Register rd, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
LongAddressGen(rd, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
LongAddressGen(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
// Add/subtract immediate
|
||||
void add(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
constexpr uint32_t Op = 0b0001'0001'0 << 23;
|
||||
@@ -145,27 +92,6 @@ public:
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0000, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0001, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0010, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0011, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
// Logical immediate
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
@@ -278,8 +204,8 @@ public:
|
||||
void sxth(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
sbfm(s, rd, rn, 0, 15);
|
||||
}
|
||||
void sxtw(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn) {
|
||||
sbfm(ARMEmitter::Size::i64Bit, rd, rn.X(), 0, 31);
|
||||
void sxtw(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn) {
|
||||
sbfm(ARMEmitter::Size::i64Bit, rd, rn, 0, 31);
|
||||
}
|
||||
void sbfx(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
LOGMAN_THROW_A_FMT(width > 0, "sbfx needs width > 0");
|
||||
@@ -308,12 +234,12 @@ public:
|
||||
|
||||
void lsl(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsl a region larger than the register");
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to asr a region larger than the register");
|
||||
ubfm(s, rd, rn, (RegSize - shift) % RegSize, RegSize - shift - 1);
|
||||
}
|
||||
void lsr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsr a region larger than the register");
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to asr a region larger than the register");
|
||||
ubfm(s, rd, rn, shift, RegSize - 1);
|
||||
}
|
||||
void ubfx(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
@@ -324,8 +250,8 @@ public:
|
||||
|
||||
void bfi(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(width > 0, "bfi needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to bfi a region larger than the register");
|
||||
LOGMAN_THROW_A_FMT(width > 0, "sbfx needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to sbfx a region larger than the register");
|
||||
bfm(s, rd, rn, (RegSize - lsb) & (RegSize - 1), width - 1);
|
||||
}
|
||||
|
||||
@@ -337,6 +263,7 @@ public:
|
||||
}
|
||||
|
||||
void ror(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm < RegSizeInBits(s), "Tried to extr a region larger than the register");
|
||||
extr(s, rd, rn, rn, Imm);
|
||||
}
|
||||
|
||||
@@ -402,26 +329,6 @@ public:
|
||||
(0b0101'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void subp(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
@@ -508,24 +415,7 @@ public:
|
||||
(s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void ctz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'10U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cnt(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'11U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void abs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0010'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
|
||||
// TODO: PAUTH
|
||||
|
||||
@@ -694,30 +584,10 @@ public:
|
||||
constexpr uint32_t Op = 0b0111'1010'000U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
|
||||
}
|
||||
|
||||
// Rotate right into flags
|
||||
void rmif(XRegister rn, uint32_t shift, uint32_t mask) {
|
||||
LOGMAN_THROW_AA_FMT(shift <= 63, "Shift must be within 0-63. Shift: {}", shift);
|
||||
LOGMAN_THROW_AA_FMT(mask <= 15, "Mask must be within 0-15. Mask: {}", mask);
|
||||
|
||||
uint32_t Op = 0b1011'1010'0000'0000'0000'0100'0000'0000;
|
||||
Op |= rn.Idx() << 5;
|
||||
Op |= shift << 15;
|
||||
Op |= mask;
|
||||
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
// TODO
|
||||
// Evaluate into flags
|
||||
void setf8(WRegister rn) {
|
||||
constexpr uint32_t Op = 0b0011'1010'0000'0000'0000'1000'0000'1101;
|
||||
EvaluateIntoFlags(Op, 0, rn);
|
||||
}
|
||||
void setf16(WRegister rn) {
|
||||
constexpr uint32_t Op = 0b0011'1010'0000'0000'0000'1000'0000'1101;
|
||||
EvaluateIntoFlags(Op, 1, rn);
|
||||
}
|
||||
|
||||
// TODO
|
||||
// Conditional compare - register
|
||||
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0011'1010'010 << 21;
|
||||
@@ -768,28 +638,28 @@ public:
|
||||
DataProcessing_3Source(Op, 0, s, rd, rn, rm, ra);
|
||||
}
|
||||
void mul(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
madd(s, rd, rn, rm, XReg::zr);
|
||||
madd(s, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void msub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'000U << 21;
|
||||
DataProcessing_3Source(Op, 1, s, rd, rn, rm, ra);
|
||||
}
|
||||
void mneg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
msub(s, rd, rn, rm, XReg::zr);
|
||||
msub(s, rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void smaddl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'001U << 21;
|
||||
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
|
||||
}
|
||||
void smull(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
smaddl(rd, rn, rm, XReg::zr);
|
||||
smaddl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void smsubl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'001U << 21;
|
||||
DataProcessing_3Source(Op, 1, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
|
||||
}
|
||||
void smnegl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
smsubl(rd, rn, rm, XReg::zr);
|
||||
smsubl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void smulh(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = 0b001'1011'010U << 21;
|
||||
@@ -800,14 +670,14 @@ public:
|
||||
DataProcessing_3Source(Op, 0, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
|
||||
}
|
||||
void umull(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
umaddl(rd, rn, rm, XReg::zr);
|
||||
umaddl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void umsubl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::XRegister ra) {
|
||||
constexpr uint32_t Op = 0b001'1011'101U << 21;
|
||||
DataProcessing_3Source(Op, 1, FEXCore::ARMEmitter::Size::i64Bit, rd, rn, rm, ra);
|
||||
}
|
||||
void umnegl(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm) {
|
||||
umsubl(rd, rn, rm, XReg::zr);
|
||||
umsubl(rd, rn, rm, FEXCore::ARMEmitter::Reg::zr);
|
||||
}
|
||||
void umulh(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = 0b001'1011'110U << 21;
|
||||
@@ -877,21 +747,6 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void MinMaxImmediate(uint32_t opc, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = 0b1'0001'11U << 22;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= opc << 18;
|
||||
Instr |= (Imm & 0xFF) << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Move Wide
|
||||
void DataProcessing_MoveWide(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
@@ -972,9 +827,6 @@ private:
|
||||
// AddSub - shifted register
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_AA_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
if (s == FEXCore::ARMEmitter::Size::i32Bit) {
|
||||
LOGMAN_THROW_AA_FMT(amt < 32, "Shift amount for 32-bit must be below 32");
|
||||
}
|
||||
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
@@ -1057,11 +909,4 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void EvaluateIntoFlags(uint32_t op, uint32_t size, WRegister rn) {
|
||||
uint32_t Instr = op;
|
||||
Instr |= size << 14;
|
||||
Instr |= rn.Idx() << 5;
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
|
||||
+639
-1868
File diff suppressed because it is too large.
Load diff
@@ -87,11 +87,6 @@ namespace FEXCore::ARMEmitter {
|
||||
return Size;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
size_t GetCursorOffsetFromAddress(const T* Address) const {
|
||||
return static_cast<size_t>(reinterpret_cast<const uint8_t*>(Address) - BufferBase);
|
||||
}
|
||||
|
||||
protected:
|
||||
|
||||
void ResetBuffer() {
|
||||
|
||||
@@ -6,13 +6,13 @@
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <type_traits>
|
||||
#include <vector>
|
||||
|
||||
/*
|
||||
* Welcome to FEX-Emu's custom AArch64 emitter.
|
||||
@@ -62,9 +62,20 @@ namespace FEXCore::ARMEmitter {
|
||||
};
|
||||
|
||||
// This allows us to get the `Size` enum in bits.
|
||||
[[nodiscard]]
|
||||
constexpr size_t RegSizeInBits(Size size) {
|
||||
return size_t{32} << FEXCore::ToUnderlying(size);
|
||||
template<Size size>
|
||||
constexpr size_t RegSizeInBits() {
|
||||
constexpr size_t RegSize[] = {
|
||||
32, 64, 128,
|
||||
};
|
||||
return RegSize[FEXCore::ToUnderlying(size)];
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static inline size_t RegSizeInBits(Size size) {
|
||||
constexpr size_t RegSize[] = {
|
||||
32, 64, 128,
|
||||
};
|
||||
return RegSize[FEXCore::ToUnderlying(size)];
|
||||
}
|
||||
|
||||
/* This `SubRegSize` enum is used for most ASIMD operations.
|
||||
@@ -79,9 +90,14 @@ namespace FEXCore::ARMEmitter {
|
||||
};
|
||||
|
||||
// This allows us to get the `SubRegSize` in bits.
|
||||
[[nodiscard]]
|
||||
constexpr size_t SubRegSizeInBits(SubRegSize size) {
|
||||
return size_t{8} << FEXCore::ToUnderlying(size);
|
||||
template<SubRegSize size>
|
||||
constexpr size_t SubRegSizeInBits() {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static inline size_t SubRegSizeInBits(SubRegSize size) {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
/* This `ScalarRegSize` enum is used for most scalar float
|
||||
@@ -101,9 +117,14 @@ namespace FEXCore::ARMEmitter {
|
||||
};
|
||||
|
||||
// This allows us to get the `ScalarRegSize` in bits.
|
||||
[[nodiscard]]
|
||||
constexpr size_t ScalarRegSizeInBits(ScalarRegSize size) {
|
||||
return size_t{8} << FEXCore::ToUnderlying(size);
|
||||
template<ScalarRegSize size>
|
||||
constexpr size_t ScalarRegSizeInBits() {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static inline size_t ScalarRegSizeInBits(ScalarRegSize size) {
|
||||
return (1 << FEXCore::ToUnderlying(size)) * 8;
|
||||
}
|
||||
|
||||
/* This `VectorRegSizePair` union allows us to have an overlapping type
|
||||
@@ -119,12 +140,12 @@ namespace FEXCore::ARMEmitter {
|
||||
};
|
||||
|
||||
// This allows us to create a `VectorRegSizePair` union.
|
||||
[[nodiscard]]
|
||||
constexpr VectorRegSizePair ToVectorSizePair(SubRegSize size) {
|
||||
[[maybe_unused]]
|
||||
static inline VectorRegSizePair ToVectorSizePair(SubRegSize size) {
|
||||
return VectorRegSizePair {.Vector = size};
|
||||
}
|
||||
[[nodiscard]]
|
||||
constexpr VectorRegSizePair ToVectorSizePair(ScalarRegSize size) {
|
||||
[[maybe_unused]]
|
||||
static inline VectorRegSizePair ToVectorSizePair(ScalarRegSize size) {
|
||||
return VectorRegSizePair {.Scalar = size};
|
||||
}
|
||||
|
||||
@@ -498,12 +519,11 @@ namespace FEXCore::ARMEmitter {
|
||||
BC,
|
||||
TEST_BRANCH,
|
||||
RELATIVE_LOAD,
|
||||
LONG_ADDRESS_GEN,
|
||||
};
|
||||
uint8_t *Location{};
|
||||
InstType Type;
|
||||
};
|
||||
fextl::vector<Instructions> Insts{};
|
||||
std::vector<Instructions> Insts{};
|
||||
};
|
||||
|
||||
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
@@ -515,44 +535,6 @@ namespace FEXCore::ARMEmitter {
|
||||
ForwardLabel Forward;
|
||||
};
|
||||
|
||||
// Some FCMA ASIMD instructions support a rotation argument.
|
||||
enum class Rotation : uint32_t {
|
||||
ROTATE_0 = 0b00,
|
||||
ROTATE_90 = 0b01,
|
||||
ROTATE_180 = 0b10,
|
||||
ROTATE_270 = 0b11,
|
||||
};
|
||||
|
||||
// Concept for contraining some instructions to accept only an XRegister or WRegister.
|
||||
// Particularly for operations that differ encodings depending on which one is used.
|
||||
template <typename T>
|
||||
concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegister>;
|
||||
|
||||
// Whether or not a given set of vector registers are sequential
|
||||
// in increasing order as far as the register file is concerned (modulo its size)
|
||||
//
|
||||
// For example, a set of registers like:
|
||||
//
|
||||
// v1, v2, v3 and
|
||||
// v31, v0, v1
|
||||
//
|
||||
// would both be considered sequential sequences, and some instructions in particular
|
||||
// limit register lists to these kind of sequences.
|
||||
//
|
||||
template <typename T, typename... Args>
|
||||
constexpr bool AreVectorsSequential(T first, const Args&... args) {
|
||||
// Ensure we always have a pair of registers to compare against.
|
||||
static_assert(sizeof...(args) >= 1, "Number of arguments must be greater than 1");
|
||||
|
||||
const auto fn = [](auto& lhs, const auto& rhs) {
|
||||
const auto result = ((lhs.Idx() + 1) % 32) == rhs.Idx();
|
||||
lhs = rhs;
|
||||
return result;
|
||||
};
|
||||
|
||||
return (fn(first, args) && ...);
|
||||
}
|
||||
|
||||
// This is an emitter that is designed around the smallest code bloat as possible.
|
||||
// Eschewing most developer convenience in order to keep code as small as possible.
|
||||
|
||||
@@ -589,7 +571,7 @@ namespace FEXCore::ARMEmitter {
|
||||
case ForwardLabel::Instructions::InstType::ADR: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575, "Unscaled offset too large");
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
@@ -601,7 +583,7 @@ namespace FEXCore::ARMEmitter {
|
||||
case ForwardLabel::Instructions::InstType::ADRP: {
|
||||
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
LOGMAN_THROW_A_FMT(Imm >= -4294967296 && Imm <= 4294963200 && (Imm & 0xFFF) == 0, "Unscaled offset too large");
|
||||
Imm >>= 12;
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
@@ -652,47 +634,6 @@ namespace FEXCore::ARMEmitter {
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN: {
|
||||
uint32_t *Instructions = reinterpret_cast<uint32_t*>(Inst.Location);
|
||||
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
auto OriginalOffset = GetCursorOffset();
|
||||
|
||||
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
|
||||
SetCursorOffset(InstOffset);
|
||||
|
||||
// We encoded the destination register in to the first instruction space.
|
||||
// Read it back.
|
||||
ARMEmitter::Register DestReg(Instructions[0]);
|
||||
|
||||
if (IsADRRange(ImmInstTwo)) {
|
||||
// If within ADR range from the second instruction, then we can emit NOP+ADR
|
||||
nop();
|
||||
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
|
||||
}
|
||||
else if (IsADRPRange(ImmInstOne)) {
|
||||
|
||||
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
|
||||
// First check if we are in non-offset range for second instruction.
|
||||
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
|
||||
// We can emit nop + adrp
|
||||
nop();
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
|
||||
}
|
||||
else {
|
||||
// Not aligned, need adrp + add
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
|
||||
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
|
||||
}
|
||||
}
|
||||
else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
SetCursorOffset(OriginalOffset);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
|
||||
}
|
||||
}
|
||||
|
||||
+565
-439
File diff suppressed because it is too large.
Load diff
+249
-76
@@ -1,8 +1,5 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
#include <compare>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::ARMEmitter {
|
||||
@@ -18,12 +15,13 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr explicit Register(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const Register&, const Register&) = default;
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator WRegister() const;
|
||||
operator XRegister() const;
|
||||
|
||||
WRegister W() const;
|
||||
XRegister X() const;
|
||||
|
||||
@@ -43,7 +41,9 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr explicit WRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const WRegister&, const WRegister&) = default;
|
||||
bool operator==(const WRegister &rhs) {
|
||||
return Idx() == rhs.Idx();
|
||||
}
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
@@ -53,7 +53,10 @@ namespace FEXCore::ARMEmitter {
|
||||
return Register(Index);
|
||||
}
|
||||
|
||||
operator XRegister() const;
|
||||
|
||||
XRegister X() const;
|
||||
|
||||
Register R() const;
|
||||
|
||||
private:
|
||||
@@ -72,7 +75,9 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr explicit XRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const XRegister&, const XRegister&) = default;
|
||||
bool operator==(const XRegister &rhs) {
|
||||
return Idx() == rhs.Idx();
|
||||
}
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
@@ -82,7 +87,10 @@ namespace FEXCore::ARMEmitter {
|
||||
return Register(Index);
|
||||
}
|
||||
|
||||
operator WRegister() const;
|
||||
|
||||
WRegister W() const;
|
||||
|
||||
Register R() const;
|
||||
|
||||
private:
|
||||
@@ -93,29 +101,45 @@ namespace FEXCore::ARMEmitter {
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
|
||||
inline WRegister Register::W() const {
|
||||
return WRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline XRegister Register::X() const {
|
||||
return XRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline Register::operator WRegister () const {
|
||||
return WRegister(Index);
|
||||
}
|
||||
|
||||
inline Register::operator XRegister () const {
|
||||
return XRegister(Index);
|
||||
}
|
||||
|
||||
inline XRegister WRegister::X() const {
|
||||
return XRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline Register WRegister::R() const {
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline WRegister::operator XRegister () const {
|
||||
return XRegister(Index);
|
||||
}
|
||||
|
||||
inline WRegister XRegister::W() const {
|
||||
return WRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline Register XRegister::R() const {
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline XRegister::operator WRegister () const {
|
||||
return WRegister(Index);
|
||||
}
|
||||
|
||||
// Namespace containing all unsized GPR register objects.
|
||||
namespace Reg {
|
||||
constexpr static Register r0(0);
|
||||
@@ -267,15 +291,20 @@ namespace FEXCore::ARMEmitter {
|
||||
class VRegister {
|
||||
public:
|
||||
VRegister() = delete;
|
||||
constexpr explicit VRegister(uint32_t Idx)
|
||||
constexpr VRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const VRegister&, const VRegister&) = default;
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator BRegister() const;
|
||||
operator HRegister() const;
|
||||
operator SRegister() const;
|
||||
operator DRegister() const;
|
||||
operator QRegister() const;
|
||||
operator ZRegister() const;
|
||||
|
||||
BRegister B() const;
|
||||
HRegister H() const;
|
||||
SRegister S() const;
|
||||
@@ -299,15 +328,16 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr explicit BRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const BRegister&, const BRegister&) = default;
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator VRegister () const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
operator VRegister() const;
|
||||
operator HRegister() const;
|
||||
operator SRegister() const;
|
||||
operator DRegister() const;
|
||||
operator QRegister() const;
|
||||
operator ZRegister() const;
|
||||
|
||||
BRegister V() const;
|
||||
HRegister H() const;
|
||||
@@ -332,15 +362,16 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr explicit HRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const HRegister&, const HRegister&) = default;
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator VRegister() const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
operator VRegister() const;
|
||||
operator BRegister() const;
|
||||
operator SRegister() const;
|
||||
operator DRegister() const;
|
||||
operator QRegister() const;
|
||||
operator ZRegister() const;
|
||||
|
||||
HRegister V() const;
|
||||
BRegister B() const;
|
||||
@@ -365,15 +396,16 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr explicit SRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const SRegister&, const SRegister&) = default;
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator VRegister() const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
operator VRegister() const;
|
||||
operator BRegister() const;
|
||||
operator HRegister() const;
|
||||
operator DRegister() const;
|
||||
operator QRegister() const;
|
||||
operator ZRegister() const;
|
||||
|
||||
SRegister V() const;
|
||||
BRegister B() const;
|
||||
@@ -399,15 +431,16 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr explicit DRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const DRegister&, const DRegister&) = default;
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator VRegister() const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
operator VRegister() const;
|
||||
operator BRegister() const;
|
||||
operator HRegister() const;
|
||||
operator SRegister() const;
|
||||
operator QRegister() const;
|
||||
operator ZRegister() const;
|
||||
|
||||
DRegister V() const;
|
||||
BRegister B() const;
|
||||
@@ -433,15 +466,16 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr explicit QRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const QRegister&, const QRegister&) = default;
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator VRegister () const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
operator VRegister() const;
|
||||
operator BRegister() const;
|
||||
operator HRegister() const;
|
||||
operator SRegister() const;
|
||||
operator DRegister() const;
|
||||
operator ZRegister() const;
|
||||
|
||||
QRegister V() const;
|
||||
BRegister B() const;
|
||||
@@ -466,8 +500,6 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr explicit ZRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const ZRegister&, const ZRegister&) = default;
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
@@ -488,22 +520,41 @@ namespace FEXCore::ARMEmitter {
|
||||
|
||||
// VRegister
|
||||
inline BRegister VRegister::B() const {
|
||||
return BRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline HRegister VRegister::H() const {
|
||||
return HRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline SRegister VRegister::S() const {
|
||||
return SRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline DRegister VRegister::D() const {
|
||||
return DRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline QRegister VRegister::Q() const {
|
||||
return QRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline ZRegister VRegister::Z() const {
|
||||
return ZRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline VRegister::operator BRegister () const {
|
||||
return BRegister(Index);
|
||||
}
|
||||
inline VRegister::operator HRegister () const {
|
||||
return HRegister(Index);
|
||||
}
|
||||
inline VRegister::operator SRegister () const {
|
||||
return SRegister(Index);
|
||||
}
|
||||
inline VRegister::operator DRegister () const {
|
||||
return DRegister(Index);
|
||||
}
|
||||
inline VRegister::operator QRegister () const {
|
||||
return QRegister(Index);
|
||||
}
|
||||
inline VRegister::operator ZRegister () const {
|
||||
return ZRegister(Index);
|
||||
}
|
||||
|
||||
// BRegister
|
||||
@@ -511,19 +562,38 @@ namespace FEXCore::ARMEmitter {
|
||||
return *this;
|
||||
}
|
||||
inline HRegister BRegister::H() const {
|
||||
return HRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline SRegister BRegister::S() const {
|
||||
return SRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline DRegister BRegister::D() const {
|
||||
return DRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline QRegister BRegister::Q() const {
|
||||
return QRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline ZRegister BRegister::Z() const {
|
||||
return ZRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline BRegister::operator VRegister () const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
inline BRegister::operator HRegister () const {
|
||||
return HRegister(Index);
|
||||
}
|
||||
inline BRegister::operator SRegister () const {
|
||||
return SRegister(Index);
|
||||
}
|
||||
inline BRegister::operator DRegister () const {
|
||||
return DRegister(Index);
|
||||
}
|
||||
inline BRegister::operator QRegister () const {
|
||||
return QRegister(Index);
|
||||
}
|
||||
inline BRegister::operator ZRegister () const {
|
||||
return ZRegister(Index);
|
||||
}
|
||||
|
||||
// HRegister
|
||||
@@ -531,19 +601,38 @@ namespace FEXCore::ARMEmitter {
|
||||
return *this;
|
||||
}
|
||||
inline BRegister HRegister::B() const {
|
||||
return BRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline SRegister HRegister::S() const {
|
||||
return SRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline DRegister HRegister::D() const {
|
||||
return DRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline QRegister HRegister::Q() const {
|
||||
return QRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline ZRegister HRegister::Z() const {
|
||||
return ZRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline HRegister::operator VRegister () const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
inline HRegister::operator BRegister () const {
|
||||
return BRegister(Index);
|
||||
}
|
||||
inline HRegister::operator SRegister () const {
|
||||
return SRegister(Index);
|
||||
}
|
||||
inline HRegister::operator DRegister () const {
|
||||
return DRegister(Index);
|
||||
}
|
||||
inline HRegister::operator QRegister () const {
|
||||
return QRegister(Index);
|
||||
}
|
||||
inline HRegister::operator ZRegister () const {
|
||||
return ZRegister(Index);
|
||||
}
|
||||
|
||||
// SRegister
|
||||
@@ -551,39 +640,77 @@ namespace FEXCore::ARMEmitter {
|
||||
return *this;
|
||||
}
|
||||
inline BRegister SRegister::B() const {
|
||||
return BRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline HRegister SRegister::H() const {
|
||||
return HRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline DRegister SRegister::D() const {
|
||||
return DRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline QRegister SRegister::Q() const {
|
||||
return QRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline ZRegister SRegister::Z() const {
|
||||
return ZRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline SRegister::operator VRegister () const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
inline SRegister::operator BRegister () const {
|
||||
return BRegister(Index);
|
||||
}
|
||||
inline SRegister::operator HRegister () const {
|
||||
return HRegister(Index);
|
||||
}
|
||||
inline SRegister::operator DRegister () const {
|
||||
return DRegister(Index);
|
||||
}
|
||||
inline SRegister::operator QRegister () const {
|
||||
return QRegister(Index);
|
||||
}
|
||||
inline SRegister::operator ZRegister () const {
|
||||
return ZRegister(Index);
|
||||
}
|
||||
|
||||
// DRegister
|
||||
inline DRegister DRegister::V() const {
|
||||
return DRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline BRegister DRegister::B() const {
|
||||
return BRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline HRegister DRegister::H() const {
|
||||
return HRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline SRegister DRegister::S() const {
|
||||
return SRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline QRegister DRegister::Q() const {
|
||||
return QRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline ZRegister DRegister::Z() const {
|
||||
return ZRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline DRegister::operator VRegister () const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
inline DRegister::operator BRegister () const {
|
||||
return BRegister(Index);
|
||||
}
|
||||
inline DRegister::operator HRegister () const {
|
||||
return HRegister(Index);
|
||||
}
|
||||
inline DRegister::operator SRegister () const {
|
||||
return SRegister(Index);
|
||||
}
|
||||
inline DRegister::operator QRegister () const {
|
||||
return QRegister(Index);
|
||||
}
|
||||
inline DRegister::operator ZRegister () const {
|
||||
return ZRegister(Index);
|
||||
}
|
||||
|
||||
// QRegister
|
||||
@@ -591,19 +718,38 @@ namespace FEXCore::ARMEmitter {
|
||||
return *this;
|
||||
}
|
||||
inline BRegister QRegister::B() const {
|
||||
return BRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline HRegister QRegister::H() const {
|
||||
return HRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline SRegister QRegister::S() const {
|
||||
return SRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline DRegister QRegister::D() const {
|
||||
return DRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
inline ZRegister QRegister::Z() const {
|
||||
return ZRegister{Index};
|
||||
return *this;
|
||||
}
|
||||
|
||||
inline QRegister::operator VRegister () const {
|
||||
return VRegister(Index);
|
||||
}
|
||||
inline QRegister::operator BRegister () const {
|
||||
return BRegister(Index);
|
||||
}
|
||||
inline QRegister::operator HRegister () const {
|
||||
return HRegister(Index);
|
||||
}
|
||||
inline QRegister::operator SRegister () const {
|
||||
return SRegister(Index);
|
||||
}
|
||||
inline QRegister::operator DRegister () const {
|
||||
return DRegister(Index);
|
||||
}
|
||||
inline QRegister::operator ZRegister () const {
|
||||
return ZRegister(Index);
|
||||
}
|
||||
|
||||
// ZRegister
|
||||
@@ -923,12 +1069,17 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr PRegister(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const PRegister&, const PRegister&) = default;
|
||||
operator uint32_t() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator PRegisterZero() const;
|
||||
operator PRegisterMerge() const;
|
||||
|
||||
PRegisterZero Zeroing() const;
|
||||
PRegisterMerge Merging() const;
|
||||
|
||||
@@ -946,13 +1097,16 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr PRegisterZero(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const PRegisterZero&, const PRegisterZero&) = default;
|
||||
operator uint32_t() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator PRegister() const;
|
||||
operator PRegisterMerge() const;
|
||||
|
||||
PRegister P() const;
|
||||
PRegisterMerge Merging() const;
|
||||
@@ -971,13 +1125,16 @@ namespace FEXCore::ARMEmitter {
|
||||
constexpr PRegisterMerge(uint32_t Idx)
|
||||
: Index {Idx} {}
|
||||
|
||||
friend constexpr auto operator<=>(const PRegisterMerge&, const PRegisterMerge&) = default;
|
||||
operator uint32_t() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
uint32_t Idx() const {
|
||||
return Index;
|
||||
}
|
||||
|
||||
operator PRegister() const;
|
||||
operator PRegisterZero() const;
|
||||
|
||||
PRegister P() const;
|
||||
PRegisterZero Zeroing() const;
|
||||
@@ -991,6 +1148,14 @@ namespace FEXCore::ARMEmitter {
|
||||
|
||||
|
||||
// PRegister
|
||||
inline PRegister::operator PRegisterZero() const {
|
||||
return PRegisterZero(Index);
|
||||
}
|
||||
|
||||
inline PRegister::operator PRegisterMerge() const {
|
||||
return PRegisterMerge(Index);
|
||||
}
|
||||
|
||||
inline PRegisterZero PRegister::Zeroing() const {
|
||||
return PRegisterZero(Idx());
|
||||
}
|
||||
@@ -1004,6 +1169,10 @@ namespace FEXCore::ARMEmitter {
|
||||
return PRegister(Index);
|
||||
}
|
||||
|
||||
inline PRegisterZero::operator PRegisterMerge() const {
|
||||
return PRegisterMerge(Index);
|
||||
}
|
||||
|
||||
inline PRegister PRegisterZero::P() const {
|
||||
return PRegister(Idx());
|
||||
}
|
||||
@@ -1017,6 +1186,10 @@ namespace FEXCore::ARMEmitter {
|
||||
return PRegisterZero(Index);
|
||||
}
|
||||
|
||||
inline PRegisterMerge::operator PRegisterZero() const {
|
||||
return PRegisterZero(Index);
|
||||
}
|
||||
|
||||
inline PRegister PRegisterMerge::P() const {
|
||||
return PRegister(Idx());
|
||||
}
|
||||
|
||||
+1876
-2555
File diff suppressed because it is too large.
Load diff
+1090
-681
File diff suppressed because it is too large.
Load diff
+6
-47
@@ -1,44 +1,29 @@
|
||||
#pragma once
|
||||
|
||||
#include "UContext.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
|
||||
#include <signal.h>
|
||||
#include <string.h>
|
||||
#ifndef _WIN32
|
||||
#include <ucontext.h>
|
||||
#endif
|
||||
#include <stdint.h>
|
||||
#include <type_traits>
|
||||
|
||||
namespace FEX::ArchHelpers::Context {
|
||||
#ifndef _WIN32
|
||||
namespace FEXCore::ArchHelpers::Context {
|
||||
|
||||
enum ContextFlags : uint32_t {
|
||||
CONTEXT_FLAG_INJIT = (1U << 0),
|
||||
CONTEXT_FLAG_32BIT = (1U << 1),
|
||||
};
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
constexpr uint64_t STACK_COOKIE_MAGIC = 0x4142434445464748ULL;
|
||||
#endif
|
||||
|
||||
struct X86ContextBackup {
|
||||
// Host State
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// During debug builds, insert a cookie on the stack.
|
||||
// This is useful for validation that the stack is trying to be restored from the correct location.
|
||||
// During stack restore, we ensure this is set to the value we expect.
|
||||
// If given an incorrect stack location, or corrupted stack then this cookie will be wrong.
|
||||
uint64_t StackCookie;
|
||||
#endif
|
||||
// RIP and RSP is stored in GPRs here
|
||||
uint64_t GPRs[23];
|
||||
FEXCore::x86_64::_libc_fpstate FPRState;
|
||||
uint64_t sa_mask;
|
||||
uint16_t InSyscallInfo;
|
||||
bool FaultToTopAndGeneratedException;
|
||||
|
||||
// Guest state
|
||||
@@ -54,9 +39,6 @@ struct X86ContextBackup {
|
||||
|
||||
struct ArmContextBackup {
|
||||
// Host State
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
uint64_t StackCookie;
|
||||
#endif
|
||||
uint64_t GPRs[31];
|
||||
uint64_t PrevSP;
|
||||
uint64_t PrevPC;
|
||||
@@ -65,7 +47,6 @@ struct ArmContextBackup {
|
||||
uint32_t FPCR;
|
||||
__uint128_t FPRs[32];
|
||||
uint64_t sa_mask;
|
||||
uint16_t InSyscallInfo;
|
||||
bool FaultToTopAndGeneratedException;
|
||||
|
||||
// Guest state
|
||||
@@ -142,10 +123,6 @@ static inline uint64_t GetArmReg(void* ucontext, uint32_t id) {
|
||||
return GetMContext(ucontext)->regs[id];
|
||||
}
|
||||
|
||||
static inline uint64_t *GetArmGPRs(void* ucontext) {
|
||||
return reinterpret_cast<uint64_t*>(GetMContext(ucontext)->regs);
|
||||
}
|
||||
|
||||
static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
|
||||
GetMContext(ucontext)->regs[id] = val;
|
||||
}
|
||||
@@ -201,12 +178,12 @@ static inline uint32_t GetProtectFlags(void* ucontext) {
|
||||
uint32_t ProtectFlags{};
|
||||
if ((ESR & ESR1_DataAbort_Level) == ESR1_DataAbort_Level_EL0) {
|
||||
// Always a user error for us.
|
||||
ProtectFlags |= FEXCore::X86State::X86_PF_USER;
|
||||
ProtectFlags |= X86State::X86_PF_USER;
|
||||
}
|
||||
|
||||
if (ESR & ESR1_WNR) {
|
||||
// Fault was due to a write
|
||||
ProtectFlags |= FEXCore::X86State::X86_PF_WRITE;
|
||||
ProtectFlags |= X86State::X86_PF_WRITE;
|
||||
}
|
||||
|
||||
// PF_PROT is not returned to user on x86, so don't return the difference between permission fault and translation fault.
|
||||
@@ -234,10 +211,6 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
|
||||
// Save the signal mask so we can restore it
|
||||
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
Backup->StackCookie = STACK_COOKIE_MAGIC;
|
||||
#endif
|
||||
} else {
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
@@ -247,10 +220,8 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
template <typename T>
|
||||
static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, ArmContextBackup>::value) {
|
||||
LOGMAN_THROW_A_FMT(Backup->StackCookie == STACK_COOKIE_MAGIC, "Stack cookie didn't match! 0x{:x}", Backup->StackCookie);
|
||||
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
@@ -312,10 +283,6 @@ static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
static inline uint64_t *GetArmGPRs(void* ucontext) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
static inline uint32_t GetProtectFlags(void* ucontext) {
|
||||
return GetMContext(ucontext)->gregs[REG_ERR];
|
||||
}
|
||||
@@ -335,10 +302,6 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
|
||||
// Save the signal mask so we can restore it
|
||||
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
Backup->StackCookie = STACK_COOKIE_MAGIC;
|
||||
#endif
|
||||
} else {
|
||||
// This must be a runtime error
|
||||
ERROR_AND_DIE_FMT("Wrong context type");
|
||||
@@ -348,8 +311,6 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
template <typename T>
|
||||
static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
if constexpr (std::is_same<T, X86ContextBackup>::value) {
|
||||
LOGMAN_THROW_A_FMT(Backup->StackCookie == STACK_COOKIE_MAGIC, "Stack cookie didn't match! 0x{:x}", Backup->StackCookie);
|
||||
|
||||
auto _ucontext = GetUContext(ucontext);
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
@@ -367,7 +328,5 @@ static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
}
|
||||
|
||||
#endif
|
||||
#else
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::ArchHelpers::Context
|
||||
+4
-5
@@ -1,4 +1,3 @@
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
@@ -58,17 +57,17 @@ auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t *>(
|
||||
FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
|
||||
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
|
||||
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
if (ThreadState->CTX->Config.GlobalJITNaming()) {
|
||||
ThreadState->CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
|
||||
+147
-78
@@ -8,20 +8,18 @@ $end_info$
|
||||
#include "Common/StringConv.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include "git_version.h"
|
||||
|
||||
#include <cstring>
|
||||
#ifdef _M_X86_64
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include <cpuid.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -76,6 +74,21 @@ static uint32_t GetCPUID() {
|
||||
return CPU;
|
||||
}
|
||||
|
||||
static uint32_t CalculateNumberOfCPUs() {
|
||||
size_t CPUs = 1;
|
||||
|
||||
while(std::filesystem::exists("/sys/devices/system/cpu/cpu" + std::to_string(CPUs))) {
|
||||
CPUs++;
|
||||
}
|
||||
|
||||
return CPUs;
|
||||
}
|
||||
|
||||
// TODO: Replace usages with CTX->HostFeatures.EnableAVX
|
||||
// when AVX implementations are further along.
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
|
||||
// #define CPUID_AMD
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
0 | // Stepping
|
||||
@@ -103,30 +116,31 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
size_t CPUs = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
size_t CPUs = CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
|
||||
uint64_t MIDR{};
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
std::error_code ec{};
|
||||
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
std::string MIDRPath = "/sys/devices/system/cpu/cpu" + std::to_string(i) + "/regs/identification/midr_el1";
|
||||
if (std::filesystem::exists(MIDRPath, ec)) {
|
||||
std::vector<char> Data{};
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFile(Data, MIDRPath, 18)) {
|
||||
uint64_t NewMIDR{};
|
||||
std::string_view MIDRView(&Data.at(0), 18);
|
||||
if (FEXCore::StrConv::Conv(MIDRView, &NewMIDR)) {
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
}
|
||||
|
||||
std::array<char, 18> Data;
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFileToBuffer(MIDRPath, Data) == sizeof(Data)) {
|
||||
uint64_t NewMIDR{};
|
||||
std::string_view MIDRView(Data.data(), sizeof(Data));
|
||||
if (FEXCore::StrConv::Conv(MIDRView, &NewMIDR)) {
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -205,8 +219,8 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
if (Hybrid) {
|
||||
// Walk the MIDRs and calculate big little designs
|
||||
fextl::vector<const CPUMIDR*> BigCores;
|
||||
fextl::vector<const CPUMIDR*> LittleCores;
|
||||
std::vector<const CPUMIDR*> BigCores;
|
||||
std::vector<const CPUMIDR*> LittleCores;
|
||||
|
||||
// Separate CPU cores out to big or little selected
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
@@ -342,28 +356,28 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
#else
|
||||
static uint32_t GetCycleCounterFrequency() {
|
||||
uint32_t data[4];
|
||||
Xbyak::util::Cpu::getCpuid(0, data);
|
||||
if (data[0] >= 0x15) {
|
||||
Xbyak::util::Cpu::getCpuid(0x15, data);
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
__cpuid(0, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x15) {
|
||||
__cpuid(0x15, eax, ebx, ecx, edx);
|
||||
|
||||
if (data[0] && data[1] && data[2]) {
|
||||
return data[2] * data[1] / data[0];
|
||||
if (eax && ebx && ecx) {
|
||||
return ecx * ebx / eax;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
uint32_t data[4];
|
||||
Xbyak::util::Cpu::getCpuid(0, data);
|
||||
if (data[0] >= 0x7) {
|
||||
Xbyak::util::Cpu::getCpuid(0x7, data);
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
__cpuid(0, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x7) {
|
||||
__cpuid(0x7, eax, ebx, ecx, edx);
|
||||
// Bit 15 of edx claims hybrid CPU
|
||||
Hybrid = (data[3] & (1U << 15)) != 0;
|
||||
Hybrid = (edx & (1U << 15)) != 0;
|
||||
}
|
||||
|
||||
size_t CPUs = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
size_t CPUs = CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
PerCPUData[i].IsBig = true;
|
||||
@@ -395,9 +409,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
|
||||
// Hypervisor bit is normally set but some applications have issues with it.
|
||||
uint32_t Hypervisor = HideHypervisorBit() ? 0 : 1;
|
||||
// XXX: Enable once the rest of the SSE4.2 instructions are emulated
|
||||
uint32_t SupportsSSE42 = CTX->HostFeatures.SupportsCRC && false ? 1 : 0;
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
@@ -427,18 +440,18 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(0 << 17) | // Process-context identifiers
|
||||
(0 << 18) | // Prefetching from memory mapped device
|
||||
(1 << 19) | // SSE4.1
|
||||
(CTX->HostFeatures.SupportsCRC << 20) | // SSE4.2
|
||||
(SupportsSSE42 << 20) | // SSE4.2
|
||||
(0 << 21) | // X2APIC
|
||||
(1 << 22) | // MOVBE
|
||||
(1 << 23) | // POPCNT
|
||||
(0 << 24) | // APIC TSC-Deadline
|
||||
(CTX->HostFeatures.SupportsAES << 25) | // AES
|
||||
(SupportsAVX() << 26) | // XSAVE
|
||||
(SupportsAVX() << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 28) | // AVX
|
||||
(0 << 26) | // XSAVE
|
||||
(0 << 27) | // OSXSAVE
|
||||
(SUPPORTS_AVX << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(Hypervisor << 31);
|
||||
(1 << 31); // Hypervisor always returns one
|
||||
|
||||
Res.edx =
|
||||
(1 << 0) | // FPU
|
||||
@@ -624,12 +637,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(SupportsAVX() << 3) | // BMI1
|
||||
(1 << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(0 << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(SupportsAVX() << 8) | // BMI2
|
||||
(1 << 8) | // BMI2
|
||||
(0 << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
@@ -644,8 +657,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
|
||||
(0 << 21) | // Reserved
|
||||
(0 << 22) | // Reserved
|
||||
(1 << 23) | // CLFLUSHOPT instruction
|
||||
(CTX->HostFeatures.SupportsCLWB << 24) | // CLWB instruction
|
||||
(0 << 23) | // CLFLUSHOPT instruction
|
||||
(0 << 24) | // CLWB instruction
|
||||
(0 << 25) | // Intel processor trace
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
@@ -693,7 +706,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 1) | // Reserved
|
||||
(0 << 2) | // AVX512_4VNNIW
|
||||
(0 << 3) | // AVX512_4FMAPS
|
||||
(1 << 4) | // Fast Short Rep Mov
|
||||
(0 << 4) | // Fast Short Rep Mov
|
||||
(0 << 5) | // Reserved
|
||||
(0 << 6) | // Reserved
|
||||
(0 << 7) | // Reserved
|
||||
@@ -730,13 +743,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
// Leaf 0
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
uint32_t XFeatureSupportedSizeMax = SupportsAVX() ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
|
||||
uint32_t XFeatureSupportedSizeMax = SUPPORTS_AVX ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
|
||||
if (Leaf == 0) {
|
||||
// XFeatureSupportedMask[31:0]
|
||||
Res.eax =
|
||||
(1 << 0) | // X87 support
|
||||
(1 << 1) | // 128-bit SSE support
|
||||
(SupportsAVX() << 2) | // 256-bit AVX support
|
||||
(SUPPORTS_AVX << 2) | // 256-bit AVX support
|
||||
(0b00 << 3) | // MPX State
|
||||
(0b000 << 5) | // AVX-512 state
|
||||
(0 << 8) | // "Used for IA32_XSS" ... Used for what?
|
||||
@@ -770,8 +783,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
Res.edx = 0;
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
Res.eax = SupportsAVX() ? 0x0000'0100 : 0; // YmmSaveStateSize
|
||||
Res.ebx = SupportsAVX() ? 0x0000'0240 : 0; // YmmSaveStateOffset
|
||||
Res.eax = SUPPORTS_AVX ? 0x0000'0100 : 0; // YmmSaveStateSize
|
||||
Res.ebx = SUPPORTS_AVX ? 0x0000'0240 : 0; // YmmSaveStateOffset
|
||||
|
||||
// Reserved
|
||||
Res.ecx = 0;
|
||||
@@ -866,13 +879,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
|
||||
|
||||
// Extended processor and feature bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
|
||||
// RDTSCP is disabled on WIN32/Wine because there is no sane way to query processor ID.
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t SUPPORTS_RDTSCP = 0;
|
||||
#else
|
||||
constexpr uint32_t SUPPORTS_RDTSCP = 1;
|
||||
#endif
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
@@ -939,7 +945,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(SUPPORTS_RDTSCP << 27) | // RDTSCP
|
||||
(1 << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
(1 << 30) | // 3DNow! Extensions
|
||||
@@ -970,14 +976,14 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[0], std::min(ssize_t{16L}, DESCRIBE_STR_SIZE));
|
||||
memcpy(&Res, &ProcessorBrand[0], std::min(16L, DESCRIBE_STR_SIZE));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[16], std::max(ssize_t{0L}, DESCRIBE_STR_SIZE - 16));
|
||||
memcpy(&Res, &ProcessorBrand[16], std::max(0L, DESCRIBE_STR_SIZE - 16));
|
||||
return Res;
|
||||
}
|
||||
|
||||
@@ -1206,26 +1212,89 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() {
|
||||
// This just returns XCR0
|
||||
FEXCore::CPUID::XCRResults Res{
|
||||
.eax = static_cast<uint32_t>(XCR0),
|
||||
.edx = static_cast<uint32_t>(XCR0 >> 32),
|
||||
};
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
void CPUIDEmu::Init(FEXCore::Context::ContextImpl *ctx) {
|
||||
void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
CTX = ctx;
|
||||
|
||||
RegisterFunction(0, &CPUIDEmu::Function_0h);
|
||||
RegisterFunction(1, &CPUIDEmu::Function_01h);
|
||||
RegisterFunction(2, &CPUIDEmu::Function_02h);
|
||||
// 3: Serial Number(previously), now reserved
|
||||
#ifndef CPUID_AMD
|
||||
// Deterministic cache parameters for each level
|
||||
RegisterFunction(0x4, &CPUIDEmu::Function_04h);
|
||||
#endif
|
||||
// 5: Monitor/mwait
|
||||
// Thermal and power management
|
||||
RegisterFunction(6, &CPUIDEmu::Function_06h);
|
||||
// Extended feature flags
|
||||
RegisterFunction(7, &CPUIDEmu::Function_07h);
|
||||
// 9: Direct Cache Access information
|
||||
// 0x0A: Architectural performance monitoring
|
||||
// 0x0B: Extended topology enumeration
|
||||
// 0x0D: Processor extended state enumeration
|
||||
RegisterFunction(0x0D, &CPUIDEmu::Function_0Dh);
|
||||
// 0x0F: Intel RDT monitoring
|
||||
// 0x10: Intel RDT allocation enumeration
|
||||
// 0x12: Intel SGX capability enumeration
|
||||
// 0x13: Reserved
|
||||
// 0x14: Intel Processor trace
|
||||
#ifndef CPUID_AMD
|
||||
// Timestamp counter information
|
||||
// Doesn't exist on AMD hardware
|
||||
RegisterFunction(0x15, &CPUIDEmu::Function_15h);
|
||||
#endif
|
||||
// 0x16: Processor frequency information
|
||||
// 0x17: SoC vendor attribute enumeration
|
||||
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
#ifndef CPUID_AMD
|
||||
RegisterFunction(0x1A, &CPUIDEmu::Function_1Ah);
|
||||
#endif
|
||||
// Hypervisor CPUID information leaf
|
||||
RegisterFunction(0x4000'0000, &CPUIDEmu::Function_4000_0000h);
|
||||
RegisterFunction(0x4000'0001, &CPUIDEmu::Function_4000_0001h);
|
||||
|
||||
// Largest extended function number
|
||||
RegisterFunction(0x8000'0000, &CPUIDEmu::Function_8000_0000h);
|
||||
// Processor vendor
|
||||
RegisterFunction(0x8000'0001, &CPUIDEmu::Function_8000_0001h);
|
||||
// Processor brand string
|
||||
RegisterFunction(0x8000'0002, &CPUIDEmu::Function_8000_0002h);
|
||||
// Processor brand string continued
|
||||
RegisterFunction(0x8000'0003, &CPUIDEmu::Function_8000_0003h);
|
||||
// Processor brand string continued
|
||||
RegisterFunction(0x8000'0004, &CPUIDEmu::Function_8000_0004h);
|
||||
// 0x8000'0005: L1 Cache and TLB identifiers
|
||||
#ifdef CPUID_AMD
|
||||
RegisterFunction(0x8000'0005, &CPUIDEmu::Function_8000_0005h);
|
||||
#else
|
||||
// This is full reserved on Intel platforms
|
||||
RegisterFunction(0x8000'0005, &CPUIDEmu::Function_Reserved);
|
||||
#endif
|
||||
// 0x8000'0006: L2 Cache identifiers
|
||||
RegisterFunction(0x8000'0006, &CPUIDEmu::Function_8000_0006h);
|
||||
// Advanced power management information
|
||||
RegisterFunction(0x8000'0007, &CPUIDEmu::Function_8000_0007h);
|
||||
// Virtual and physical address sizes
|
||||
RegisterFunction(0x8000'0008, &CPUIDEmu::Function_8000_0008h);
|
||||
|
||||
// 0x8000'000A: SVM Revision
|
||||
// TLB 1GB page identifiers
|
||||
RegisterFunction(0x8000'0019, &CPUIDEmu::Function_8000_0019h);
|
||||
|
||||
// 0x8000'001A: Performance optimization identifiers
|
||||
// 0x8000'001B: Instruction based sampling identifiers
|
||||
// 0x8000'001C: Lightweight profiling capabilities
|
||||
// 0x8000'001D: Cache properties
|
||||
#ifdef CPUID_AMD
|
||||
// Deterministic cache parameters for each level
|
||||
RegisterFunction(0x8000'001D, &CPUIDEmu::Function_8000_001Dh);
|
||||
#endif
|
||||
// 0x8000'001E: Extended APIC ID
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
// TODO: Enable once AVX is supported.
|
||||
if (false && CTX->HostFeatures.SupportsAVX) {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+18
-218
@@ -1,21 +1,18 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
class ContextImpl;
|
||||
struct Context;
|
||||
}
|
||||
|
||||
// Debugging define to switch what family of CPU we execute as.
|
||||
// Might be useful if an application makes an assumption about a CPU.
|
||||
// #define CPUID_AMD
|
||||
class CPUIDEmu final {
|
||||
private:
|
||||
constexpr static uint32_t CPUID_VENDOR_INTEL1 = 0x756E6547; // "Genu"
|
||||
@@ -31,27 +28,16 @@ public:
|
||||
// if we report anything differently then applications are likely to break
|
||||
constexpr static uint64_t CACHELINE_SIZE = 64;
|
||||
|
||||
void Init(FEXCore::Context::ContextImpl *ctx);
|
||||
void Init(FEXCore::Context::Context *ctx);
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) {
|
||||
if (Function < Primary.size()) {
|
||||
const auto Handler = Primary[Function];
|
||||
return (this->*Handler)(Leaf);
|
||||
const auto Handler = FunctionHandlers.find(Function);
|
||||
|
||||
if (Handler == FunctionHandlers.end()) {
|
||||
return Function_Reserved(Leaf);
|
||||
}
|
||||
|
||||
constexpr uint32_t HypervisorBase = 0x4000'0000;
|
||||
if (Function >= HypervisorBase && Function < (HypervisorBase + Hypervisor.size())) {
|
||||
const auto Handler = Hypervisor[Function - HypervisorBase];
|
||||
return (this->*Handler)(Leaf);
|
||||
}
|
||||
|
||||
constexpr uint32_t ExtendedBase = 0x8000'0000;
|
||||
if (Function >= ExtendedBase && Function < (ExtendedBase + Extended.size())) {
|
||||
const auto Handler = Extended[Function - ExtendedBase];
|
||||
return (this->*Handler)(Leaf);
|
||||
}
|
||||
|
||||
return Function_Reserved(Leaf);
|
||||
return (this->*Handler->second)(Leaf);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
@@ -63,52 +49,17 @@ public:
|
||||
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) {
|
||||
if (Function >= 1) {
|
||||
// XCR function 1 is not yet supported.
|
||||
return {};
|
||||
}
|
||||
|
||||
return XCRFunction_0h();
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
FEXCore::Context::Context *CTX;
|
||||
bool Hybrid{};
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
// Mask that configures what features are enabled on the CPU.
|
||||
// Affects XSAVE and XRSTOR when modified.
|
||||
// Bit layout is as follows.
|
||||
// [0] - x87 enabled
|
||||
// [1] - SSE enabled
|
||||
// [2] - YMM enabled (256-bit SSE)
|
||||
// [8:3] - Reserved. MBZ.
|
||||
// [9] - MPK
|
||||
// [10] - Reserved. MBZ.
|
||||
// [11] - CET_U
|
||||
// [12] - CET_S
|
||||
// [61:13] - Reserved. MBZ.
|
||||
// [62] - LWP (Lightweight profiling)
|
||||
// [63] - Reserved for XCR bit vector expansion. MBZ.
|
||||
// Always enable x87 and SSE by default.
|
||||
constexpr static uint64_t XCR0_X87 = 1ULL << 0;
|
||||
constexpr static uint64_t XCR0_SSE = 1ULL << 1;
|
||||
constexpr static uint64_t XCR0_AVX = 1ULL << 2;
|
||||
|
||||
uint64_t XCR0 {
|
||||
XCR0_X87 |
|
||||
XCR0_SSE
|
||||
};
|
||||
|
||||
uint32_t SupportsAVX() const {
|
||||
return (XCR0 & XCR0_AVX) ? 1 : 0;
|
||||
}
|
||||
|
||||
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
|
||||
void RegisterFunction(uint32_t Function, FunctionHandler Handler) {
|
||||
FunctionHandlers.insert_or_assign(Function, Handler);
|
||||
}
|
||||
|
||||
std::unordered_map<uint32_t, FunctionHandler> FunctionHandlers;
|
||||
struct CPUData {
|
||||
const char *ProductName{};
|
||||
#ifdef _M_ARM_64
|
||||
@@ -116,7 +67,7 @@ private:
|
||||
#endif
|
||||
bool IsBig{};
|
||||
};
|
||||
fextl::vector<CPUData> PerCPUData{};
|
||||
std::vector<CPUData> PerCPUData{};
|
||||
|
||||
// Functions
|
||||
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
|
||||
@@ -144,163 +95,12 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0008h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0009h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
|
||||
|
||||
FEXCore::CPUID::XCRResults XCRFunction_0h();
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
static constexpr std::array<FunctionHandler, 27> Primary = {
|
||||
// 0: Highest function parameter and ID
|
||||
&CPUIDEmu::Function_0h,
|
||||
// 1: Processor info
|
||||
&CPUIDEmu::Function_01h,
|
||||
// 2: Cache and TLB info
|
||||
&CPUIDEmu::Function_02h,
|
||||
// 3: Serial Number(previously), now reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#ifndef CPUID_AMD
|
||||
// 4: Deterministic cache parameters for each level
|
||||
&CPUIDEmu::Function_04h,
|
||||
#else
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#endif
|
||||
// 5: Monitor/mwait
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 6: Thermal and power management
|
||||
&CPUIDEmu::Function_06h,
|
||||
// 7: Extended feature flags
|
||||
&CPUIDEmu::Function_07h,
|
||||
// 0x08: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 9: Direct Cache Access information
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x0A: Architectural performance monitoring
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x0B: Extended topology enumeration
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x0C: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x0D: Processor extended state enumeration
|
||||
&CPUIDEmu::Function_0Dh,
|
||||
// 0x0E: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x0F: Intel RDT monitoring
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x10: Intel RDT allocation enumeration
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x12: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x12: Intel SGX capability enumeration
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x13: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x14: Intel Processor trace
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#ifndef CPUID_AMD
|
||||
// Timestamp counter information
|
||||
// Doesn't exist on AMD hardware
|
||||
&CPUIDEmu::Function_15h,
|
||||
#else
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#endif
|
||||
// 0x16: Processor frequency information
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x17: SoC vendor attribute enumeration
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x18: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x19: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#ifndef CPUID_AMD
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
&CPUIDEmu::Function_1Ah,
|
||||
#else
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#endif
|
||||
};
|
||||
|
||||
static constexpr std::array<FunctionHandler, 2> Hypervisor = {
|
||||
// Hypervisor CPUID information leaf
|
||||
&CPUIDEmu::Function_4000_0000h,
|
||||
// FEX-Emu specific leaf
|
||||
&CPUIDEmu::Function_4000_0001h,
|
||||
};
|
||||
|
||||
static constexpr std::array<FunctionHandler, 32> Extended = {
|
||||
// Largest extended function number
|
||||
&CPUIDEmu::Function_8000_0000h,
|
||||
// Processor vendor
|
||||
&CPUIDEmu::Function_8000_0001h,
|
||||
// Processor brand string
|
||||
&CPUIDEmu::Function_8000_0002h,
|
||||
// Processor brand string continued
|
||||
&CPUIDEmu::Function_8000_0003h,
|
||||
// Processor brand string continued
|
||||
&CPUIDEmu::Function_8000_0004h,
|
||||
#ifdef CPUID_AMD
|
||||
// 0x8000'0005: L1 Cache and TLB identifiers
|
||||
&CPUIDEmu::Function_8000_0005h,
|
||||
#else
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#endif
|
||||
// 0x8000'0006: L2 Cache identifiers
|
||||
&CPUIDEmu::Function_8000_0006h,
|
||||
// 0x8000'0007: Advanced power management information
|
||||
&CPUIDEmu::Function_8000_0007h,
|
||||
// 0x8000'0008: Virtual and physical address sizes
|
||||
&CPUIDEmu::Function_8000_0008h,
|
||||
// 0x8000'0009: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'000A: SVM Revision
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'000B: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'000C: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'000D: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'000E: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'000F: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0010: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0011: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0012: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0013: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0014: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0015: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0016: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0017: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0018: Reserved?
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'0019: TLB 1GB page identifiers
|
||||
&CPUIDEmu::Function_8000_0019h,
|
||||
// 0x8000'001A: Performance optimization identifiers
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'001B: Instruction based sampling identifiers
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'001C: Lightweight profiling capabilities
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#ifdef CPUID_AMD
|
||||
// 0x8000'001D: Cache properties
|
||||
&CPUIDEmu::Function_8000_001Dh,
|
||||
#else
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#endif
|
||||
// 0x8000'001E: Extended APIC ID
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
};
|
||||
};
|
||||
}
|
||||
+224
-229
@@ -8,7 +8,6 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "FEXCore/Utils/DeferredSignalMutex.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
@@ -20,13 +19,10 @@ $end_info$
|
||||
#include "Interface/Core/Interpreter/InterpreterCore.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
@@ -36,6 +32,7 @@ $end_info$
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
@@ -45,15 +42,9 @@ $end_info$
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TodoDefines.h>
|
||||
|
||||
@@ -62,24 +53,33 @@ $end_info$
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <condition_variable>
|
||||
#include <fcntl.h>
|
||||
#include <filesystem>
|
||||
#include <functional>
|
||||
#include <fstream>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <queue>
|
||||
#include <set>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <type_traits>
|
||||
#include <unistd.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <xxhash.h>
|
||||
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
bool CreateCPUCore(Context::ContextImpl *CTX) {
|
||||
bool CreateCPUCore(FEXCore::Context::Context *CTX) {
|
||||
// This should be used for generating things that are shared between threads
|
||||
CTX->CPUID.Init(CTX);
|
||||
return true;
|
||||
@@ -147,17 +147,16 @@ std::string_view const& GetGRegName(unsigned Reg) {
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl()
|
||||
Context::Context()
|
||||
: IRCaptureCache {this} {
|
||||
#ifdef BLOCKSTATS
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
#endif
|
||||
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
CodeObjectCacheService = fextl::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
CodeObjectCacheService = std::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
}
|
||||
if (!Config.Is64BitMode()) {
|
||||
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
|
||||
Config.VirtualMemSize = 1ULL << 32;
|
||||
if (!Config.EnableAVX) {
|
||||
HostFeatures.SupportsAVX = false;
|
||||
}
|
||||
|
||||
if (Config.BlockJITNaming() ||
|
||||
@@ -166,16 +165,9 @@ namespace FEXCore::Context {
|
||||
// Only initialize symbols file if enabled. Ensures we don't pollute /tmp with empty files.
|
||||
Symbols.InitFile();
|
||||
}
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
ContextImpl::~ContextImpl() {
|
||||
if (ParentThread) {
|
||||
DestroyThread(ParentThread);
|
||||
}
|
||||
|
||||
Context::~Context() {
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
CodeObjectCacheService->Shutdown();
|
||||
@@ -194,57 +186,44 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
static FEXCore::Core::CPUState CreateDefaultCPUState() {
|
||||
FEXCore::Core::CPUState NewThreadState{};
|
||||
|
||||
if (InlineHeader) {
|
||||
auto InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
// Check if the host PC is currently within a code block.
|
||||
// If it is then RIP can be reconstructed from the beginning of the code block.
|
||||
// This is currently as close as FEX can get RIP reconstructions.
|
||||
if (HostPC >= reinterpret_cast<uint64_t>(BlockBegin) &&
|
||||
HostPC < reinterpret_cast<uint64_t>(BlockBegin + InlineTail->Size)) {
|
||||
|
||||
// Reconstruct RIP from JIT entries for this block.
|
||||
uint64_t StartingHostPC = BlockBegin;
|
||||
uint64_t StartingGuestRIP = InlineTail->RIP;
|
||||
|
||||
for (uint32_t i = 0; i < InlineTail->NumberOfRIPEntries; ++i) {
|
||||
const auto &RIPEntry = RIPEntries[i];
|
||||
if (HostPC >= (StartingHostPC + RIPEntry.HostPCOffset)) {
|
||||
// We are beyond this entry, keep going forward.
|
||||
StartingHostPC += RIPEntry.HostPCOffset;
|
||||
StartingGuestRIP += RIPEntry.GuestRIPOffset;
|
||||
}
|
||||
else {
|
||||
// Passed where the Host PC is at. Break now.
|
||||
break;
|
||||
}
|
||||
}
|
||||
return StartingGuestRIP;
|
||||
}
|
||||
// Initialize default CPU state
|
||||
NewThreadState.rip = ~0ULL;
|
||||
for (auto& greg : NewThreadState.gregs) {
|
||||
greg = 0;
|
||||
}
|
||||
|
||||
// Fallback to what is stored in the RIP currently.
|
||||
return Frame->State.rip;
|
||||
for (auto& xmm : NewThreadState.xmm.avx.data) {
|
||||
xmm[0] = 0xDEADBEEFULL;
|
||||
xmm[1] = 0xBAD0DAD1ULL;
|
||||
xmm[2] = 0xDEADCAFEULL;
|
||||
xmm[3] = 0xBAD2CAD3ULL;
|
||||
}
|
||||
memset(NewThreadState.flags, 0, Core::CPUState::NUM_EFLAG_BITS);
|
||||
NewThreadState.flags[1] = 1;
|
||||
NewThreadState.flags[9] = 1;
|
||||
NewThreadState.FCW = 0x37F;
|
||||
NewThreadState.FTW = 0xFFFF;
|
||||
return NewThreadState;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::InitCore(uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
FEXCore::Core::InternalThreadState* Context::InitCore(uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
// Initialize the CPU core signal handlers & DispatcherConfig
|
||||
switch (Config.Core) {
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
case FEXCore::Config::CONFIG_INTERPRETER:
|
||||
FEXCore::CPU::InitializeInterpreterSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetInterpreterBackendFeatures();
|
||||
break;
|
||||
#endif
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
FEXCore::CPU::InitializeX86JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetX86JITBackendFeatures();
|
||||
#elif (_M_ARM_64 && JIT_ARM64) || defined(VIXL_SIMULATOR)
|
||||
FEXCore::CPU::InitializeArm64JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetArm64JITBackendFeatures();
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
|
||||
@@ -268,37 +247,24 @@ namespace FEXCore::Context {
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled with an unknown target");
|
||||
#endif
|
||||
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
|
||||
.StaticRegisterAllocation = DispatcherConfig.StaticRegisterAllocation,
|
||||
.SupportsAVX = HostFeatures.SupportsAVX,
|
||||
// Initialize common signal handlers
|
||||
|
||||
.DispatcherBegin = Dispatcher->Start,
|
||||
.DispatcherEnd = Dispatcher->End,
|
||||
|
||||
.AbsoluteLoopTopAddressFillSRA = Dispatcher->AbsoluteLoopTopAddressFillSRA,
|
||||
.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress,
|
||||
.SignalHandlerReturnAddressRT = Dispatcher->SignalHandlerReturnAddressRT,
|
||||
|
||||
.PauseReturnInstruction = Dispatcher->PauseReturnInstruction,
|
||||
.ThreadPauseHandlerAddressSpillSRA = Dispatcher->ThreadPauseHandlerAddressSpillSRA,
|
||||
.ThreadPauseHandlerAddress = Dispatcher->ThreadPauseHandlerAddress,
|
||||
|
||||
// Stop handlers.
|
||||
.ThreadStopHandlerAddressSpillSRA = Dispatcher->ThreadStopHandlerAddressSpillSRA,
|
||||
.ThreadStopHandlerAddress = Dispatcher->ThreadStopHandlerAddress,
|
||||
|
||||
// SRA information.
|
||||
.SRAGPRCount = Dispatcher->GetSRAGPRCount(),
|
||||
.SRAFPRCount = Dispatcher->GetSRAFPRCount(),
|
||||
auto PauseHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSignalPause(Thread, Signal, info, ucontext);
|
||||
};
|
||||
|
||||
Dispatcher->GetSRAGPRMapping(SignalConfig.SRAGPRMapping);
|
||||
Dispatcher->GetSRAFPRMapping(SignalConfig.SRAFPRMapping);
|
||||
SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, PauseHandler, true);
|
||||
|
||||
// Give this configuration to the SignalDelegator.
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleGuestSignal(Thread, Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
|
||||
// Initialize GDBServer after the signal handlers are installed
|
||||
// It may install its own handlers that need to be executed AFTER the CPU cores
|
||||
if (Config.GdbServer) {
|
||||
StartGdbServer();
|
||||
}
|
||||
@@ -306,13 +272,12 @@ namespace FEXCore::Context {
|
||||
StopGdbServer();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#endif
|
||||
ThunkHandler.reset(FEXCore::ThunkHandler::Create());
|
||||
|
||||
using namespace FEXCore::Core;
|
||||
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(nullptr, 0);
|
||||
FEXCore::Core::CPUState NewThreadState = CreateDefaultCPUState();
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(&NewThreadState, 0);
|
||||
|
||||
// We are the parent thread
|
||||
ParentThread = Thread;
|
||||
@@ -325,26 +290,30 @@ namespace FEXCore::Context {
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::StartGdbServer() {
|
||||
#ifndef _WIN32
|
||||
void Context::StartGdbServer() {
|
||||
if (!DebugServer) {
|
||||
DebugServer = fextl::make_unique<GdbServer>(this);
|
||||
DebugServer = std::make_unique<GdbServer>(this);
|
||||
StartPaused = true;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::StopGdbServer() {
|
||||
#ifndef _WIN32
|
||||
void Context::StopGdbServer() {
|
||||
DebugServer.reset();
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
|
||||
void Context::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
Thread->CTX->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
|
||||
void ContextImpl::WaitForIdle() {
|
||||
void Context::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SignalDelegation->RegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void Context::RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SignalDelegation->RegisterFrontendHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void Context::WaitForIdle() {
|
||||
std::unique_lock<std::mutex> lk(IdleWaitMutex);
|
||||
IdleWaitCV.wait(lk, [this] {
|
||||
return IdleWaitRefCount.load() == 0;
|
||||
@@ -353,7 +322,7 @@ namespace FEXCore::Context {
|
||||
Running = false;
|
||||
}
|
||||
|
||||
void ContextImpl::WaitForIdleWithTimeout() {
|
||||
void Context::WaitForIdleWithTimeout() {
|
||||
std::unique_lock<std::mutex> lk(IdleWaitMutex);
|
||||
bool WaitResult = IdleWaitCV.wait_for(lk, std::chrono::milliseconds(1500),
|
||||
[this] {
|
||||
@@ -371,16 +340,20 @@ namespace FEXCore::Context {
|
||||
WaitForIdle();
|
||||
}
|
||||
|
||||
void ContextImpl::NotifyPause() {
|
||||
void Context::NotifyPause() {
|
||||
|
||||
// Tell all the threads that they should pause
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
SignalDelegation->SignalThread(Thread, FEXCore::Core::SignalEvent::Pause);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Pause);
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
// Only attempt to stop this thread if it is running
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::Pause() {
|
||||
void Context::Pause() {
|
||||
// If we aren't running, WaitForIdle will never compete.
|
||||
if (Running) {
|
||||
NotifyPause();
|
||||
@@ -389,7 +362,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::Run() {
|
||||
void Context::Run() {
|
||||
// Spin up all the threads
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
@@ -401,7 +374,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::WaitForThreadsToRun() {
|
||||
void Context::WaitForThreadsToRun() {
|
||||
size_t NumThreads{};
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
@@ -417,7 +390,7 @@ namespace FEXCore::Context {
|
||||
Running = true;
|
||||
}
|
||||
|
||||
void ContextImpl::Step() {
|
||||
void Context::Step() {
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
// Walk the threads and tell them to clear their caches
|
||||
@@ -437,7 +410,7 @@ namespace FEXCore::Context {
|
||||
this->Config.MaxInstPerBlock = PreviousMaxIntPerBlock;
|
||||
}
|
||||
|
||||
void ContextImpl::Stop(bool IgnoreCurrentThread) {
|
||||
void Context::Stop(bool IgnoreCurrentThread) {
|
||||
pid_t tid = FHU::Syscalls::gettid();
|
||||
FEXCore::Core::InternalThreadState* CurrentThread{};
|
||||
|
||||
@@ -475,19 +448,21 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
void Context::StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Thread->RunningEvents.Running.exchange(false)) {
|
||||
SignalDelegation->SignalThread(Thread, FEXCore::Core::SignalEvent::Stop);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Stop);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event) {
|
||||
void Context::SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event) {
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
SignalDelegation->SignalThread(Thread, Event);
|
||||
Thread->SignalReason.store(Event);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason ContextImpl::RunUntilExit() {
|
||||
FEXCore::Context::ExitReason Context::RunUntilExit() {
|
||||
if(!StartPaused) {
|
||||
// We will only have one thread at this point, but just in case run notify everything
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
@@ -508,16 +483,16 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
int ContextImpl::GetProgramStatus() const {
|
||||
int Context::GetProgramStatus() const {
|
||||
return ParentThread->StatusCode;
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeThreadData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
void Context::InitializeThreadData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CPUBackend->Initialize();
|
||||
}
|
||||
|
||||
struct ExecutionThreadHandler {
|
||||
ContextImpl *This;
|
||||
FEXCore::Context::Context *This;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
@@ -528,7 +503,7 @@ namespace FEXCore::Context {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
void Context::InitializeThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// This will create the execution thread but it won't actually start executing
|
||||
ExecutionThreadHandler *Arg = reinterpret_cast<ExecutionThreadHandler*>(FEXCore::Allocator::malloc(sizeof(ExecutionThreadHandler)));
|
||||
Arg->This = this;
|
||||
@@ -550,27 +525,25 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
void Context::InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Let's do some initial bookkeeping here
|
||||
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
|
||||
Thread->ThreadManager.PID = ::getpid();
|
||||
SignalDelegation->RegisterTLSState(Thread);
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
}
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
}
|
||||
|
||||
void ContextImpl::RunThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
void Context::RunThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Tell the thread to start executing
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
void Context::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Thread->OpDispatcher = std::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
|
||||
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
|
||||
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(this);
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
|
||||
Thread->LookupCache = std::make_unique<FEXCore::LookupCache>(this);
|
||||
Thread->FrontendDecoder = std::make_unique<FEXCore::Frontend::Decoder>(this);
|
||||
Thread->PassManager = std::make_unique<FEXCore::IR::PassManager>();
|
||||
Thread->PassManager->RegisterExitHandler([this]() {
|
||||
Stop(false /* Ignore current thread */);
|
||||
});
|
||||
@@ -617,13 +590,11 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState* Context::CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{};
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
if (NewThreadState) {
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
// Set up the thread manager state
|
||||
@@ -632,9 +603,6 @@ namespace FEXCore::Context {
|
||||
InitializeCompiler(Thread);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
|
||||
|
||||
// Insert after the Thread object has been fully initialized
|
||||
{
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
@@ -644,7 +612,7 @@ namespace FEXCore::Context {
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
void Context::DestroyThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// remove new thread object
|
||||
{
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
@@ -660,22 +628,10 @@ namespace FEXCore::Context {
|
||||
// To be able to delete a thread from itself, we need to detached the std::thread object
|
||||
Thread->ExecutionThread->detach();
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(Thread->CurrentFrame->State.DeferredSignalFaultAddress), 4096);
|
||||
delete Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState *LiveThread, bool Child) {
|
||||
Allocator::UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
CodeInvalidationMutex.unlock();
|
||||
return;
|
||||
}
|
||||
|
||||
void Context::CleanupAfterFork(FEXCore::Core::InternalThreadState *LiveThread) {
|
||||
// This function is called after fork
|
||||
// We need to cleanup some of the thread data that is dead
|
||||
for (auto &DeadThread : Threads) {
|
||||
@@ -714,16 +670,11 @@ namespace FEXCore::Context {
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
}
|
||||
|
||||
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
CodeInvalidationMutex.lock();
|
||||
Allocator::LockBeforeFork(Thread);
|
||||
}
|
||||
|
||||
void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
void Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
{
|
||||
@@ -739,46 +690,49 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, IR::IREmitter *IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
FEXCore::File::File FD;
|
||||
const auto DumpIRStr = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR();
|
||||
FILE* f = nullptr;
|
||||
bool CloseAfter = false;
|
||||
const auto DumpIRStr = Thread->CTX->Config.DumpIR();
|
||||
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpIRStr =="stderr" || DumpIRStr =="no") {
|
||||
FD = FEXCore::File::File::GetStdERR();
|
||||
f = stderr;
|
||||
}
|
||||
else if (DumpIRStr =="stdout") {
|
||||
FD = FEXCore::File::File::GetStdOUT();
|
||||
f = stdout;
|
||||
}
|
||||
else {
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIRStr, GuestRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
FD = FEXCore::File::File(fileName.c_str(),
|
||||
FEXCore::File::FileModes::WRITE |
|
||||
FEXCore::File::FileModes::CREATE |
|
||||
FEXCore::File::FileModes::TRUNCATE);
|
||||
const auto fileName = fmt::format("{}/{:x}{}", DumpIRStr, GuestRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
f = fopen(fileName.c_str(), "w");
|
||||
CloseAfter = true;
|
||||
}
|
||||
|
||||
if (FD.IsValid()) {
|
||||
fextl::stringstream out;
|
||||
if (f) {
|
||||
std::stringstream out;
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
fmt::print(f,"IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
|
||||
if (CloseAfter) {
|
||||
fclose(f);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
static void ValidateIR(ContextImpl *ctx, IR::IREmitter *IREmitter) {
|
||||
static void ValidateIR(FEXCore::Context::Context *ctx, IR::IREmitter *IREmitter) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
fextl::stringstream out;
|
||||
std::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction(ctx->OpDispatcherAllocator);
|
||||
compaction->Run(IREmitter);
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
FEXCore::Utils::PooledAllocatorMalloc Allocator;
|
||||
auto reparsed = IR::Parse(Allocator, out);
|
||||
auto reparsed = IR::Parse(Allocator, &out);
|
||||
if (reparsed == nullptr) {
|
||||
LOGMAN_MSG_A_FMT("Failed to parse IR\n");
|
||||
} else {
|
||||
fextl::stringstream out2;
|
||||
std::stringstream out2;
|
||||
auto NewIR2 = reparsed->ViewIR();
|
||||
Dump(&out2, &NewIR2, nullptr);
|
||||
if (out.str() != out2.str()) {
|
||||
@@ -789,7 +743,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
ContextImpl::GenerateIRResult ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
@@ -816,7 +770,7 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Thread, Start, Length);
|
||||
Thread->CTX->SyscallHandler->MarkGuestExecutableRange(Start, Length);
|
||||
}
|
||||
});
|
||||
|
||||
@@ -836,6 +790,13 @@ namespace FEXCore::Context {
|
||||
// Reset any block-specific state
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
|
||||
if (Config.x86dec_SynchronizeRIPOnAllBlocks) {
|
||||
// Ensure the RIP is synchronized to the context on block entry.
|
||||
// In the case of block linking, the RIP may not have synchronized.
|
||||
auto NewRIP = Thread->OpDispatcher->_EntrypointOffset(Block.Entry - GuestRIP, GPRSize);
|
||||
Thread->OpDispatcher->_StoreContext(GPRSize, IR::GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
}
|
||||
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
@@ -846,7 +807,7 @@ namespace FEXCore::Context {
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
|
||||
if (ExtendedDebugInfo) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
}
|
||||
|
||||
@@ -890,9 +851,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (TableInfo) {
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry - GuestRIP, GPRSize));
|
||||
@@ -930,14 +888,14 @@ namespace FEXCore::Context {
|
||||
|
||||
IR::IREmitter *IREmitter = Thread->OpDispatcher.get();
|
||||
|
||||
auto ShouldDump = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR() != "no" || Thread->OpDispatcher->ShouldDump;
|
||||
auto ShouldDump = Thread->CTX->Config.DumpIR() != "no" || Thread->OpDispatcher->ShouldDump;
|
||||
// Debug
|
||||
{
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP, nullptr);
|
||||
}
|
||||
|
||||
if (static_cast<ContextImpl*>(Thread->CTX)->Config.ValidateIRarser) {
|
||||
if (Thread->CTX->Config.ValidateIRarser) {
|
||||
ValidateIR(this, IREmitter);
|
||||
}
|
||||
}
|
||||
@@ -967,7 +925,7 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
Context::CompileCodeResult Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
@@ -995,7 +953,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
if (AOTIRCacheEntry.Entry && !AOTIRCacheEntry.Entry->ContainsCode) {
|
||||
AOTIRCacheEntry.Entry->SourcecodeMap =
|
||||
SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
|
||||
@@ -1004,7 +962,7 @@ namespace FEXCore::Context {
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(Thread, GuestRIP, IRList);
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(GuestRIP, IRList);
|
||||
if (_GeneratedIR) {
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
@@ -1027,6 +985,9 @@ namespace FEXCore::Context {
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
|
||||
// Increment stats
|
||||
Thread->Stats.BlocksCompiled.fetch_add(1);
|
||||
|
||||
// These blocks aren't already in the cache
|
||||
GeneratedIR = true;
|
||||
}
|
||||
@@ -1036,10 +997,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
return {
|
||||
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
|
||||
// In the future with code caching getting wired up, we will pass the rest of the data forward.
|
||||
// TODO: Pass the data forward when code caching is wired up to this.
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get(), GetGdbServerStatus()).BlockEntry,
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get(), GetGdbServerStatus()),
|
||||
.IRData = IRList,
|
||||
.DebugData = DebugData,
|
||||
.RAData = std::move(RAData),
|
||||
@@ -1049,7 +1007,7 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
void ContextImpl::CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
void Context::CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto NewBlock = CompileBlock(Frame, GuestRIP);
|
||||
|
||||
if (NewBlock == 0) {
|
||||
@@ -1060,12 +1018,12 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(CodeInvalidationMutex, Thread);
|
||||
std::shared_lock lk(CodeInvalidationMutex);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
@@ -1097,7 +1055,7 @@ namespace FEXCore::Context {
|
||||
auto FragmentBasePtr = reinterpret_cast<uint8_t *>(CodePtr);
|
||||
|
||||
if (DebugData) {
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
@@ -1122,7 +1080,7 @@ namespace FEXCore::Context {
|
||||
if (CodeObjectCacheService &&
|
||||
Config.CacheObjectCodeCompilation == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_READWRITE &&
|
||||
DebugData) {
|
||||
CodeObjectCacheService->AsyncAddSerializationJob(fextl::make_unique<CodeSerialize::AsyncJobHandler::SerializationJobData>(
|
||||
CodeObjectCacheService->AsyncAddSerializationJob(std::make_unique<CodeSerialize::AsyncJobHandler::SerializationJobData>(
|
||||
CodeSerialize::AsyncJobHandler::SerializationJobData {
|
||||
.GuestRIP = GuestRIP,
|
||||
.GuestCodeLength = Length,
|
||||
@@ -1160,19 +1118,18 @@ namespace FEXCore::Context {
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
void ContextImpl::ExecutionThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
void Context::ExecutionThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Core::ThreadData.Thread = Thread;
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
|
||||
InitializeThreadTLSData(Thread);
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
|
||||
++IdleWaitRefCount;
|
||||
|
||||
// Now notify the thread that we are initialized
|
||||
Thread->ThreadWaiting.NotifyAll();
|
||||
|
||||
if (Thread != static_cast<ContextImpl*>(Thread->CTX)->ParentThread || StartPaused || Thread->StartPaused) {
|
||||
if (Thread != Thread->CTX->ParentThread || StartPaused || Thread->StartPaused) {
|
||||
// Parent thread doesn't need to wait to run
|
||||
Thread->StartRunning.Wait();
|
||||
}
|
||||
@@ -1184,7 +1141,7 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
Thread->CTX->Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
Thread->RunningEvents.Running = false;
|
||||
}
|
||||
@@ -1210,11 +1167,10 @@ namespace FEXCore::Context {
|
||||
--IdleWaitRefCount;
|
||||
IdleWaitCV.notify_all();
|
||||
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
SignalDelegation->UninstallTLSState(Thread);
|
||||
|
||||
// If the parent thread is waiting to join, then we can't destroy our thread object
|
||||
if (!Thread->DestroyedByParent && Thread != static_cast<ContextImpl*>(Thread->CTX)->ParentThread) {
|
||||
if (!Thread->DestroyedByParent && Thread != Thread->CTX->ParentThread) {
|
||||
Thread->CTX->DestroyThread(Thread);
|
||||
}
|
||||
}
|
||||
@@ -1227,43 +1183,36 @@ namespace FEXCore::Context {
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (auto Address: it->second) {
|
||||
ContextImpl::ThreadRemoveCodeEntry(Thread, Address);
|
||||
Context::ThreadRemoveCodeEntry(Thread, Address);
|
||||
}
|
||||
it->second.clear();
|
||||
}
|
||||
}
|
||||
|
||||
static void InvalidateGuestCodeRangeInternal(ContextImpl *CTX, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard lk(static_cast<ContextImpl*>(CTX)->ThreadCreationMutex);
|
||||
static void InvalidateGuestCodeRangeInternal(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard lk(CTX->ThreadCreationMutex);
|
||||
|
||||
for (auto &Thread : static_cast<ContextImpl*>(CTX)->Threads) {
|
||||
for (auto &Thread : CTX->Threads) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CTX->CodeInvalidationMutex);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
InvalidateGuestCodeRangeInternal(CTX, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CTX->CodeInvalidationMutex);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
InvalidateGuestCodeRangeInternal(CTX, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::MarkMemoryShared() {
|
||||
void Context::MarkMemoryShared() {
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
|
||||
@@ -1281,14 +1230,18 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
void MarkMemoryShared(FEXCore::Context::Context *CTX) {
|
||||
CTX->MarkMemoryShared();
|
||||
}
|
||||
|
||||
void Context::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::shared_lock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
void Context::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(Thread->CTX->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
@@ -1296,7 +1249,7 @@ namespace FEXCore::Context {
|
||||
Thread->LookupCache->Erase(GuestRIP);
|
||||
}
|
||||
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
CustomIRResult Context::AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::unique_lock lk(CustomIRMutex);
|
||||
@@ -1312,23 +1265,66 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
|
||||
void Context::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::scoped_lock lk(CustomIRMutex);
|
||||
|
||||
InvalidateGuestCodeRange(nullptr, Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
InvalidateGuestCodeRange(this, Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
CustomIRHandlers.erase(Entrypoint);
|
||||
});
|
||||
}
|
||||
|
||||
// Debug interface
|
||||
void Context::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
uint64_t RIPBackup = Thread->CurrentFrame->State.rip;
|
||||
Thread->CurrentFrame->State.rip = RIP;
|
||||
|
||||
// Erase the RIP from all the storage backings if it exists
|
||||
ThreadRemoveCodeEntry(Thread, RIP);
|
||||
|
||||
// We don't care if compilation passes or not
|
||||
CompileBlock(Thread->CurrentFrame, RIP);
|
||||
|
||||
Thread->CurrentFrame->State.rip = RIPBackup;
|
||||
}
|
||||
|
||||
uint64_t Context::GetThreadCount() const {
|
||||
return Threads.size();
|
||||
}
|
||||
|
||||
FEXCore::Core::RuntimeStats *Context::GetRuntimeStatsForThread(uint64_t Thread) {
|
||||
return &Threads[Thread]->Stats;
|
||||
}
|
||||
|
||||
bool Context::GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
std::lock_guard<std::recursive_mutex> lk(ParentThread->LookupCache->WriteLock);
|
||||
auto it = ParentThread->DebugStore.find(RIP);
|
||||
if (it == ParentThread->DebugStore.end()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
memcpy(Data, it->second.DebugData.get(), sizeof(FEXCore::Core::DebugData));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Context::FindHostCodeForRIP(uint64_t RIP, uint8_t **Code) {
|
||||
uintptr_t HostCode = ParentThread->LookupCache->FindBlock(RIP);
|
||||
if (!HostCode) {
|
||||
return false;
|
||||
}
|
||||
|
||||
*Code = reinterpret_cast<uint8_t*>(HostCode);
|
||||
return true;
|
||||
}
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args) {
|
||||
uint64_t Result{};
|
||||
Result = Handler->HandleSyscall(Frame, Args);
|
||||
return Result;
|
||||
}
|
||||
|
||||
IR::AOTIRCacheEntry *ContextImpl::LoadAOTIRCacheEntry(const fextl::string &filename) {
|
||||
IR::AOTIRCacheEntry *Context::LoadAOTIRCacheEntry(const std::string &filename) {
|
||||
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
@@ -1336,20 +1332,19 @@ namespace FEXCore::Context {
|
||||
return rv;
|
||||
}
|
||||
|
||||
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry) {
|
||||
void Context::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
|
||||
void Context::AppendThunkDefinitions(std::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
|
||||
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
|
||||
}
|
||||
|
||||
+3
-3
@@ -5,7 +5,7 @@ namespace FEXCore {
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -17,7 +17,7 @@ namespace FEXCore::CPU {
|
||||
*
|
||||
* @return true if core was able to be create
|
||||
*/
|
||||
bool CreateCPUCore(FEXCore::Context::ContextImpl *CTX);
|
||||
bool CreateCPUCore(FEXCore::Context::Context *CTX);
|
||||
|
||||
bool LoadCode(FEXCore::Context::ContextImpl *CTX, FEXCore::CodeLoader *Loader);
|
||||
bool LoadCode(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader);
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
|
||||
@@ -11,7 +12,6 @@
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <array>
|
||||
@@ -28,13 +28,14 @@
|
||||
#include <code-buffer-vixl.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
|
||||
#ifdef VIXL_SIMULATOR
|
||||
, Simulator {&Decoder}
|
||||
@@ -129,6 +130,13 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// VIXL simulator can't run syscalls.
|
||||
constexpr bool SignalSafeCompile = false;
|
||||
#else
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
#endif
|
||||
|
||||
ARMEmitter::ForwardLabel NoBlock;
|
||||
|
||||
{
|
||||
@@ -147,16 +155,14 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
|
||||
// The the full LookupCacheEntry with a single LDP.
|
||||
// Check the guest address first to ensure it maps to the address we are currently at.
|
||||
// Load the guest address first to ensure it maps to the address we are currently at
|
||||
// This fixes aliasing problems
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode));
|
||||
cmp(ARMEmitter::XReg::x1, RipReg);
|
||||
b(ARMEmitter::Condition::CC_NE, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
// Now load the actual host block to execute if we can
|
||||
ldr(ARMEmitter::XReg::x3, ARMEmitter::Reg::r0, offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode));
|
||||
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
@@ -176,7 +182,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -190,11 +196,26 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
}
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
@@ -206,17 +227,26 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
|
||||
mov(ARMEmitter::XReg::x4, ARMEmitter::XReg::x0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
mov(ARMEmitter::XReg::x0, ARMEmitter::XReg::x4);
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, ARMEmitter::XReg::x1, 0);
|
||||
|
||||
br(ARMEmitter::Reg::r0);
|
||||
}
|
||||
|
||||
@@ -225,11 +255,29 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
Bind(&NoBlock);
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Reload x2 to bring back RIP
|
||||
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, 8);
|
||||
}
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
@@ -242,17 +290,23 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, TMP1, 0);
|
||||
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
@@ -264,21 +318,13 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
hlt(0);
|
||||
}
|
||||
|
||||
{
|
||||
SignalHandlerReturnAddressRT = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Now to get back to our old location we need to do a fault dance
|
||||
// We can't use SIGTRAP here since gdb catches it and never gives it to the application!
|
||||
hlt(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
@@ -289,7 +335,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
brk(0);
|
||||
}
|
||||
@@ -300,28 +346,20 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
if (CTX->ExitOnHLTEnabled()) {
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::r0, 0);
|
||||
PopCalleeSavedRegisters();
|
||||
ret();
|
||||
}
|
||||
else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
}
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
// We are pausing, this means the frontend should be waiting for this thread to idle
|
||||
@@ -398,7 +436,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -420,7 +458,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -442,7 +480,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -464,7 +502,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
@@ -495,7 +533,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
ClearICache(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
fextl::string Name = fextl::fmt::format("Dispatch_{}", FHU::Syscalls::gettid());
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
@@ -530,10 +568,10 @@ size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t Gues
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(FEXCore::Context::ContextImpl::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
static_assert(sizeof(FEXCore::Context::Context::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Thread));
|
||||
emit.ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, offsetof(FEXCore::Core::InternalThreadState, CTX)); // Get Context
|
||||
emit.ldr(ARMEmitter::WReg::w0, ARMEmitter::Reg::r0, offsetof(FEXCore::Context::ContextImpl, Config.RunningMode));
|
||||
emit.ldr(ARMEmitter::WReg::w0, ARMEmitter::Reg::r0, offsetof(FEXCore::Context::Context, Config.RunningMode));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cbz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, &RunBlock);
|
||||
@@ -578,6 +616,28 @@ size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
|
||||
for (size_t i = 0; i < SRA64.size(); i++) {
|
||||
if (IgnoreMask & (1U << SRA64[i].Idx())) {
|
||||
// Skip this one, it's already spilled
|
||||
continue;
|
||||
}
|
||||
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].Idx());
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].Idx());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].Idx());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
@@ -592,7 +652,6 @@ void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thr
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
|
||||
|
||||
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
|
||||
AArch64.LUDIVHandler = LUDIVHandlerAddress;
|
||||
@@ -602,8 +661,8 @@ void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thr
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
|
||||
return fextl::make_unique<Arm64Dispatcher>(CTX, Config);
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<Arm64Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -7,6 +7,10 @@
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
@@ -18,7 +22,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
public:
|
||||
Arm64Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config);
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
@@ -30,25 +34,8 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
|
||||
void EmitDispatcher();
|
||||
|
||||
uint16_t GetSRAGPRCount() const override {
|
||||
return StaticRegisters.size();
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const override {
|
||||
return StaticFPRegisters.size();
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
void GetSRAFPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticFPRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
protected:
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
|
||||
|
||||
private:
|
||||
// Long division helpers
|
||||
|
||||
@@ -1,11 +1,13 @@
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
@@ -20,7 +22,7 @@
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void Dispatcher::SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame) {
|
||||
void Dispatcher::SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuStateFrame *Frame) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
--ctx->IdleWaitRefCount;
|
||||
@@ -38,15 +40,801 @@ void Dispatcher::SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
}
|
||||
|
||||
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext) {
|
||||
// We can end up getting a signal at any point in our host state
|
||||
// Jump to a handler that saves all state so we can safely return
|
||||
uint64_t OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
uintptr_t NewSP = OldSP;
|
||||
|
||||
size_t StackOffset = sizeof(ArchHelpers::Context::ContextBackup);
|
||||
|
||||
// We need to back up behind the host's red zone
|
||||
// We do this on the guest side as well
|
||||
// (does nothing on arm hosts)
|
||||
NewSP -= ArchHelpers::Context::ContextBackup::RedZoneSize;
|
||||
|
||||
NewSP -= StackOffset;
|
||||
NewSP = AlignDown(NewSP, 16);
|
||||
|
||||
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
|
||||
ArchHelpers::Context::BackupContext(ucontext, Context);
|
||||
|
||||
// Retain the action pointer so we can see it when we return
|
||||
Context->Signal = Signal;
|
||||
|
||||
// Save guest state
|
||||
// We can't guarantee if registers are in context or host GPRs
|
||||
// So we need to save everything
|
||||
memcpy(&Context->GuestState, Thread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Set the new SP
|
||||
ArchHelpers::Context::SetSp(ucontext, NewSP);
|
||||
|
||||
// Signal frames are only used on the interpreter
|
||||
// The JITS require the stack to be setup correctly on rt_sigreturn
|
||||
if (CTX->Config.Core() == FEXCore::Config::CONFIG_INTERPRETER) {
|
||||
SignalFrames.push(NewSP);
|
||||
}
|
||||
|
||||
Context->Flags = 0;
|
||||
Context->FPStateLocation = 0;
|
||||
Context->UContextLocation = 0;
|
||||
Context->SigInfoLocation = 0;
|
||||
|
||||
// Store fault to top status and then reset it
|
||||
Context->FaultToTopAndGeneratedException = Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
|
||||
return Context;
|
||||
}
|
||||
|
||||
void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext) {
|
||||
uint64_t OldSP{};
|
||||
if (CTX->Config.Core() == FEXCore::Config::CONFIG_IRJIT) {
|
||||
OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(!SignalFrames.empty(), "Trying to restore a signal frame when we don't have any");
|
||||
OldSP = SignalFrames.top();
|
||||
SignalFrames.pop();
|
||||
}
|
||||
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
uintptr_t NewSP = OldSP;
|
||||
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
|
||||
|
||||
// First thing, reset the guest state
|
||||
memcpy(Thread->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Now restore host state
|
||||
ArchHelpers::Context::RestoreContext(ucontext, Context);
|
||||
|
||||
if (Context->UContextLocation) {
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (Context->Flags &ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT) {
|
||||
// XXX: Unsupported since it needs state reconstruction
|
||||
// If we are in the JIT then SRA might need to be restored to values from the context
|
||||
// We can't currently support this since it might result in tearing without real state reconstruction
|
||||
}
|
||||
|
||||
if (!(Context->Flags & ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT)) {
|
||||
auto *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(Context->UContextLocation);
|
||||
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<siginfo_t*>(Context->SigInfoLocation);
|
||||
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] ||
|
||||
Context->FaultToTopAndGeneratedException) {
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
|
||||
// XXX: Full context setting
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
Frame->State.flags[1] = 1;
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x];
|
||||
COPY_REG(R8);
|
||||
COPY_REG(R9);
|
||||
COPY_REG(R10);
|
||||
COPY_REG(R11);
|
||||
COPY_REG(R12);
|
||||
COPY_REG(R13);
|
||||
COPY_REG(R14);
|
||||
COPY_REG(R15);
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][0], &fpstate->_xmm[i], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][2], &xstate->ymmh.ymmh_space[i], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.FTW = fpstate->ftw;
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(Context->UContextLocation);
|
||||
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(Context->SigInfoLocation);
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] ||
|
||||
Context->FaultToTopAndGeneratedException) {
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
// XXX: Full context setting
|
||||
// First 32-bytes of flags is EFLAGS broken out
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
Frame->State.flags[1] = 1;
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
|
||||
Frame->State.cs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
|
||||
Frame->State.cs_cached = Frame->State.gdt[Frame->State.cs_idx >> 3].base;
|
||||
Frame->State.ds_cached = Frame->State.gdt[Frame->State.ds_idx >> 3].base;
|
||||
Frame->State.es_cached = Frame->State.gdt[Frame->State.es_idx >> 3].base;
|
||||
Frame->State.fs_cached = Frame->State.gdt[Frame->State.fs_idx >> 3].base;
|
||||
Frame->State.gs_cached = Frame->State.gdt[Frame->State.gs_idx >> 3].base;
|
||||
Frame->State.ss_cached = Frame->State.gdt[Frame->State.ss_idx >> 3].base;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.FTW = fpstate->ftw;
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static uint32_t ConvertSignalToTrapNo(int Signal, siginfo_t *HostSigInfo) {
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
if (HostSigInfo->si_code == SEGV_MAPERR ||
|
||||
HostSigInfo->si_code == SEGV_ACCERR) {
|
||||
// Protection fault
|
||||
return X86State::X86_TRAPNO_PF;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// Unknown mapping, fall back to old behaviour and just pass signal
|
||||
return Signal;
|
||||
}
|
||||
|
||||
static uint32_t ConvertSignalToError(void *ucontext, int Signal, siginfo_t *HostSigInfo) {
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
if (HostSigInfo->si_code == SEGV_MAPERR ||
|
||||
HostSigInfo->si_code == SEGV_ACCERR) {
|
||||
// Protection fault
|
||||
// Always a user fault for us
|
||||
return ArchHelpers::Context::GetProtectFlags(ucontext);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// Not a page fault issue
|
||||
return 0;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void SetXStateInfo(T* xstate, bool is_avx_enabled) {
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
fpstate->sw_reserved.magic1 = x86_64::fpx_sw_bytes::FP_XSTATE_MAGIC;
|
||||
fpstate->sw_reserved.extended_size = is_avx_enabled ? sizeof(T) : 0;
|
||||
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_FP |
|
||||
x86_64::fpx_sw_bytes::FEATURE_SSE;
|
||||
if (is_avx_enabled) {
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_YMM;
|
||||
}
|
||||
|
||||
fpstate->sw_reserved.xstate_size = fpstate->sw_reserved.extended_size;
|
||||
|
||||
if (is_avx_enabled) {
|
||||
xstate->xstate_hdr.xfeatures = 0;
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
auto ContextBackup = StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
|
||||
// Set the new PC
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
uint64_t OldGuestSP = Frame->State.gregs[X86State::REG_RSP];
|
||||
uint64_t NewGuestSP = OldGuestSP;
|
||||
|
||||
// Pulling from context here
|
||||
const bool Is64BitMode = CTX->Config.Is64BitMode;
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
const uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
|
||||
|
||||
// Spill the SRA regardless of signal handler type
|
||||
// We are going to be returning to the top of the dispatcher which will fill again
|
||||
// Otherwise we might load garbage
|
||||
if (config.StaticRegisterAllocation) {
|
||||
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
|
||||
uint32_t IgnoreMask{};
|
||||
#ifdef _M_ARM_64
|
||||
if (Frame->InSyscallInfo != 0) {
|
||||
// We are in a syscall, this means we are in a weird register state
|
||||
// We need to spill SRA but only some of it, since some values have already been spilled
|
||||
// Lower 16 bits tells us which registers are already spilled to the context
|
||||
// So we ignore spilling those ones
|
||||
uint16_t NumRegisters = std::popcount(Frame->InSyscallInfo & 0xFFFF);
|
||||
if (NumRegisters >= 4) {
|
||||
// Unhandled case
|
||||
IgnoreMask = 0;
|
||||
}
|
||||
else {
|
||||
IgnoreMask = Frame->InSyscallInfo & 0xFFFF;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We must spill everything
|
||||
IgnoreMask = 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
// We are in jit, SRA must be spilled
|
||||
SpillSRA(Thread, ucontext, IgnoreMask);
|
||||
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT;
|
||||
} else {
|
||||
if (!IsAddressInDispatcher(OldPC)) {
|
||||
// This is likely to cause issues but in some cases it isn't fatal
|
||||
// This can also happen if we have put a signal on hold, then we just reenabled the signal
|
||||
// So we are in the syscall handler
|
||||
// Only throw a log message in this case
|
||||
if constexpr (false) {
|
||||
// XXX: Messages in the signal handler can cause us to crash
|
||||
LogMan::Msg::EFmt("Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// altstack is only used if the signal handler was setup with SA_ONSTACK
|
||||
if (GuestAction->sa_flags & SA_ONSTACK) {
|
||||
// Additionally the altstack is only used if the enabled (SS_DISABLE flag is not set)
|
||||
if (!(GuestStack->ss_flags & SS_DISABLE)) {
|
||||
// If our guest is already inside of the alternative stack
|
||||
// Then that means we are hitting recursive signals and we need to walk back the stack correctly
|
||||
uint64_t AltStackBase = reinterpret_cast<uint64_t>(GuestStack->ss_sp);
|
||||
uint64_t AltStackEnd = AltStackBase + GuestStack->ss_size;
|
||||
if (OldGuestSP >= AltStackBase &&
|
||||
OldGuestSP <= AltStackEnd) {
|
||||
// We are already in the alt stack, the rest of the code will handle adjusting this
|
||||
}
|
||||
else {
|
||||
NewGuestSP = AltStackEnd;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Is64BitMode) {
|
||||
// Back up past the redzone, which is 128bytes
|
||||
// 32-bit doesn't have a redzone
|
||||
NewGuestSP -= 128;
|
||||
}
|
||||
|
||||
// siginfo_t
|
||||
siginfo_t *HostSigInfo = reinterpret_cast<siginfo_t*>(info);
|
||||
|
||||
// Backup where we think the RIP currently is
|
||||
ContextBackup->OriginalRIP = Frame->State.rip;
|
||||
|
||||
if (GuestAction->sa_flags & SA_SIGINFO) {
|
||||
// Setup ucontext a bit
|
||||
if (Is64BitMode) {
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86_64::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86_64::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::_libc_fpstate));
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::ucontext_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::ucontext_t));
|
||||
uint64_t UContextLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(siginfo_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(siginfo_t));
|
||||
uint64_t SigInfoLocation = NewGuestSP;
|
||||
|
||||
ContextBackup->FPStateLocation = FPStateLocation;
|
||||
ContextBackup->UContextLocation = UContextLocation;
|
||||
ContextBackup->SigInfoLocation = SigInfoLocation;
|
||||
|
||||
FEXCore::x86_64::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(UContextLocation);
|
||||
siginfo_t *guest_siginfo = reinterpret_cast<siginfo_t*>(SigInfoLocation);
|
||||
|
||||
// We have extended float information
|
||||
guest_uctx->uc_flags = FEXCore::x86_64::UC_FP_XSTATE;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CSGSFS] = 0;
|
||||
|
||||
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
|
||||
// SI_USER could also potentially have random data in it, needs to be bit perfect
|
||||
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
|
||||
// Overwrite si_code
|
||||
guest_siginfo->si_code = Thread->CurrentFrame->SynchronousFaultData.si_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(ucontext, Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_OLDMASK] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CR2] = 0;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
|
||||
COPY_REG(R8);
|
||||
COPY_REG(R9);
|
||||
COPY_REG(R10);
|
||||
COPY_REG(R11);
|
||||
COPY_REG(R12);
|
||||
COPY_REG(R13);
|
||||
COPY_REG(R14);
|
||||
COPY_REG(R15);
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
fpstate->ftw = Frame->State.FTW;
|
||||
|
||||
// Reconstruct FSW
|
||||
fpstate->fsw =
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] << 10) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] << 14);
|
||||
|
||||
// Copy over signal stack information
|
||||
guest_uctx->uc_stack.ss_flags = GuestStack->ss_flags;
|
||||
guest_uctx->uc_stack.ss_sp = GuestStack->ss_sp;
|
||||
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
|
||||
|
||||
Frame->State.gregs[X86State::REG_RSI] = SigInfoLocation;
|
||||
Frame->State.gregs[X86State::REG_RDX] = UContextLocation;
|
||||
}
|
||||
else {
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT;
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::_libc_fpstate));
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::ucontext_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::ucontext_t));
|
||||
uint64_t UContextLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::siginfo_t);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::siginfo_t));
|
||||
uint64_t SigInfoLocation = NewGuestSP;
|
||||
|
||||
ContextBackup->FPStateLocation = FPStateLocation;
|
||||
ContextBackup->UContextLocation = UContextLocation;
|
||||
ContextBackup->SigInfoLocation = SigInfoLocation;
|
||||
|
||||
FEXCore::x86::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(UContextLocation);
|
||||
FEXCore::x86::siginfo_t *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(SigInfoLocation);
|
||||
|
||||
// We have extended float information
|
||||
guest_uctx->uc_flags = FEXCore::x86::UC_FP_XSTATE;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = static_cast<uint32_t>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss_idx;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(ucontext, Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = 0;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
fpstate->ftw = Frame->State.FTW;
|
||||
// Reconstruct FSW
|
||||
fpstate->fsw =
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] << 10) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] << 14);
|
||||
|
||||
// Copy over signal stack information
|
||||
guest_uctx->uc_stack.ss_flags = GuestStack->ss_flags;
|
||||
guest_uctx->uc_stack.ss_sp = static_cast<uint32_t>(reinterpret_cast<uint64_t>(GuestStack->ss_sp));
|
||||
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
|
||||
|
||||
// These three elements are in every siginfo
|
||||
guest_siginfo->si_signo = HostSigInfo->si_signo;
|
||||
guest_siginfo->si_errno = HostSigInfo->si_errno;
|
||||
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
case SIGBUS:
|
||||
// Macro expansion to get the si_addr
|
||||
// This is the address trying to be accessed, not the RIP
|
||||
guest_siginfo->_sifields._sigfault.addr = static_cast<uint32_t>(reinterpret_cast<uintptr_t>(HostSigInfo->si_addr));
|
||||
break;
|
||||
case SIGFPE:
|
||||
case SIGILL:
|
||||
// Macro expansion to get the si_addr
|
||||
// Can't really give a real result here. Pull from the context for now
|
||||
guest_siginfo->_sifields._sigfault.addr = Frame->State.rip;
|
||||
break;
|
||||
case SIGCHLD:
|
||||
guest_siginfo->_sifields._sigchld.pid = HostSigInfo->si_pid;
|
||||
guest_siginfo->_sifields._sigchld.uid = HostSigInfo->si_uid;
|
||||
guest_siginfo->_sifields._sigchld.status = HostSigInfo->si_status;
|
||||
guest_siginfo->_sifields._sigchld.utime = HostSigInfo->si_utime;
|
||||
guest_siginfo->_sifields._sigchld.stime = HostSigInfo->si_stime;
|
||||
break;
|
||||
case SIGALRM:
|
||||
case SIGVTALRM:
|
||||
guest_siginfo->_sifields._timer.tid = HostSigInfo->si_timerid;
|
||||
guest_siginfo->_sifields._timer.overrun = HostSigInfo->si_overrun;
|
||||
guest_siginfo->_sifields._timer.sigval.sival_int = HostSigInfo->si_int;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled siginfo_t for signal: {}\n", Signal);
|
||||
break;
|
||||
}
|
||||
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = UContextLocation;
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = SigInfoLocation;
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = Signal;
|
||||
}
|
||||
|
||||
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.sigaction);
|
||||
}
|
||||
else {
|
||||
if (!Is64BitMode) {
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = Signal;
|
||||
}
|
||||
|
||||
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.handler);
|
||||
}
|
||||
|
||||
if (Is64BitMode) {
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RDI] = Signal;
|
||||
|
||||
// Set up the new SP for stack handling
|
||||
NewGuestSP -= 8;
|
||||
*(uint64_t*)NewGuestSP = SignalReturn;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
}
|
||||
else {
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = SignalReturn;
|
||||
LOGMAN_THROW_AA_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
}
|
||||
|
||||
// The guest starts its signal frame with a zero initialized FPU
|
||||
// Set that up now. Little bit costly but it's a requirement
|
||||
// This state will be restored on rt_sigreturn
|
||||
memset(Frame->State.xmm.avx.data, 0, sizeof(Frame->State.xmm));
|
||||
memset(Frame->State.mm, 0, sizeof(Frame->State.mm));
|
||||
Frame->State.FCW = 0x37F;
|
||||
Frame->State.FTW = 0xFFFF;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == SignalHandlerReturnAddress) {
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == PauseReturnInstruction) {
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = Thread->SignalReason.load();
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
|
||||
// Store our thread state so we can come back to this
|
||||
StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
|
||||
}
|
||||
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Stop) {
|
||||
// Our thread is stopping
|
||||
// We don't care about anything at this point
|
||||
// Set the stack to our starting location when we entered the core and get out safely
|
||||
ArchHelpers::Context::SetSp(ucontext, Frame->ReturningStackLocation);
|
||||
|
||||
// Our ref counting doesn't matter anymore
|
||||
Thread->CurrentFrame->SignalHandlerRefCounter = 0;
|
||||
|
||||
// Set the new PC
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
|
||||
}
|
||||
|
||||
// We need to be a little bit careful here
|
||||
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
|
||||
// Then we need to ensure we don't double decrement our idle thread counter
|
||||
if (Thread->RunningEvents.ThreadSleeping) {
|
||||
// If the thread was sleeping then its idle counter was decremented
|
||||
// Reincrement it here to not break logic
|
||||
++Thread->CTX->IdleWaitRefCount;
|
||||
}
|
||||
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Return) {
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
uint64_t Dispatcher::GetCompileBlockPtr() {
|
||||
using ClassPtrType = void (FEXCore::Context::ContextImpl::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
|
||||
using ClassPtrType = void (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
|
||||
union PtrCast {
|
||||
ClassPtrType ClassPtr;
|
||||
uintptr_t Data;
|
||||
};
|
||||
|
||||
PtrCast CompileBlockPtr;
|
||||
CompileBlockPtr.ClassPtr = &FEXCore::Context::ContextImpl::CompileBlockJit;
|
||||
CompileBlockPtr.ClassPtr = &FEXCore::Context::Context::CompileBlockJit;
|
||||
return CompileBlockPtr.Data;
|
||||
}
|
||||
|
||||
|
||||
+22
-24
@@ -1,13 +1,14 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <stack>
|
||||
#include <tuple>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore {
|
||||
struct GuestSigAction;
|
||||
@@ -19,7 +20,7 @@ struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -43,7 +44,6 @@ public:
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA{};
|
||||
uint64_t ExitFunctionLinkerAddress{};
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t SignalHandlerReturnAddressRT{};
|
||||
uint64_t GuestSignal_SIGILL{};
|
||||
uint64_t GuestSignal_SIGTRAP{};
|
||||
uint64_t GuestSignal_SIGSEGV{};
|
||||
@@ -56,6 +56,14 @@ public:
|
||||
uint64_t Start{};
|
||||
uint64_t End{};
|
||||
|
||||
bool HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
|
||||
bool IsAddressInDispatcher(uint64_t Address) const {
|
||||
return Address >= Start && Address < End;
|
||||
}
|
||||
|
||||
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
// These are across all arches for now
|
||||
@@ -65,8 +73,8 @@ public:
|
||||
virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0;
|
||||
virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0;
|
||||
|
||||
static fextl::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
|
||||
static fextl::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
|
||||
virtual void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
DispatchPtr(Frame);
|
||||
@@ -76,32 +84,22 @@ public:
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
|
||||
virtual uint16_t GetSRAGPRCount() const {
|
||||
return 0U;
|
||||
}
|
||||
|
||||
virtual uint16_t GetSRAFPRCount() const {
|
||||
return 0U;
|
||||
}
|
||||
|
||||
virtual void GetSRAGPRMapping(uint8_t Mapping[16]) const {
|
||||
}
|
||||
|
||||
virtual void GetSRAFPRMapping(uint8_t Mapping[16]) const {
|
||||
}
|
||||
|
||||
const DispatcherConfig& GetConfig() const { return config; }
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config)
|
||||
Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &Config)
|
||||
: CTX {ctx}
|
||||
, config {Config}
|
||||
{}
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext);
|
||||
void RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext);
|
||||
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
|
||||
|
||||
virtual void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {}
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
DispatcherConfig config;
|
||||
|
||||
static void SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame);
|
||||
static void SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuStateFrame *Frame);
|
||||
|
||||
static uint64_t GetCompileBlockPtr();
|
||||
|
||||
|
||||
@@ -1,4 +1,3 @@
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
@@ -12,14 +11,14 @@
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/mman.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
[STATE + offsetof(FEXCore::Core::STATE_TYPE, FIELD)]
|
||||
@@ -28,10 +27,10 @@ namespace FEXCore::CPU {
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE r14
|
||||
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: Dispatcher(ctx, config)
|
||||
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
|
||||
FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true),
|
||||
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
|
||||
nullptr) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "X86 dispatcher does not support SRA");
|
||||
@@ -170,11 +169,36 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
// Block creation
|
||||
{
|
||||
L(NoBlock);
|
||||
|
||||
inc(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, reinterpret_cast<uint64_t>(CTX));
|
||||
@@ -183,15 +207,24 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
call(rax);
|
||||
|
||||
dec(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
Label AfterStore;
|
||||
// Skip the deferred fault address if the refcount isn't zero
|
||||
jne(AfterStore);
|
||||
mov(rax, qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)]);
|
||||
mov(rax, qword [rax]);
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
L(AfterStore);
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
|
||||
// rdx already contains RIP here
|
||||
jmp(LoopTop);
|
||||
@@ -199,7 +232,31 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = getCurr<uint64_t>();
|
||||
inc(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rax, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi}
|
||||
mov(rdi, STATE);
|
||||
@@ -207,17 +264,27 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
|
||||
dec(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
Label AfterStore;
|
||||
// Skip the deferred fault address if the refcount isn't zero
|
||||
jne(AfterStore);
|
||||
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)]);
|
||||
mov(qword [rbx], rbx);
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
L(AfterStore);
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
jmp(rax);
|
||||
jmp(r9);
|
||||
}
|
||||
else {
|
||||
jmp(rax);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
@@ -277,12 +344,6 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// RT Signal return handler
|
||||
SignalHandlerReturnAddressRT = getCurr<uint64_t>();
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
@@ -340,7 +401,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
End = Start + getSize();
|
||||
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
fextl::string Name = fextl::fmt::format("Dispatch_{}", FHU::Syscalls::gettid());
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(Start), End-Start, Name);
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
@@ -349,13 +410,15 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
}
|
||||
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline
|
||||
static thread_local Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
|
||||
size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
emit.setNewBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
|
||||
Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
@@ -364,7 +427,7 @@ size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestR
|
||||
emit.mov(rax, reinterpret_cast<uint64_t>(CTX));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cmp(dword [rax + (offsetof(FEXCore::Context::ContextImpl, Config.RunningMode))], 0);
|
||||
emit.cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
emit.je(RunBlock);
|
||||
{
|
||||
// Make sure RIP is syncronized to the context
|
||||
@@ -388,11 +451,10 @@ size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
emit.setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
Label InlineIRData;
|
||||
|
||||
|
||||
emit.mov(rdi, STATE);
|
||||
emit.lea(rsi, ptr[rip + InlineIRData]);
|
||||
emit.call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
@@ -407,7 +469,7 @@ size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
}
|
||||
|
||||
X86Dispatcher::~X86Dispatcher() {
|
||||
FEXCore::Allocator::VirtualFree(top_, MAX_DISPATCHER_CODE_SIZE);
|
||||
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
|
||||
}
|
||||
|
||||
void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
@@ -424,15 +486,14 @@ void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Threa
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
|
||||
|
||||
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
|
||||
(uintptr_t&)Interpreter.CallbackReturn = IntCallbackReturnAddress;
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
|
||||
return fextl::make_unique<X86Dispatcher>(CTX, Config);
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<X86Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,24 +1,13 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/unordered_set.h>
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#define XBYAK64
|
||||
#define XBYAK_CUSTOM_ALLOC
|
||||
#define XBYAK_CUSTOM_MALLOC FEXCore::Allocator::malloc
|
||||
#define XBYAK_CUSTOM_FREE FEXCore::Allocator::free
|
||||
#define XBYAK_CUSTOM_SETS
|
||||
#define XBYAK_STD_UNORDERED_SET fextl::unordered_set
|
||||
#define XBYAK_STD_UNORDERED_MAP fextl::unordered_map
|
||||
#define XBYAK_STD_UNORDERED_MULTIMAP fextl::unordered_multimap
|
||||
#define XBYAK_STD_LIST fextl::list
|
||||
#define XBYAK_NO_EXCEPTION
|
||||
|
||||
#include <xbyak/xbyak.h>
|
||||
#include <xbyak/xbyak_util.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
@@ -28,7 +17,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config);
|
||||
X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
+258
-157
@@ -7,7 +7,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
@@ -15,13 +14,15 @@ $end_info$
|
||||
#include <cstring>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
#include <sys/mman.h>
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
#include "Interface/Core/VSyscall/VSyscall.inc"
|
||||
@@ -31,6 +32,26 @@ using namespace FEXCore::X86Tables;
|
||||
static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool HasREX, bool HasXMM, bool HasMM, uint8_t InvalidOffset = 16) {
|
||||
using GPRArray = std::array<uint32_t, 16>;
|
||||
|
||||
static constexpr GPRArray GPRIndexes = {
|
||||
// Classical ordering?
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RCX,
|
||||
FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_RBX,
|
||||
FEXCore::X86State::REG_RSP,
|
||||
FEXCore::X86State::REG_RBP,
|
||||
FEXCore::X86State::REG_RSI,
|
||||
FEXCore::X86State::REG_RDI,
|
||||
FEXCore::X86State::REG_R8,
|
||||
FEXCore::X86State::REG_R9,
|
||||
FEXCore::X86State::REG_R10,
|
||||
FEXCore::X86State::REG_R11,
|
||||
FEXCore::X86State::REG_R12,
|
||||
FEXCore::X86State::REG_R13,
|
||||
FEXCore::X86State::REG_R14,
|
||||
FEXCore::X86State::REG_R15,
|
||||
};
|
||||
|
||||
static constexpr GPRArray GPR8BitHighIndexes = {
|
||||
// Classical ordering?
|
||||
FEXCore::X86State::REG_RAX,
|
||||
@@ -51,34 +72,112 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
FEXCore::X86State::REG_R15,
|
||||
};
|
||||
|
||||
static constexpr GPRArray XMMIndexes = {
|
||||
FEXCore::X86State::REG_XMM_0,
|
||||
FEXCore::X86State::REG_XMM_1,
|
||||
FEXCore::X86State::REG_XMM_2,
|
||||
FEXCore::X86State::REG_XMM_3,
|
||||
FEXCore::X86State::REG_XMM_4,
|
||||
FEXCore::X86State::REG_XMM_5,
|
||||
FEXCore::X86State::REG_XMM_6,
|
||||
FEXCore::X86State::REG_XMM_7,
|
||||
FEXCore::X86State::REG_XMM_8,
|
||||
FEXCore::X86State::REG_XMM_9,
|
||||
FEXCore::X86State::REG_XMM_10,
|
||||
FEXCore::X86State::REG_XMM_11,
|
||||
FEXCore::X86State::REG_XMM_12,
|
||||
FEXCore::X86State::REG_XMM_13,
|
||||
FEXCore::X86State::REG_XMM_14,
|
||||
FEXCore::X86State::REG_XMM_15,
|
||||
};
|
||||
|
||||
static constexpr GPRArray MMIndexes = {
|
||||
FEXCore::X86State::REG_MM_0,
|
||||
FEXCore::X86State::REG_MM_1,
|
||||
FEXCore::X86State::REG_MM_2,
|
||||
FEXCore::X86State::REG_MM_3,
|
||||
FEXCore::X86State::REG_MM_4,
|
||||
FEXCore::X86State::REG_MM_5,
|
||||
FEXCore::X86State::REG_MM_6,
|
||||
FEXCore::X86State::REG_MM_7,
|
||||
FEXCore::X86State::REG_INVALID,
|
||||
FEXCore::X86State::REG_INVALID,
|
||||
FEXCore::X86State::REG_INVALID,
|
||||
FEXCore::X86State::REG_INVALID,
|
||||
FEXCore::X86State::REG_INVALID,
|
||||
FEXCore::X86State::REG_INVALID,
|
||||
FEXCore::X86State::REG_INVALID,
|
||||
FEXCore::X86State::REG_INVALID
|
||||
};
|
||||
|
||||
const GPRArray *GPRs = &GPRIndexes;
|
||||
if (HasXMM) {
|
||||
GPRs = &XMMIndexes;
|
||||
}
|
||||
else if (HasMM) {
|
||||
GPRs = &MMIndexes;
|
||||
}
|
||||
else if (HighBits && !HasREX) {
|
||||
GPRs = &GPR8BitHighIndexes;
|
||||
}
|
||||
|
||||
uint8_t Offset = (REX << 3) | bits;
|
||||
|
||||
if (Offset == InvalidOffset) {
|
||||
return FEXCore::X86State::REG_INVALID;
|
||||
}
|
||||
|
||||
if (HasXMM) {
|
||||
return FEXCore::X86State::REG_XMM_0 + Offset;
|
||||
}
|
||||
else if (HasMM) {
|
||||
return FEXCore::X86State::REG_MM_0 + Offset;
|
||||
}
|
||||
else if (!(HighBits && !HasREX)) {
|
||||
return FEXCore::X86State::REG_RAX + Offset;
|
||||
}
|
||||
|
||||
return GPR8BitHighIndexes[Offset];
|
||||
return (*GPRs)[(REX << 3) | bits];
|
||||
}
|
||||
|
||||
static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
|
||||
using GPRArray = std::array<uint32_t, 16>;
|
||||
|
||||
static constexpr GPRArray GPRIndexes = {
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RCX,
|
||||
FEXCore::X86State::REG_RDX,
|
||||
FEXCore::X86State::REG_RBX,
|
||||
FEXCore::X86State::REG_RSP,
|
||||
FEXCore::X86State::REG_RBP,
|
||||
FEXCore::X86State::REG_RSI,
|
||||
FEXCore::X86State::REG_RDI,
|
||||
FEXCore::X86State::REG_R8,
|
||||
FEXCore::X86State::REG_R9,
|
||||
FEXCore::X86State::REG_R10,
|
||||
FEXCore::X86State::REG_R11,
|
||||
FEXCore::X86State::REG_R12,
|
||||
FEXCore::X86State::REG_R13,
|
||||
FEXCore::X86State::REG_R14,
|
||||
FEXCore::X86State::REG_R15,
|
||||
};
|
||||
|
||||
static constexpr GPRArray XMMIndexes = {
|
||||
FEXCore::X86State::REG_XMM_0,
|
||||
FEXCore::X86State::REG_XMM_1,
|
||||
FEXCore::X86State::REG_XMM_2,
|
||||
FEXCore::X86State::REG_XMM_3,
|
||||
FEXCore::X86State::REG_XMM_4,
|
||||
FEXCore::X86State::REG_XMM_5,
|
||||
FEXCore::X86State::REG_XMM_6,
|
||||
FEXCore::X86State::REG_XMM_7,
|
||||
FEXCore::X86State::REG_XMM_8,
|
||||
FEXCore::X86State::REG_XMM_9,
|
||||
FEXCore::X86State::REG_XMM_10,
|
||||
FEXCore::X86State::REG_XMM_11,
|
||||
FEXCore::X86State::REG_XMM_12,
|
||||
FEXCore::X86State::REG_XMM_13,
|
||||
FEXCore::X86State::REG_XMM_14,
|
||||
FEXCore::X86State::REG_XMM_15,
|
||||
};
|
||||
|
||||
if (HasXMM) {
|
||||
return FEXCore::X86State::REG_XMM_0 + vvvv;
|
||||
return XMMIndexes[vvvv];
|
||||
} else {
|
||||
return FEXCore::X86State::REG_RAX + vvvv;
|
||||
return GPRIndexes[vvvv];
|
||||
}
|
||||
}
|
||||
|
||||
Decoder::Decoder(FEXCore::Context::ContextImpl *ctx)
|
||||
Decoder::Decoder(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx}
|
||||
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN }
|
||||
, PoolObject {ctx->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
@@ -107,7 +206,7 @@ uint64_t Decoder::ReadData(uint8_t Size) {
|
||||
uint64_t Res = 0;
|
||||
std::memcpy(&Res, &InstStream[InstructionSize], Size);
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
#ifndef NDEBUG
|
||||
for(size_t i = 0; i < Size; ++i) {
|
||||
ReadByte();
|
||||
}
|
||||
@@ -285,6 +384,12 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
// XXX: Once we support 32bit x86 then this will be necessary to support
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
|
||||
LogMan::Msg::DFmt("Legacy Prefix");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
@@ -303,35 +408,27 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
(Options.w && CTX->Config.Is64BitMode);
|
||||
const bool HasNarrowingDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST) != 0;
|
||||
|
||||
const bool HasXMMFlags = (Info->Flags & InstFlags::FLAGS_XMM_FLAGS) != 0;
|
||||
bool HasXMMSrc = HasXMMFlags &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_MMX_SRC);
|
||||
bool HasXMMDst = HasXMMFlags &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_DST_GPR) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_MMX_DST);
|
||||
bool HasMMSrc = HasXMMFlags &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_MMX_SRC);
|
||||
bool HasMMDst = HasXMMFlags &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_DST_GPR) &&
|
||||
HAS_XMM_SUBFLAG(Info->Flags, InstFlags::FLAGS_SF_MMX_DST);
|
||||
bool HasXMMSrc = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_MMX_SRC);
|
||||
bool HasXMMDst = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_GPR) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_MMX_DST);
|
||||
bool HasMMSrc = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_GPR) &&
|
||||
HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_MMX_SRC);
|
||||
bool HasMMDst = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
|
||||
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_GPR) &&
|
||||
HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_MMX_DST);
|
||||
|
||||
// Is ModRM present via explicit instruction encoded or REX?
|
||||
const bool HasMODRM = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM);
|
||||
|
||||
const bool HasREX = !!(DecodeInst->Flags & DecodeFlags::FLAG_REX_PREFIX);
|
||||
const bool HasHighXMM = HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_HIGH_XMM_REG);
|
||||
const bool Has16BitAddressing = !CTX->Config.Is64BitMode &&
|
||||
DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
|
||||
|
||||
// This is used for ModRM register modification
|
||||
// For both modrm.reg and modrm.rm(when mod == 0b11) when value is >= 0b100
|
||||
// then it changes from expected registers to the high 8bits of the lower registers
|
||||
// Bit annoying to support
|
||||
// In the case of no modrm (REX in byte situation) then it is unaffected
|
||||
bool Is8BitSrc{};
|
||||
bool Is8BitDest{};
|
||||
|
||||
// If we require ModRM and haven't decoded it yet, do it now
|
||||
// Some instructions have to read modrm upfront, others do it later
|
||||
if (HasMODRM && !DecodeInst->DecodedModRM) {
|
||||
@@ -348,7 +445,6 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_8BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_8BIT);
|
||||
DestSize = 1;
|
||||
Is8BitDest = true;
|
||||
}
|
||||
else if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_16BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_16BIT);
|
||||
@@ -391,7 +487,6 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
// Decode sources
|
||||
if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_8BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_8BIT);
|
||||
Is8BitSrc = true;
|
||||
}
|
||||
else if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_16BIT) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
|
||||
@@ -425,6 +520,14 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
}
|
||||
}
|
||||
|
||||
// This is used for ModRM register modification
|
||||
// For both modrm.reg and modrm.rm(when mod == 0b11) when value is >= 0b100
|
||||
// then it changes from expected registers to the high 8bits of the lower registers
|
||||
// Bit annoying to support
|
||||
// In the case of no modrm (REX in byte situation) then it is unaffected
|
||||
const bool Is8BitSrc = (DecodeFlags::GetSizeSrcFlags(DecodeInst->Flags) == DecodeFlags::SIZE_8BIT);
|
||||
const bool Is8BitDest = (DecodeFlags::GetSizeDstFlags(DecodeInst->Flags) == DecodeFlags::SIZE_8BIT);
|
||||
|
||||
auto *CurrentDest = &DecodeInst->Dest;
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ||
|
||||
@@ -435,7 +538,8 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
CurrentDest->Data.GPR.GPR = HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
|
||||
CurrentDest = &DecodeInst->Src[0];
|
||||
}
|
||||
else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_AA_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
@@ -443,7 +547,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
// ADDITIONALLY:
|
||||
// If there is a REX prefix then that allows extended GPR usage
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Dest.Data.GPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100);
|
||||
DecodeInst->Dest.Data.GPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100) || HasHighXMM;
|
||||
CurrentDest->Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
|
||||
|
||||
if (CurrentDest->Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
@@ -471,7 +575,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
|
||||
// Decode the GPR source first
|
||||
GPR.Type = DecodedOperand::OpType::GPR;
|
||||
GPR.Data.GPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX);
|
||||
GPR.Data.GPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX) || HasHighXMM;
|
||||
GPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_R ? 1 : 0, ModRM.reg, GPR8Bit, HasREX, HasXMMGPR, HasMMGPR);
|
||||
|
||||
if (GPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
@@ -481,7 +585,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
// ModRM.Mod != 0b11 == Register-direct addressing
|
||||
if (ModRM.mod == 0b11) {
|
||||
NonGPR.Type = DecodedOperand::OpType::GPR;
|
||||
NonGPR.Data.GPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX);
|
||||
NonGPR.Data.GPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX) || HasHighXMM;
|
||||
NonGPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, NonGPR8Bit, HasREX, HasXMMNonGPR, HasMMNonGPR);
|
||||
if (NonGPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
return false;
|
||||
@@ -502,12 +606,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) != 0) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
|
||||
// If we have XMM flags at all, then SRC 1 cannot be a GPR. The only case where
|
||||
// this is possible is with BMI1 and BMI2 instructions (which are all GPR-based
|
||||
// and don't use XMM flags)
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMFlags);
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
@@ -585,6 +684,12 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
// XXX: Once we support 32bit x86 then this will be necessary to support
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
|
||||
LogMan::Msg::DFmt("Legacy Prefix");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
@@ -598,11 +703,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
LOGMAN_THROW_AA_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
|
||||
"REX PREFIX should have been decoded before this!");
|
||||
|
||||
// A normal instruction is the most likely.
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INST) [[likely]] {
|
||||
return NormalOp(Info, Op);
|
||||
}
|
||||
else if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
|
||||
if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
|
||||
Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
@@ -750,8 +851,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
return NormalOp(&EVEXTableOps[EVEXOp], EVEXOp);
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Invalid instruction decoding type");
|
||||
FEX_UNREACHABLE;
|
||||
return NormalOp(Info, Op);
|
||||
}
|
||||
|
||||
bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
@@ -770,106 +870,105 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
case 0x0F: {// Escape Op
|
||||
uint8_t EscapeOp = ReadByte();
|
||||
switch (EscapeOp) {
|
||||
case 0x0F: [[unlikely]] { // 3DNow!
|
||||
// 3DNow! Instruction Encoding: 0F 0F [ModRM] [SIB] [Displacement] [Opcode]
|
||||
// Decode ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
case 0x0F: [[unlikely]] { // 3DNow!
|
||||
// 3DNow! Instruction Encoding: 0F 0F [ModRM] [SIB] [Displacement] [Opcode]
|
||||
// Decode ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
|
||||
const bool Has16BitAddressing = !CTX->Config.Is64BitMode &&
|
||||
DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
|
||||
const bool Has16BitAddressing = !CTX->Config.Is64BitMode &&
|
||||
DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
|
||||
|
||||
// All 3DNow! instructions have the second argument as the rm handler
|
||||
// We need to decode it upfront to get the displacement out of the way
|
||||
if (ModRM.mod != 0b11) {
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&DecodeInst->Src[0], ModRM);
|
||||
}
|
||||
|
||||
// Take a peek at the op just past the displacement
|
||||
uint8_t LocalOp = ReadByte();
|
||||
return NormalOpHeader(&FEXCore::X86Tables::DDDNowOps[LocalOp], LocalOp);
|
||||
break;
|
||||
// All 3DNow! instructions have the second argument as the rm handler
|
||||
// We need to decode it upfront to get the displacement out of the way
|
||||
if (ModRM.mod != 0b11) {
|
||||
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
|
||||
(this->*Disp)(&DecodeInst->Src[0], ModRM);
|
||||
}
|
||||
case 0x38: { // F38 Table!
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
uint16_t Prefix = PF_38_NONE;
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_OPERAND_SIZE) {
|
||||
Prefix |= PF_38_66;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REPNE_PREFIX) {
|
||||
Prefix |= PF_38_F2;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REP_PREFIX) {
|
||||
Prefix |= PF_38_F3;
|
||||
}
|
||||
// Take a peek at the op just past the displacement
|
||||
uint8_t LocalOp = ReadByte();
|
||||
return NormalOpHeader(&FEXCore::X86Tables::DDDNowOps[LocalOp], LocalOp);
|
||||
break;
|
||||
}
|
||||
case 0x38: { // F38 Table!
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
return NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp);
|
||||
break;
|
||||
uint16_t Prefix = PF_38_NONE;
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_OPERAND_SIZE) {
|
||||
Prefix |= PF_38_66;
|
||||
}
|
||||
case 0x3A: { // F3A Table!
|
||||
constexpr uint16_t PF_3A_NONE = 0;
|
||||
constexpr uint16_t PF_3A_66 = (1 << 0);
|
||||
constexpr uint16_t PF_3A_REX = (1 << 1);
|
||||
|
||||
uint16_t Prefix = PF_3A_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0x66) // Operand Size
|
||||
Prefix = PF_3A_66;
|
||||
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REX_WIDENING)
|
||||
Prefix |= PF_3A_REX;
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
return NormalOpHeader(&FEXCore::X86Tables::H0F3ATableOps[LocalOp], LocalOp);
|
||||
break;
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REPNE_PREFIX) {
|
||||
Prefix |= PF_38_F2;
|
||||
}
|
||||
default: [[likely]] { // Two byte table!
|
||||
// x86-64 abuses three legacy prefixes to extend the table encodings
|
||||
// 0x66 - Operand Size prefix
|
||||
// 0xF2 - REPNE prefix
|
||||
// 0xF3 - REP prefix
|
||||
// If any of these three prefixes are used then it falls down the subtable
|
||||
// Additionally: If you hit repeat of differnt prefixes then only the LAST one before this one works for subtable selection
|
||||
|
||||
bool NoOverlay = (FEXCore::X86Tables::SecondBaseOps[EscapeOp].Flags & InstFlags::FLAGS_NO_OVERLAY) != 0;
|
||||
bool NoOverlay66 = (FEXCore::X86Tables::SecondBaseOps[EscapeOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
|
||||
|
||||
if (NoOverlay) { // This section of the table ignores prefix extention
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
DecodeFlags::PopOpAddrIf(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
|
||||
return NormalOpHeader(&FEXCore::X86Tables::OpSizeModOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else {
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
break;
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REP_PREFIX) {
|
||||
Prefix |= PF_38_F3;
|
||||
}
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
return NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp);
|
||||
break;
|
||||
}
|
||||
case 0x3A: { // F3A Table!
|
||||
constexpr uint16_t PF_3A_NONE = 0;
|
||||
constexpr uint16_t PF_3A_66 = (1 << 0);
|
||||
constexpr uint16_t PF_3A_REX = (1 << 1);
|
||||
|
||||
uint16_t Prefix = PF_3A_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0x66) // Operand Size
|
||||
Prefix = PF_3A_66;
|
||||
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REX_WIDENING)
|
||||
Prefix |= PF_3A_REX;
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
return NormalOpHeader(&FEXCore::X86Tables::H0F3ATableOps[LocalOp], LocalOp);
|
||||
break;
|
||||
}
|
||||
default: // Two byte table!
|
||||
// x86-64 abuses three legacy prefixes to extend the table encodings
|
||||
// 0x66 - Operand Size prefix
|
||||
// 0xF2 - REPNE prefix
|
||||
// 0xF3 - REP prefix
|
||||
// If any of these three prefixes are used then it falls down the subtable
|
||||
// Additionally: If you hit repeat of differnt prefixes then only the LAST one before this one works for subtable selection
|
||||
|
||||
bool NoOverlay = (FEXCore::X86Tables::SecondBaseOps[EscapeOp].Flags & InstFlags::FLAGS_NO_OVERLAY) != 0;
|
||||
bool NoOverlay66 = (FEXCore::X86Tables::SecondBaseOps[EscapeOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
|
||||
|
||||
if (NoOverlay) { // This section of the table ignores prefix extention
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
DecodeFlags::PopOpAddrIf(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
|
||||
return NormalOpHeader(&FEXCore::X86Tables::OpSizeModOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
else {
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -922,7 +1021,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
case 0x65: // GS prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_GS_PREFIX;
|
||||
break;
|
||||
default: [[likely]] { // Default base table
|
||||
default: { // Default base table
|
||||
auto Info = &FEXCore::X86Tables::BaseOps[Op];
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
|
||||
@@ -1113,7 +1212,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
|
||||
uint64_t CurrentCodePage = PC & FHU::FEX_PAGE_MASK;
|
||||
|
||||
fextl::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
@@ -1141,19 +1240,24 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
auto OpMinPage = OpMinAddress & FHU::FEX_PAGE_MASK;
|
||||
auto OpMaxPage = OpMaxAddress & FHU::FEX_PAGE_MASK;
|
||||
|
||||
|
||||
if (OpMinPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMinPage;
|
||||
CodePages.insert(CurrentCodePage);
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpMaxPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMaxPage;
|
||||
CodePages.insert(CurrentCodePage);
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) [[unlikely]] {
|
||||
if (ErrorDuringDecoding) {
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", RIPToDecode + PCOffset, PC);
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
@@ -1210,9 +1314,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
}
|
||||
|
||||
for (auto CodePage : CodePages) {
|
||||
AddContainedCodePage(PC, CodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
|
||||
// sort for better branching
|
||||
std::sort(Blocks.begin(), Blocks.end(), [](const FEXCore::Frontend::Decoder::DecodedBlocks& a, const FEXCore::Frontend::Decoder::DecodedBlocks& b) {
|
||||
|
||||
+12
-13
@@ -1,18 +1,17 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <set>
|
||||
#include <stddef.h>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
@@ -26,11 +25,11 @@ public:
|
||||
bool HasInvalidInstruction{};
|
||||
};
|
||||
|
||||
Decoder(FEXCore::Context::ContextImpl *ctx);
|
||||
Decoder(FEXCore::Context::Context *ctx);
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
|
||||
fextl::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
return &Blocks;
|
||||
}
|
||||
|
||||
@@ -38,7 +37,7 @@ public:
|
||||
uint64_t DecodedMaxAddress {~0ULL};
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
|
||||
void SetExternalBranches(fextl::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
@@ -53,7 +52,7 @@ private:
|
||||
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
|
||||
};
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
FEXCore::Context::Context *CTX;
|
||||
const FEXCore::HLE::SyscallOSABI OSABI{};
|
||||
|
||||
bool DecodeInstruction(uint64_t PC);
|
||||
@@ -90,10 +89,10 @@ private:
|
||||
uint64_t SymbolMinAddress {~0ULL};
|
||||
uint64_t SectionMaxAddress {~0ULL};
|
||||
|
||||
fextl::vector<DecodedBlocks> Blocks;
|
||||
fextl::set<uint64_t> BlocksToDecode;
|
||||
fextl::set<uint64_t> HasBlocks;
|
||||
fextl::set<uint64_t> *ExternalBranches {nullptr};
|
||||
std::vector<DecodedBlocks> Blocks;
|
||||
std::set<uint64_t> BlocksToDecode;
|
||||
std::set<uint64_t> HasBlocks;
|
||||
std::set<uint64_t> *ExternalBranches {nullptr};
|
||||
|
||||
// ModRM rm decoding
|
||||
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
|
||||
+119
-125
@@ -8,8 +8,11 @@ $end_info$
|
||||
#include <cstdlib>
|
||||
#include <cstdio>
|
||||
#include <iomanip>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <vector>
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
@@ -24,45 +27,39 @@ $end_info$
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/NetStream.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstring>
|
||||
#ifndef _WIN32
|
||||
#include <elf.h>
|
||||
#include <netdb.h>
|
||||
#include <sys/socket.h>
|
||||
#endif
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <fstream>
|
||||
#include <fmt/format.h>
|
||||
#include <netdb.h>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <string_view>
|
||||
#include <sys/socket.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "GdbServer.h"
|
||||
|
||||
namespace FEXCore
|
||||
{
|
||||
#ifndef _WIN32
|
||||
|
||||
void GdbServer::Break(int signal) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
if (!CommsStream) {
|
||||
return;
|
||||
}
|
||||
|
||||
const fextl::string str = fextl::fmt::format("S{:02x}", signal);
|
||||
const auto str = fmt::format("S{:02x}", signal);
|
||||
SendPacket(*CommsStream, str);
|
||||
}
|
||||
|
||||
@@ -71,11 +68,11 @@ void GdbServer::WaitForThreadWakeup() {
|
||||
ThreadBreakEvent.Wait();
|
||||
}
|
||||
|
||||
GdbServer::GdbServer(FEXCore::Context::ContextImpl *ctx) : CTX(ctx) {
|
||||
GdbServer::GdbServer(FEXCore::Context::Context *ctx) : CTX(ctx) {
|
||||
// Pass all signals by default
|
||||
std::fill(PassSignals.begin(), PassSignals.end(), true);
|
||||
|
||||
ctx->SetExitHandler([this](uint64_t ThreadId, FEXCore::Context::ExitReason ExitReason) {
|
||||
Context::SetExitHandler(ctx, [this](uint64_t ThreadId, FEXCore::Context::ExitReason ExitReason) {
|
||||
if (ExitReason == FEXCore::Context::ExitReason::EXIT_DEBUG) {
|
||||
this->Break(SIGTRAP);
|
||||
}
|
||||
@@ -104,7 +101,7 @@ GdbServer::GdbServer(FEXCore::Context::ContextImpl *ctx) : CTX(ctx) {
|
||||
StartThread();
|
||||
}
|
||||
|
||||
static int calculateChecksum(const fextl::string &packet) {
|
||||
static int calculateChecksum(const std::string &packet) {
|
||||
unsigned char checksum = 0;
|
||||
for (const char &c : packet) {
|
||||
checksum += c;
|
||||
@@ -112,8 +109,8 @@ static int calculateChecksum(const fextl::string &packet) {
|
||||
return checksum;
|
||||
}
|
||||
|
||||
static fextl::string hexstring(fextl::istringstream &ss, int delm) {
|
||||
fextl::string ret;
|
||||
static std::string hexstring(std::istringstream &ss, int delm) {
|
||||
std::string ret;
|
||||
|
||||
char hexString[3] = {0, 0, 0};
|
||||
while (ss.peek() != delm) {
|
||||
@@ -128,8 +125,8 @@ static fextl::string hexstring(fextl::istringstream &ss, int delm) {
|
||||
return ret;
|
||||
}
|
||||
|
||||
static fextl::string encodeHex(const unsigned char *data, size_t length) {
|
||||
fextl::ostringstream ss;
|
||||
static std::string encodeHex(const unsigned char *data, size_t length) {
|
||||
std::ostringstream ss;
|
||||
|
||||
for (size_t i=0; i < length; i++) {
|
||||
ss << std::setfill('0') << std::setw(2) << std::hex << int(data[i]);
|
||||
@@ -137,19 +134,26 @@ static fextl::string encodeHex(const unsigned char *data, size_t length) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
static fextl::string getThreadName(uint32_t ThreadID) {
|
||||
const auto ThreadFile = fextl::fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
|
||||
fextl::string ThreadName {"<No Name>"};
|
||||
FEXCore::FileLoading::LoadFile(ThreadName, ThreadFile);
|
||||
return ThreadName;
|
||||
static std::string getThreadName(uint32_t ThreadID) {
|
||||
const auto ThreadFile = fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
|
||||
std::fstream fs(ThreadFile, std::fstream::in | std::fstream::binary);
|
||||
|
||||
if (fs.is_open()) {
|
||||
std::string ThreadName;
|
||||
fs >> ThreadName;
|
||||
fs.close();
|
||||
return ThreadName;
|
||||
}
|
||||
|
||||
return "<No Name>";
|
||||
}
|
||||
|
||||
// Packet parser
|
||||
// Takes a serial stream and reads a single packet
|
||||
// Un-escapes chars, checks the checksum and request a retransmit if it fails.
|
||||
// Once the checksum is validated, it acknowledges and returns the packet in a string
|
||||
fextl::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
fextl::string packet{};
|
||||
std::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
std::string packet{};
|
||||
|
||||
// The GDB "Remote Serial Protocal" was originally 7bit clean for use on serial ports.
|
||||
// Binary data is useally hex encoded. However some later extentions just put
|
||||
@@ -168,7 +172,7 @@ fextl::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
LogMan::Msg::EFmt("Dropping unexpected data: \"{}\"", packet);
|
||||
|
||||
// clear any existing data, must have been a mistake.
|
||||
packet = fextl::string();
|
||||
packet = std::string();
|
||||
break;
|
||||
case '}': // escape char
|
||||
{
|
||||
@@ -199,8 +203,8 @@ fextl::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
return "";
|
||||
}
|
||||
|
||||
static fextl::string escapePacket(const fextl::string& packet) {
|
||||
fextl::ostringstream ss;
|
||||
static std::string escapePacket(const std::string& packet) {
|
||||
std::ostringstream ss;
|
||||
|
||||
for(const auto &c : packet) {
|
||||
switch (c) {
|
||||
@@ -221,9 +225,9 @@ static fextl::string escapePacket(const fextl::string& packet) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
void GdbServer::SendPacket(std::ostream &stream, const fextl::string& packet) {
|
||||
void GdbServer::SendPacket(std::ostream &stream, const std::string& packet) {
|
||||
const auto escaped = escapePacket(packet);
|
||||
const auto str = fextl::fmt::format("${}#{:02x}", escaped, calculateChecksum(escaped));
|
||||
const auto str = fmt::format("${}#{:02x}", escaped, calculateChecksum(escaped));
|
||||
|
||||
stream << str << std::flush;
|
||||
}
|
||||
@@ -259,7 +263,7 @@ struct FEX_PACKED GDBContextDefinition {
|
||||
uint32_t mxcsr;
|
||||
};
|
||||
|
||||
fextl::string GdbServer::readRegs() {
|
||||
std::string GdbServer::readRegs() {
|
||||
GDBContextDefinition GDB{};
|
||||
FEXCore::Core::CPUState state{};
|
||||
|
||||
@@ -307,11 +311,11 @@ fextl::string GdbServer::readRegs() {
|
||||
return encodeHex((unsigned char *)&GDB, sizeof(GDBContextDefinition));
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
size_t addr;
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.get(); // Drop first letter
|
||||
ss >> std::hex >> addr;
|
||||
GdbServer::HandledPacketType GdbServer::readReg(const std::string& packet) {
|
||||
size_t addr;
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.get(); // Drop first letter
|
||||
ss >> std::hex >> addr;
|
||||
|
||||
FEXCore::Core::CPUState state{};
|
||||
|
||||
@@ -391,8 +395,8 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
fextl::string buildTargetXML() {
|
||||
fextl::ostringstream xml;
|
||||
std::string buildTargetXML() {
|
||||
std::ostringstream xml;
|
||||
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
xml << "<!DOCTYPE target SYSTEM 'gdb-target.dtd'>\n";
|
||||
@@ -444,7 +448,7 @@ fextl::string buildTargetXML() {
|
||||
|
||||
// x87 stack
|
||||
for (int i=0; i < 8; i++) {
|
||||
reg(fextl::fmt::format("st{}", i), "i387_ext", 80);
|
||||
reg("st" + std::to_string(i), "i387_ext", 80);
|
||||
}
|
||||
|
||||
// x87 control
|
||||
@@ -480,7 +484,7 @@ fextl::string buildTargetXML() {
|
||||
|
||||
// SSE regs
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
reg(fextl::fmt::format("xmm{}", i), "vec128", 128);
|
||||
reg("xmm" + std::to_string(i), "vec128", 128);
|
||||
}
|
||||
|
||||
reg("mxcsr", "int", 32);
|
||||
@@ -516,8 +520,8 @@ fextl::string buildTargetXML() {
|
||||
return xml.str();
|
||||
}
|
||||
|
||||
fextl::string buildOSData() {
|
||||
fextl::ostringstream xml;
|
||||
std::string buildOSData() {
|
||||
std::ostringstream xml;
|
||||
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
|
||||
@@ -537,27 +541,24 @@ void GdbServer::buildLibraryMap() {
|
||||
return;
|
||||
}
|
||||
|
||||
fextl::ostringstream xml;
|
||||
std::ostringstream xml;
|
||||
|
||||
fextl::string MapsFile;
|
||||
FEXCore::FileLoading::LoadFile(MapsFile, "/proc/self/maps");
|
||||
fextl::istringstream MapsStream(MapsFile);
|
||||
|
||||
fextl::string Line;
|
||||
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::string Line;
|
||||
|
||||
struct FileData {
|
||||
uint64_t Begin;
|
||||
};
|
||||
|
||||
fextl::map<fextl::string, fextl::vector<FileData>> SegmentMaps;
|
||||
std::map<std::string, std::vector<FileData>> SegmentMaps;
|
||||
|
||||
// 7ff5dd6d2000-7ff5dd6d3000 rw-p 0000a000 103:0b 1881447 /usr/lib/x86_64-linux-gnu/libnss_compat.so.2
|
||||
fextl::string const &RuntimeExecutable = Filename();
|
||||
while (std::getline(MapsStream, Line)) {
|
||||
auto ss = fextl::istringstream(Line);
|
||||
fextl::string Tmp;
|
||||
fextl::string Begin;
|
||||
fextl::string Name;
|
||||
std::string const &RuntimeExecutable = Filename();
|
||||
while (std::getline(fs, Line)) {
|
||||
auto ss = std::istringstream(Line);
|
||||
std::string Tmp;
|
||||
std::string Begin;
|
||||
std::string Name;
|
||||
std::getline(ss, Begin, '-');
|
||||
std::getline(ss, Tmp, ' '); // End
|
||||
std::getline(ss, Tmp, ' '); // Perm
|
||||
@@ -608,18 +609,18 @@ void GdbServer::buildLibraryMap() {
|
||||
LibraryMapChanged = false;
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet) {
|
||||
fextl::string object;
|
||||
fextl::string rw;
|
||||
fextl::string annex;
|
||||
GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
std::string object;
|
||||
std::string rw;
|
||||
std::string annex;
|
||||
int annex_pid;
|
||||
int offset;
|
||||
int length;
|
||||
|
||||
// Parse Xfer message
|
||||
{
|
||||
auto ss = fextl::istringstream(packet);
|
||||
fextl::string expectXfer;
|
||||
auto ss = std::istringstream(packet);
|
||||
std::string expectXfer;
|
||||
char expectComma;
|
||||
|
||||
std::getline(ss, expectXfer, ':');
|
||||
@@ -630,7 +631,7 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet)
|
||||
annex_pid = getpid();
|
||||
}
|
||||
else {
|
||||
auto ss_pid = fextl::istringstream(annex);
|
||||
auto ss_pid = std::istringstream(annex);
|
||||
ss_pid >> std::hex >> annex_pid;
|
||||
}
|
||||
ss >> std::hex >> offset;
|
||||
@@ -643,7 +644,7 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet)
|
||||
}
|
||||
|
||||
// Lambda to correctly encode any reply
|
||||
auto encode = [&](fextl::string data) -> fextl::string {
|
||||
auto encode = [&](std::string data) -> std::string {
|
||||
if (offset == data.size())
|
||||
return "l";
|
||||
if (offset >= data.size())
|
||||
@@ -673,7 +674,7 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet)
|
||||
auto Threads = CTX->GetThreads();
|
||||
|
||||
ThreadString.clear();
|
||||
fextl::ostringstream ss;
|
||||
std::ostringstream ss;
|
||||
ss << "<?xml version=\"1.0\"?>\n";
|
||||
ss << "<threads>\n";
|
||||
for (auto &Thread : *Threads) {
|
||||
@@ -709,7 +710,7 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet)
|
||||
auto CodeLoader = CTX->SyscallHandler->GetCodeLoader();
|
||||
uint64_t auxv_ptr, auxv_size;
|
||||
CodeLoader->GetAuxv(auxv_ptr, auxv_size);
|
||||
fextl::string data;
|
||||
std::string data;
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
data.resize(auxv_size);
|
||||
memcpy(data.data(), reinterpret_cast<void*>(auxv_ptr), data.size());
|
||||
@@ -735,14 +736,11 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet)
|
||||
|
||||
static size_t CheckMemMapping(uint64_t Address, size_t Size) {
|
||||
uint64_t AddressEnd = Address + Size;
|
||||
fextl::string MapsFile;
|
||||
FEXCore::FileLoading::LoadFile(MapsFile, "/proc/self/maps");
|
||||
fextl::istringstream MapsStream(MapsFile);
|
||||
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::string Line;
|
||||
|
||||
fextl::string Line;
|
||||
|
||||
while (std::getline(MapsStream, Line)) {
|
||||
if (MapsStream.eof()) break;
|
||||
while (std::getline(fs, Line)) {
|
||||
if (fs.eof()) break;
|
||||
uint64_t Begin, End;
|
||||
char r,w,x,p;
|
||||
if (sscanf(Line.c_str(), "%lx-%lx %c%c%c%c", &Begin, &End, &r, &w, &x, &p) == 6) {
|
||||
@@ -763,17 +761,17 @@ static size_t CheckMemMapping(uint64_t Address, size_t Size) {
|
||||
GdbServer::HandledPacketType GdbServer::handleProgramOffsets() {
|
||||
auto CodeLoader = CTX->SyscallHandler->GetCodeLoader();
|
||||
uint64_t BaseOffset = CodeLoader->GetBaseOffset();
|
||||
fextl::string str = fextl::fmt::format("Text={:x};Data={:x};Bss={:x}", BaseOffset, BaseOffset, BaseOffset);
|
||||
auto str = fmt::format("Text={:x};Data={:x};Bss={:x}", BaseOffset, BaseOffset, BaseOffset);
|
||||
return {std::move(str), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleMemory(const fextl::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::handleMemory(const std::string &packet) {
|
||||
bool write;
|
||||
size_t addr;
|
||||
size_t length;
|
||||
fextl::string data;
|
||||
std::string data;
|
||||
|
||||
auto ss = fextl::istringstream(packet);
|
||||
auto ss = std::istringstream(packet);
|
||||
write = ss.get() == 'M';
|
||||
ss >> std::hex >> addr;
|
||||
ss.get(); // discard comma
|
||||
@@ -808,22 +806,22 @@ GdbServer::HandledPacketType GdbServer::handleMemory(const fextl::string &packet
|
||||
}
|
||||
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
const auto MatchStr = [](const fextl::string &Str, const char *str) -> bool { return Str.rfind(str, 0) == 0; };
|
||||
const auto MatchStr = [](const std::string &Str, const char *str) -> bool { return Str.rfind(str, 0) == 0; };
|
||||
|
||||
const auto split = [](const fextl::string &Str, char deliminator) -> fextl::vector<fextl::string> {
|
||||
fextl::vector<fextl::string> Elements;
|
||||
fextl::istringstream Input(Str);
|
||||
for (fextl::string line;
|
||||
const auto split = [](const std::string &Str, char deliminator) -> std::vector<std::string> {
|
||||
std::vector<std::string> Elements;
|
||||
std::istringstream Input(Str);
|
||||
for (std::string line;
|
||||
std::getline(Input, line);
|
||||
Elements.emplace_back(line));
|
||||
return Elements;
|
||||
};
|
||||
|
||||
if (match("QNonStop:")) {
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("QNonStop:").size());
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("QNonStop:").size());
|
||||
ss.get(); // discard colon
|
||||
ss >> NonStopMode;
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
@@ -834,7 +832,7 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
|
||||
// For feature documentation
|
||||
// https://sourceware.org/gdb/current/onlinedocs/gdb/General-Query-Packets.html#qSupported
|
||||
fextl::string SupportedFeatures{};
|
||||
std::string SupportedFeatures{};
|
||||
|
||||
// Required features
|
||||
SupportedFeatures += "PacketSize=32768;";
|
||||
@@ -903,7 +901,7 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
if (match("qfThreadInfo")) {
|
||||
auto Threads = CTX->GetThreads();
|
||||
|
||||
fextl::ostringstream ss;
|
||||
std::ostringstream ss;
|
||||
ss << "m";
|
||||
for (size_t i = 0; i < Threads->size(); ++i) {
|
||||
auto Thread = Threads->at(i);
|
||||
@@ -918,8 +916,8 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
return {"l", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("qThreadExtraInfo")) {
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("qThreadExtraInfo").size());
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("qThreadExtraInfo").size());
|
||||
ss.get(); // discard comma
|
||||
uint32_t ThreadID;
|
||||
ss >> std::hex >> ThreadID;
|
||||
@@ -928,7 +926,7 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
}
|
||||
if (match("qC")) {
|
||||
// Returns the current Thread ID
|
||||
fextl::ostringstream ss;
|
||||
std::ostringstream ss;
|
||||
ss << "m" << std::hex << CTX->ParentThread->ThreadManager.TID;
|
||||
return {ss.str(), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
@@ -937,10 +935,10 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("qSymbol")) {
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("qSymbol").size());
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("qSymbol").size());
|
||||
ss.get(); // discard colon
|
||||
fextl::string Symbol_Val, Symbol_name;
|
||||
std::string Symbol_Val, Symbol_name;
|
||||
std::getline(ss, Symbol_Val, ':');
|
||||
std::getline(ss, Symbol_name, ':');
|
||||
|
||||
@@ -957,13 +955,13 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
std::fill(PassSignals.begin(), PassSignals.end(), false);
|
||||
|
||||
// eg: QPassSignals:e;10;14;17;1a;1b;1c;21;24;25;2c;4c;97;
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("QPassSignals").size());
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("QPassSignals").size());
|
||||
ss.get(); // discard colon
|
||||
|
||||
// We now have a semi-colon deliminated list of signals to pass to the guest process
|
||||
for (fextl::string tmp; std::getline(ss, tmp, ';'); ) {
|
||||
uint32_t Signal = std::stoi(tmp.c_str(), nullptr, 16);
|
||||
for (std::string tmp; std::getline(ss, tmp, ';'); ) {
|
||||
uint32_t Signal = std::stoi(tmp, nullptr, 16);
|
||||
if (Signal < SignalDelegator::MAX_SIGNALS) {
|
||||
PassSignals[Signal] = true;
|
||||
}
|
||||
@@ -986,7 +984,7 @@ GdbServer::HandledPacketType GdbServer::ThreadAction(char action, uint32_t tid)
|
||||
case 's': {
|
||||
CTX->Step();
|
||||
SendPacketPair({"OK", HandledPacketType::TYPE_ACK});
|
||||
fextl::string str = fextl::fmt::format("T05thread:{:02x};", getpid());
|
||||
auto str = fmt::format("T05thread:{:02x};", getpid());
|
||||
if (LibraryMapChanged) {
|
||||
// If libraries have changed then let gdb know
|
||||
str += "library:1;";
|
||||
@@ -1004,26 +1002,26 @@ GdbServer::HandledPacketType GdbServer::ThreadAction(char action, uint32_t tid)
|
||||
}
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleV(const fextl::string& packet) {
|
||||
const auto match = [&](const fextl::string& str) -> std::optional<fextl::istringstream> {
|
||||
GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
|
||||
const auto match = [&](const std::string& str) -> std::optional<std::istringstream> {
|
||||
if (packet.rfind(str, 0) == 0) {
|
||||
auto ss = fextl::istringstream(packet);
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(str.size());
|
||||
return ss;
|
||||
}
|
||||
return std::nullopt;
|
||||
};
|
||||
|
||||
const auto F = [](int result) -> fextl::string { return fextl::fmt::format("F{:x}", result); };
|
||||
const auto F_error = []() -> fextl::string { return fextl::fmt::format("F-1,{:x}", errno); };
|
||||
const auto F_data = [](int result, const fextl::string& data) -> fextl::string {
|
||||
const auto F = [](int result) { return fmt::format("F{:x}", result); };
|
||||
const auto F_error = [] { return fmt::format("F-1,{:x}", errno); };
|
||||
const auto F_data = [](int result, const std::string& data) {
|
||||
// Binary encoded data is raw appended to the end
|
||||
return fextl::fmt::format("F{:#x};", result) + data;
|
||||
return fmt::format("F{:#x};", result) + data;
|
||||
};
|
||||
|
||||
std::optional<fextl::istringstream> ss;
|
||||
std::optional<std::istringstream> ss;
|
||||
if((ss = match("vFile:open:"))) {
|
||||
fextl::string filename;
|
||||
std::string filename;
|
||||
int flags;
|
||||
int mode;
|
||||
|
||||
@@ -1055,7 +1053,7 @@ GdbServer::HandledPacketType GdbServer::handleV(const fextl::string& packet) {
|
||||
ss->get(); // discard comma
|
||||
*ss >> std::hex >> offset;
|
||||
|
||||
fextl::string data(count, '\0');
|
||||
std::string data(count, '\0');
|
||||
if (lseek(fd, offset, SEEK_SET) < 0) {
|
||||
return {F_error(), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
@@ -1095,14 +1093,14 @@ GdbServer::HandledPacketType GdbServer::handleV(const fextl::string& packet) {
|
||||
return {"", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleThreadOp(const fextl::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet) {
|
||||
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
|
||||
if (match("Hc")) {
|
||||
// Sets thread to this ID for stepping
|
||||
// This is deprecated and vCont should be used instead
|
||||
auto ss = fextl::istringstream(packet);
|
||||
ss.seekg(fextl::string("Hc").size());
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("Hc").size());
|
||||
ss >> std::hex >> CurrentDebuggingThread;
|
||||
|
||||
CTX->Pause();
|
||||
@@ -1111,7 +1109,7 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(const fextl::string &pack
|
||||
|
||||
if (match("Hg")) {
|
||||
// Sets thread for "other" operations
|
||||
auto ss = fextl::istringstream(packet);
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string_view("Hg").size());
|
||||
ss >> std::hex >> CurrentDebuggingThread;
|
||||
|
||||
@@ -1123,8 +1121,8 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(const fextl::string &pack
|
||||
return {"", HandledPacketType::TYPE_UNKNOWN};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleBreakpoint(const fextl::string &packet) {
|
||||
auto ss = fextl::istringstream(packet);
|
||||
GdbServer::HandledPacketType GdbServer::handleBreakpoint(const std::string &packet) {
|
||||
auto ss = std::istringstream(packet);
|
||||
|
||||
// Don't do anything with set breakpoints yet
|
||||
[[maybe_unused]] bool Set{};
|
||||
@@ -1140,13 +1138,13 @@ GdbServer::HandledPacketType GdbServer::handleBreakpoint(const fextl::string &pa
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::ProcessPacket(const fextl::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::ProcessPacket(const std::string &packet) {
|
||||
switch (packet[0]) {
|
||||
case '?': {
|
||||
// Indicates the reason that the thread has stopped
|
||||
// Behaviour changes if the target is in non-stop mode
|
||||
// Binja doesn't support S response here
|
||||
fextl::string str = fextl::fmt::format("T00thread:{:x};", getpid());
|
||||
auto str = fmt::format("T00thread:{:x};", getpid());
|
||||
return {std::move(str), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
case 'c':
|
||||
@@ -1227,7 +1225,7 @@ void GdbServer::GdbServerLoop() {
|
||||
while ((c = CommsStream->get()) >= 0 ) {
|
||||
switch (c) {
|
||||
case '$': {
|
||||
auto packet = ReadPacket(*CommsStream);
|
||||
std::string packet = ReadPacket(*CommsStream);
|
||||
response = ProcessPacket(packet);
|
||||
SendPacketPair(response);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
|
||||
@@ -1247,7 +1245,7 @@ void GdbServer::GdbServerLoop() {
|
||||
break;
|
||||
case '\x03': { // ASCII EOT
|
||||
CTX->Pause();
|
||||
fextl::string str = fextl::fmt::format("T02thread:{:02x};", getpid());
|
||||
auto str = fmt::format("T02thread:{:02x};", getpid());
|
||||
if (LibraryMapChanged) {
|
||||
// If libraries have changed then let gdb know
|
||||
str += "library:1;";
|
||||
@@ -1281,8 +1279,7 @@ void GdbServer::StartThread() {
|
||||
}
|
||||
|
||||
void GdbServer::OpenListenSocket() {
|
||||
// getaddrinfo allocates memory that can't be removed.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
// open socket
|
||||
struct addrinfo hints, *res;
|
||||
|
||||
memset(&hints, 0, sizeof(hints));
|
||||
@@ -1311,11 +1308,9 @@ void GdbServer::OpenListenSocket() {
|
||||
}
|
||||
|
||||
listen(ListenSocket, 1);
|
||||
|
||||
freeaddrinfo(res);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
std::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
// Block until a connection arrives
|
||||
struct sockaddr_storage their_addr{};
|
||||
socklen_t addr_size{};
|
||||
@@ -1323,8 +1318,7 @@ fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
LogMan::Msg::IFmt("GdbServer, waiting for connection on localhost:8086");
|
||||
int new_fd = accept(ListenSocket, (struct sockaddr *)&their_addr, &addr_size);
|
||||
|
||||
return fextl::make_unique<FEXCore::Utils::NetStream>(new_fd);
|
||||
return std::make_unique<FEXCore::Utils::NetStream>(new_fd);
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore
|
||||
+22
-23
@@ -8,24 +8,23 @@ $end_info$
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Context {
|
||||
class ContextImpl;
|
||||
struct Context;
|
||||
}
|
||||
|
||||
class GdbServer {
|
||||
public:
|
||||
GdbServer(FEXCore::Context::ContextImpl *ctx);
|
||||
GdbServer(FEXCore::Context::Context *ctx);
|
||||
|
||||
// Public for threading
|
||||
void GdbServerLoop();
|
||||
@@ -38,10 +37,10 @@ private:
|
||||
void Break(int signal);
|
||||
|
||||
void OpenListenSocket();
|
||||
fextl::unique_ptr<std::iostream> OpenSocket();
|
||||
std::unique_ptr<std::iostream> OpenSocket();
|
||||
void StartThread();
|
||||
fextl::string ReadPacket(std::iostream &stream);
|
||||
void SendPacket(std::ostream &stream, const fextl::string& packet);
|
||||
std::string ReadPacket(std::iostream &stream);
|
||||
void SendPacket(std::ostream &stream, const std::string& packet);
|
||||
|
||||
void SendACK(std::ostream &stream, bool NACK);
|
||||
|
||||
@@ -49,7 +48,7 @@ private:
|
||||
void WaitForThreadWakeup();
|
||||
|
||||
struct HandledPacketType {
|
||||
fextl::string Response{};
|
||||
std::string Response{};
|
||||
enum ResponseType {
|
||||
TYPE_NONE,
|
||||
TYPE_UNKNOWN,
|
||||
@@ -62,32 +61,32 @@ private:
|
||||
};
|
||||
|
||||
void SendPacketPair(const HandledPacketType& packetPair);
|
||||
HandledPacketType ProcessPacket(const fextl::string &packet);
|
||||
HandledPacketType handleQuery(const fextl::string &packet);
|
||||
HandledPacketType handleXfer(const fextl::string &packet);
|
||||
HandledPacketType handleMemory(const fextl::string &packet);
|
||||
HandledPacketType handleV(const fextl::string& packet);
|
||||
HandledPacketType handleThreadOp(const fextl::string &packet);
|
||||
HandledPacketType handleBreakpoint(const fextl::string &packet);
|
||||
HandledPacketType ProcessPacket(const std::string &packet);
|
||||
HandledPacketType handleQuery(const std::string &packet);
|
||||
HandledPacketType handleXfer(const std::string &packet);
|
||||
HandledPacketType handleMemory(const std::string &packet);
|
||||
HandledPacketType handleV(const std::string& packet);
|
||||
HandledPacketType handleThreadOp(const std::string &packet);
|
||||
HandledPacketType handleBreakpoint(const std::string &packet);
|
||||
HandledPacketType handleProgramOffsets();
|
||||
|
||||
HandledPacketType ThreadAction(char action, uint32_t tid);
|
||||
|
||||
fextl::string readRegs();
|
||||
HandledPacketType readReg(const fextl::string& packet);
|
||||
std::string readRegs();
|
||||
HandledPacketType readReg(const std::string& packet);
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
fextl::unique_ptr<std::iostream> CommsStream;
|
||||
FEXCore::Context::Context *CTX;
|
||||
std::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
std::unique_ptr<std::iostream> CommsStream;
|
||||
std::mutex sendMutex;
|
||||
bool SettingNoAckMode{false};
|
||||
bool NoAckMode{false};
|
||||
bool NonStopMode{false};
|
||||
fextl::string ThreadString{};
|
||||
fextl::string OSDataString{};
|
||||
std::string ThreadString{};
|
||||
std::string OSDataString{};
|
||||
void buildLibraryMap();
|
||||
std::atomic<bool> LibraryMapChanged = true;
|
||||
fextl::string LibraryMapString{};
|
||||
std::string LibraryMapString{};
|
||||
|
||||
// Used to keep track of which signals to pass to the guest
|
||||
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
|
||||
|
||||
+9
-23
@@ -9,7 +9,7 @@
|
||||
#endif
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include <xbyak/xbyak_util.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -17,12 +17,12 @@ namespace FEXCore {
|
||||
// Data Zero Prohibited flag
|
||||
// 0b0 = ZVA/GVA/GZVA permitted
|
||||
// 0b1 = ZVA/GVA/GZVA prohibited
|
||||
[[maybe_unused]] constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
|
||||
constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
|
||||
// Log2 of the blocksize in 32-bit words
|
||||
[[maybe_unused]] constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
|
||||
constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
[[maybe_unused]] static uint32_t GetDCZID() {
|
||||
static uint32_t GetDCZID() {
|
||||
uint64_t Result{};
|
||||
__asm("mrs %[Res], DCZID_EL0"
|
||||
: [Res] "=r" (Result));
|
||||
@@ -54,12 +54,7 @@ HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
#else
|
||||
#ifndef _WIN32
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#else
|
||||
// Need to use ID registers in WINE.
|
||||
auto Features = vixl::CPUFeatures::InferFromIDRegisters();
|
||||
#endif
|
||||
#endif
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
@@ -84,7 +79,6 @@ HostFeatures::HostFeatures() {
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
SupportsCLWB = true;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
@@ -134,18 +128,16 @@ HostFeatures::HostFeatures() {
|
||||
SupportsSHA = Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
SupportsBMI2 = Features.has(Xbyak::util::Cpu::tCLWB);
|
||||
SupportsPMULL_128Bit = Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
|
||||
uint32_t data[4];
|
||||
Xbyak::util::Cpu::getCpuid(0x8000'0000, data);
|
||||
if (data[0] >= 0x8000'0008U) {
|
||||
__cpuid(0x8000'0000, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
Xbyak::util::Cpu::getCpuid(0x8000'0008, data);
|
||||
SupportsCLZERO = data[1] & 1;
|
||||
__cpuid(0x8000'0008, eax, ebx, ecx, edx);
|
||||
SupportsCLZERO = ebx & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
@@ -166,11 +158,5 @@ HostFeatures::HostFeatures() {
|
||||
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Disable AVX if the configuration explicitly has disabled it.
|
||||
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
|
||||
if (!EnableAVX) {
|
||||
SupportsAVX = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -17,8 +17,20 @@ $end_info$
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->SignalThread(Thread, FEXCore::Core::SignalEvent::Return);
|
||||
|
||||
LOGMAN_MSG_A_FMT("unreachable");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
SignalReturn(Data->State);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
Data->State->CurrentFrame->Pointers.Interpreter.CallbackReturn(Data->State, Data->StackEntry);
|
||||
}
|
||||
@@ -79,7 +91,7 @@ DEF_OP(Syscall) {
|
||||
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
|
||||
}
|
||||
|
||||
uint64_t Res = FEXCore::Context::HandleSyscall(static_cast<Context::ContextImpl*>(Data->State->CTX)->SyscallHandler, Data->State->CurrentFrame, &Args);
|
||||
uint64_t Res = FEXCore::Context::HandleSyscall(Data->State->CTX->SyscallHandler, Data->State->CurrentFrame, &Args);
|
||||
GD = Res;
|
||||
}
|
||||
|
||||
@@ -114,7 +126,7 @@ DEF_OP(InlineSyscall) {
|
||||
DEF_OP(Thunk) {
|
||||
auto Op = IROp->C<IR::IROp_Thunk>();
|
||||
|
||||
auto thunkFn = static_cast<Context::ContextImpl*>(Data->State->CTX)->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
auto thunkFn = Data->State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
thunkFn(*GetSrc<void**>(Data->SSAData, Op->ArgPtr));
|
||||
}
|
||||
|
||||
@@ -130,7 +142,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
static_cast<Context::ContextImpl*>(Data->State->CTX)->ThreadRemoveCodeEntryFromJit(Data->State->CurrentFrame, Data->CurrentEntry);
|
||||
Data->State->CTX->ThreadRemoveCodeEntryFromJit(Data->State->CurrentFrame, Data->CurrentEntry);
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
@@ -139,19 +151,10 @@ DEF_OP(CPUID) {
|
||||
const uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Function);
|
||||
const uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Leaf);
|
||||
|
||||
auto Results = Data->State->CTX->RunCPUIDFunction(Arg, Leaf);
|
||||
auto Results = Data->State->CTX->CPUID.RunFunction(Arg, Leaf);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
uint32_t *DstPtr = GetDest<uint32_t*>(Data->SSAData, Node);
|
||||
const uint32_t Function = *GetSrc<uint32_t*>(Data->SSAData, Op->Function);
|
||||
|
||||
auto Results = Data->State->CTX->RunXCRFunction(Function);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 2);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -62,23 +62,6 @@ DEF_OP(VCastFromGPR) {
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Src), Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(VDupFromGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VDupFromGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto NumElements = OpSize / IROp->ElementSize;
|
||||
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const auto *Src = GetSrc<void*>(Data->SSAData, Op->Src);
|
||||
for (size_t i = 0; i < NumElements; i++) {
|
||||
memcpy(Tmp + (i * ElementSize), Src, ElementSize);
|
||||
}
|
||||
|
||||
memcpy(GDP, Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
|
||||
@@ -8,7 +8,7 @@ $end_info$
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/F80Fallbacks.h"
|
||||
#include "F80Ops.h"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
@@ -417,6 +417,7 @@ DEF_OP(F64SCALE) {
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+7
-3
@@ -4,9 +4,12 @@
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
template<IR::IROps Op>
|
||||
struct OpHandlers {
|
||||
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
static X80SoftFloat handle4(float src) {
|
||||
@@ -392,4 +395,5 @@ struct OpHandlers<IR::OP_F80LOADFCW> {
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
|
||||
}
|
||||
-75
@@ -1,75 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
enum IROps : uint8_t;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
// Base template for fallback handling.
|
||||
//
|
||||
// Registering and hooking up fallback is currently like so:
|
||||
//
|
||||
// 1. Go to InterpreterFallbacks.cpp and create a template specialization of
|
||||
// the GetFallbackInfo member function.
|
||||
//
|
||||
// This member function should reasonably define what the fallback you're
|
||||
// going to create will take as parameters and return as a result. For example:
|
||||
//
|
||||
// template<>
|
||||
// FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double), Core::FallbackHandlerIndex Index) {
|
||||
// return {FABI_F80_F64, (void*)fn, Index};
|
||||
// }
|
||||
//
|
||||
// Defines info about a fallback that takes a double as an argument and
|
||||
// returns a X80SoftFloat instance.
|
||||
//
|
||||
// You will also want to define a new FallbackHandlerIndex enum member and use it
|
||||
// to set up the new info handler into the Info array in FillFallbackIndexPointers.
|
||||
//
|
||||
// 1.1. (potentially optional). Define a new ABI element in the FallbackAPI enum.
|
||||
// This ABI enum value will be used to tell the JITs how to handle the fallback
|
||||
// properly. These enum values specify the return type followed by its argument types.
|
||||
//
|
||||
// So, FABI_I64_F80_F80, for example indicates that the function will behave like a
|
||||
// function as if were defined as:
|
||||
//
|
||||
// uint64_t fn(X80SoftFloat, X80SoftFloat)
|
||||
//
|
||||
// 1.2. (potentially optional). If you needed to define a new enum ABI type like in 1.1, then
|
||||
// you need to add the handling for it in the JITs, which can be found in the respective
|
||||
// JIT's JIT.cpp file in a function called Op_Unhandled
|
||||
//
|
||||
// You need to add a new case to the ABI switch statement using the new ABI type
|
||||
// and do the necessary moving of data from register-allocated JIT parameters
|
||||
// into that platform's registers that respects the calling convention. After this is
|
||||
// done, most of the necessary background boilerplate is finished.
|
||||
//
|
||||
// 2. Now, make a specialization of this class with a member function named 'handle()'
|
||||
// that takes the same parameters as the ones described in the fallback info function
|
||||
// specialization.
|
||||
//
|
||||
// For example, if you have the fallback info from the example in step 1, it would be:
|
||||
//
|
||||
// template <>
|
||||
// struct OpHandlers<IR::CoolNewIROpcode> {
|
||||
// static X80SoftFloat handle(double src) {
|
||||
// return ...;
|
||||
// }
|
||||
// };
|
||||
//
|
||||
// 3. Fill out the behavior of the OpHandler specialization to perform what you would like
|
||||
// the fallback to do.
|
||||
//
|
||||
// 4. Add an implementation of the IR op to the Interpreter that passes through to the
|
||||
// OpHandler implementation.
|
||||
//
|
||||
// 5. Done.
|
||||
//
|
||||
template <IR::IROps Op>
|
||||
struct OpHandlers {
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
-408
@@ -1,408 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
enum class AggregationOp {
|
||||
EqualAny = 0b00,
|
||||
Ranges = 0b01,
|
||||
EqualEach = 0b10,
|
||||
EqualOrdered = 0b11,
|
||||
};
|
||||
|
||||
enum class SourceData {
|
||||
U8,
|
||||
U16,
|
||||
S8,
|
||||
S16,
|
||||
};
|
||||
|
||||
enum class Polarity {
|
||||
Positive,
|
||||
Negative,
|
||||
PositiveMasked,
|
||||
NegativeMasked,
|
||||
};
|
||||
|
||||
static uint32_t handle(uint64_t RAX, uint64_t RDX, __uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetExplicitLength(RAX, control) - 1;
|
||||
const auto valid_rhs = GetExplicitLength(RDX, control) - 1;
|
||||
|
||||
return MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
|
||||
// Main PCMPXSTRX algorithm body. Allows for reuse with both implicit and explicit length variants.
|
||||
static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
|
||||
const uint32_t aggregation = PerformAggregation(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
const int32_t upper_limit = (16 >> (control & 1)) - 1;
|
||||
|
||||
// Bits are arranged as:
|
||||
// Bit #: 3 2 1 0
|
||||
// [OF | CF | SF | ZF]
|
||||
uint32_t flags = 0;
|
||||
flags |= (valid_rhs < upper_limit) ? 0b01 : 0b00;
|
||||
flags |= (valid_lhs < upper_limit) ? 0b10 : 0b00;
|
||||
|
||||
const uint32_t result = HandlePolarity(aggregation, control, upper_limit, valid_rhs);
|
||||
if (result != 0) {
|
||||
flags |= 0b0100;
|
||||
}
|
||||
if ((result & 1) != 0) {
|
||||
flags |= 0b1000;
|
||||
}
|
||||
|
||||
// We tack the flags on top of the result to avoid needing to handle
|
||||
// multiple return values in the JITs.
|
||||
return result | (flags << 16);
|
||||
}
|
||||
|
||||
static int32_t GetExplicitLength(uint64_t reg, uint16_t control) {
|
||||
// Bit 8 controls whether or not the reg value is 64-bit or 32-bit.
|
||||
int64_t value = 0;
|
||||
if (((control >> 8) & 1) != 0) {
|
||||
value = static_cast<int64_t>(reg);
|
||||
} else {
|
||||
// We need a sign extend in this case.
|
||||
value = static_cast<int32_t>(reg);
|
||||
}
|
||||
|
||||
// If control[0] is set, then we're dealing with words instead of bytes
|
||||
const int64_t limit = (control & 1) != 0 ? 8 : 16;
|
||||
|
||||
// Length needs to saturate to 16 (if bytes) or 8 (if words)
|
||||
// when the length value is greater than 16 (if bytes)/8 (if words)
|
||||
// or if the length value is less than -16 (if bytes)/-8 (if words).
|
||||
if (value < -limit || value > limit) {
|
||||
return limit;
|
||||
}
|
||||
|
||||
return std::abs(static_cast<int>(value));
|
||||
}
|
||||
|
||||
static int32_t GetElement(const __uint128_t& vec, int32_t index, uint16_t control) {
|
||||
const auto* vec_ptr = reinterpret_cast<const uint8_t*>(&vec);
|
||||
|
||||
// Control bits [1:0] define the data type being dealt with.
|
||||
switch (static_cast<SourceData>(control & 0b11)) {
|
||||
case SourceData::U8:
|
||||
return static_cast<int32_t>(vec_ptr[index]);
|
||||
case SourceData::U16: {
|
||||
uint16_t value{};
|
||||
std::memcpy(&value, vec_ptr + (sizeof(uint16_t) * static_cast<size_t>(index)), sizeof(value));
|
||||
return value;
|
||||
}
|
||||
case SourceData::S8:
|
||||
return static_cast<int8_t>(vec_ptr[index]);
|
||||
case SourceData::S16:
|
||||
default: {
|
||||
int16_t value{};
|
||||
std::memcpy(&value, vec_ptr + (sizeof(int16_t) * static_cast<size_t>(index)), sizeof(value));
|
||||
return value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static uint32_t PerformAggregation(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
switch (static_cast<AggregationOp>((control >> 2) & 0b11)) {
|
||||
case AggregationOp::EqualAny:
|
||||
return HandleEqualAny(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::Ranges:
|
||||
return HandleRanges(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::EqualEach:
|
||||
return HandleEqualEach(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::EqualOrdered:
|
||||
default:
|
||||
return HandleEqualOrdered(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
}
|
||||
|
||||
static uint32_t HandlePolarity(uint32_t value, uint16_t control, int upper_limit, int valid_rhs) {
|
||||
switch (static_cast<Polarity>((control >> 4) & 0b11)) {
|
||||
case Polarity::Negative:
|
||||
return value ^ ((2U << upper_limit) - 1);
|
||||
case Polarity::NegativeMasked:
|
||||
return value ^ ((1U << (valid_rhs + 1)) - 1);
|
||||
case Polarity::Positive:
|
||||
case Polarity::PositiveMasked:
|
||||
default:
|
||||
// Both positive masking and positive polarity are documented
|
||||
// as both being equivalent to "IntRes2 = IntRes1", where IntRes1
|
||||
// is our 'value' parameter, so we don't need to do anything in
|
||||
// these cases except return the same value.
|
||||
return value;
|
||||
}
|
||||
}
|
||||
|
||||
// Finds characters from an overall character set.
|
||||
//
|
||||
// Scans through RHS trying to find any characters contained in LHS.
|
||||
// Think of this as a sort of vectorized version of strspn (kind of).
|
||||
//
|
||||
// e.g. Assume operating on two character vectors as unsigned words
|
||||
//
|
||||
// 0 1 2 3 4 5 6 7
|
||||
// LHS -> [a, b, c, d, e, f, g, n]
|
||||
// RHS -> [z, k, v, c, d, o, p, n]
|
||||
//
|
||||
// With both explicit lengths for each string being 8 (the max length for words),
|
||||
// this would result in an intermediate result like:
|
||||
//
|
||||
// 0b1001'1000
|
||||
// │ │ │
|
||||
// 'n' match ───┘ │ │
|
||||
// │ │
|
||||
// 'd' match ──────┘ │
|
||||
// │
|
||||
// 'c' match ────────┘
|
||||
//
|
||||
static uint32_t HandleEqualAny(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
uint32_t result = 0;
|
||||
|
||||
for (int j = valid_rhs; j >= 0; j--) {
|
||||
result <<= 1;
|
||||
|
||||
const int rhs_value = GetElement(rhs, j, control);
|
||||
for (int i = valid_lhs; i >= 0; i--) {
|
||||
const int lhs_value = GetElement(lhs, i, control);
|
||||
result |= static_cast<uint32_t>(rhs_value == lhs_value);
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// Determines if a character falls within a limited range
|
||||
//
|
||||
// Scans through rhs using a range denoted by two elements
|
||||
// in lhs and determines if the respective character in rhs
|
||||
// falls within its range.
|
||||
//
|
||||
// i.e.
|
||||
// lhs_upper_bound >= rhs_value && lhs_lower_bound <= rhs_value
|
||||
//
|
||||
// e.g. Assume operating on two character vectors as unsigned words
|
||||
//
|
||||
// 0 1 2 3 4 5 6 7
|
||||
// LHS -> [a, z, A, Z, 0, 0, 0, 0]
|
||||
// RHS -> [z, k, ., C, M, ;, \, ']
|
||||
//
|
||||
// With LHS's length being 4 and RHS's lenth being 8,
|
||||
// this would result in an intermediate result like:
|
||||
//
|
||||
// 0b0001'1011
|
||||
// │ │ ││
|
||||
// 'z' >= 'M' && 'a' <= 'M' ─────┘ │ ││
|
||||
// │ ││
|
||||
// 'z' >= 'C' && 'a' <= 'C' ───────┘ ││
|
||||
// ││
|
||||
// 'Z' >= 'k' && 'A' <= 'k' ─────────┘│
|
||||
// │
|
||||
// 'Z' >= 'z' && 'A' <= 'z' ──────────┘
|
||||
//
|
||||
static uint32_t HandleRanges(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
uint32_t result = 0;
|
||||
|
||||
for (int j = valid_rhs; j >= 0; j--) {
|
||||
result <<= 1;
|
||||
|
||||
const int element = GetElement(rhs, j, control);
|
||||
for (int i = (valid_lhs - 1) | 1; i >= 0; i -= 2) {
|
||||
const int upper_bound = GetElement(lhs, i - 0, control);
|
||||
const int lower_bound = GetElement(lhs, i - 1, control);
|
||||
|
||||
const bool ge = upper_bound >= element;
|
||||
const bool le = lower_bound <= element;
|
||||
|
||||
result |= static_cast<uint32_t>(ge && le);
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// Determines if each character is equal to one another (string compare)
|
||||
//
|
||||
// Essentially the PCMPXSTRX variant of memcmp/strcmp. Sets the bit of the
|
||||
// resulting mask if both elements are equal to one another. Otherwise
|
||||
// sets it to false.
|
||||
//
|
||||
// e.g. Assume operating on two character vectors as unsigned words
|
||||
//
|
||||
// 0 1 2 3 4 5 6 7
|
||||
// LHS -> [a, b, c, d, e, f, g, n]
|
||||
// RHS -> [a, b, c, d, e, f, e, x]
|
||||
//
|
||||
// With both explicit lengths for each string being 8 (the max length for words),
|
||||
// this would result in an intermediate result like:
|
||||
//
|
||||
// 0b0011'1111
|
||||
// ││ ││││
|
||||
// 'f' == 'f' ────┘│ ││││
|
||||
// │ ││││
|
||||
// 'e' == 'e' ─────┘ ││││
|
||||
// ││││
|
||||
// 'd' == 'd' ───────┘│││
|
||||
// │││
|
||||
// 'c' == 'c' ────────┘││
|
||||
// ││
|
||||
// 'b' == 'b' ─────────┘│
|
||||
// │
|
||||
// 'a' == 'a' ──────────┘
|
||||
//
|
||||
static uint32_t HandleEqualEach(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
const auto upper_limit = (16 >> (control & 1)) - 1;
|
||||
const auto max_valid = std::max(valid_lhs, valid_rhs);
|
||||
const auto min_valid = std::min(valid_lhs, valid_rhs);
|
||||
|
||||
// All values past the end of string must be forced to true.
|
||||
// (See 4.1.6 Valid/Invalid Override of Comparisons in the Intel Software Development Manual)
|
||||
// So we can calculate this part of the mask ahead of time and set all those to-be bits to true
|
||||
// and then progressively shift them into place over the course of execution.
|
||||
uint32_t result = (1U << (upper_limit - max_valid)) - 1;
|
||||
result <<= (max_valid - min_valid);
|
||||
|
||||
for (int i = min_valid; i >= 0; i--) {
|
||||
const int lhs_element = GetElement(lhs, i, control);
|
||||
const int rhs_element = GetElement(rhs, i, control);
|
||||
|
||||
result <<= 1;
|
||||
result |= static_cast<uint32_t>(lhs_element == rhs_element);
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// Determines if a substring exists within an overall string
|
||||
//
|
||||
// Somewhat equivalent to the behavior of strstr.
|
||||
//
|
||||
// Sets the corresponding index in the result where a substring is found.
|
||||
//
|
||||
// e.g. Assume operating on two character vectors as unsigned words
|
||||
//
|
||||
// 0 1 2 3 4 5 6 7
|
||||
// LHS -> [b, a, x, z, y, v, o, m]
|
||||
// RHS -> [b, a, d, b, a, n, k, s]
|
||||
//
|
||||
// With the length of LHS being 2 and the length of RHS being 8, we have a composition like:
|
||||
//
|
||||
// Substring to look for
|
||||
// ┌──┴──┐
|
||||
// LHS -> [b, a, x, z, y, v, o, m]
|
||||
// RHS -> [b, a, d, b, a, n, k, s]
|
||||
// └───────────┬────────────┘
|
||||
// Entire string to search
|
||||
//
|
||||
// And we end up with a result like:
|
||||
//
|
||||
// 0b0000'1001
|
||||
// │ │
|
||||
// At index 3 ───────┘ │
|
||||
// │
|
||||
// At index 0 ──────────┘
|
||||
//
|
||||
static uint32_t HandleEqualOrdered(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
const auto upper_limit = (16 >> (control & 1)) - 1;
|
||||
|
||||
// Edge case!
|
||||
// If we have *no* valid characters in our inner string, then
|
||||
// we need to return the intermediate result as
|
||||
// 0xFF (if operating on words) or 0xFFFF (if operating on bytes)
|
||||
if (valid_lhs == -1) {
|
||||
return (2U << upper_limit) - 1;
|
||||
}
|
||||
|
||||
uint32_t result = 0;
|
||||
const int initial = valid_rhs == upper_limit ? valid_rhs
|
||||
: valid_rhs - valid_lhs;
|
||||
for (int j = initial; j >= 0; j--) {
|
||||
result <<= 1;
|
||||
|
||||
uint32_t value = 1;
|
||||
const int start = std::min(valid_rhs - j, valid_lhs);
|
||||
for (int i = start; i >= 0; i--) {
|
||||
const int lhs_value = GetElement(lhs, i + 0, control);
|
||||
const int rhs_value = GetElement(rhs, i + j, control);
|
||||
|
||||
value &= static_cast<uint32_t>(lhs_value == rhs_value);
|
||||
}
|
||||
|
||||
result |= value;
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPISTRX> {
|
||||
// Essentially the same in terms of behavior with VPCMPESTRX instructions,
|
||||
// with the only difference being that the length of the string is encoded
|
||||
// as part of the data vectors passed in.
|
||||
//
|
||||
// i.e. Length is determined by the presence of a NUL (all-zero) character
|
||||
// within the data.
|
||||
//
|
||||
// If no NUL character exists, then the length of the strings are assumed
|
||||
// to be the max length possible for the given character size specified
|
||||
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
|
||||
//
|
||||
static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
|
||||
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
|
||||
|
||||
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
|
||||
static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
|
||||
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
int32_t length = 0;
|
||||
|
||||
if (is_using_words) {
|
||||
const auto get_word = [data_u8](int32_t index) {
|
||||
const auto* src = data_u8 + (index * sizeof(uint16_t));
|
||||
|
||||
uint16_t element{};
|
||||
std::memcpy(&element, src, sizeof(uint16_t));
|
||||
return element;
|
||||
};
|
||||
|
||||
while (length < 8 && get_word(length) != 0) {
|
||||
length++;
|
||||
}
|
||||
} else {
|
||||
while (length < 16 && data_u8[length] != 0) {
|
||||
length++;
|
||||
}
|
||||
}
|
||||
|
||||
return length;
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -6,24 +6,27 @@
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Dispatcher;
|
||||
class X86DispatchGenerator;
|
||||
class Arm64DispatchGenerator;
|
||||
|
||||
using DestMapType = fextl::vector<uint32_t>;
|
||||
#define DESTMAP_AS_MAP 0
|
||||
#if DESTMAP_AS_MAP
|
||||
using DestMapType = std::unordered_map<uint32_t, uint32_t>;
|
||||
#else
|
||||
using DestMapType = std::vector<uint32_t>;
|
||||
#endif
|
||||
|
||||
class InterpreterCore final : public CPUBackend {
|
||||
public:
|
||||
explicit InterpreterCore(Dispatcher *Dispatch,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
[[nodiscard]] fextl::string GetName() override { return "Interpreter"; }
|
||||
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
|
||||
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
@@ -32,8 +35,8 @@ public:
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::ContextImpl *CTX);
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
private:
|
||||
|
||||
@@ -1,14 +1,17 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#include <memory>
|
||||
#include <signal.h>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
@@ -46,21 +49,27 @@ InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::Internal
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
const auto IRSize = AlignUp(IR->GetInlineSize(), 16);
|
||||
const auto MaxSize = IRSize + Dispatcher::MaxInterpreterTrampolineSize + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
|
||||
if ((BufferUsed + MaxSize) > CurrentCodeBuffer->Size) {
|
||||
static_cast<Context::ContextImpl*>(ThreadState->CTX)->ClearCodeCache(ThreadState);
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode CodeData{};
|
||||
const auto BufferStart = CurrentCodeBuffer->Ptr + BufferUsed;
|
||||
|
||||
const auto BufferStartOffset = BufferUsed;
|
||||
CodeData.BlockBegin = CodeData.BlockEntry = CurrentCodeBuffer->Ptr + BufferStartOffset;
|
||||
|
||||
auto DestBuffer = CodeData.BlockBegin;
|
||||
auto DestBuffer = BufferStart;
|
||||
|
||||
if (GDBEnabled) {
|
||||
const auto GDBSize = Dispatch->GenerateGDBPauseCheck(DestBuffer, Entry);
|
||||
@@ -77,9 +86,7 @@ CPUBackend::CompiledCode InterpreterCore::CompileCode(uint64_t Entry, [[maybe_un
|
||||
DestBuffer += IRSize;
|
||||
BufferUsed += IRSize;
|
||||
|
||||
CodeData.Size = BufferUsed - BufferStartOffset;
|
||||
|
||||
return CodeData;
|
||||
return BufferStart;
|
||||
}
|
||||
|
||||
void InterpreterCore::ClearCache() {
|
||||
@@ -88,8 +95,12 @@ void InterpreterCore::ClearCache() {
|
||||
BufferUsed = 0;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return fextl::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
|
||||
}
|
||||
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
InterpreterCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures() {
|
||||
|
||||
@@ -1,10 +1,9 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
@@ -15,9 +14,9 @@ namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
struct DispatcherConfig;
|
||||
|
||||
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::ContextImpl *ctx,
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::ContextImpl *CTX);
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+2
-25
@@ -1,8 +1,6 @@
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
|
||||
#include "FEXCore/Core/CoreState.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/F80Fallbacks.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
#include "Interface/Core/Interpreter/F80Ops.h"
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
@@ -89,16 +87,6 @@ FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat), FEXC
|
||||
return {FABI_F80_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint32_t(*fn)(uint64_t, uint64_t, __uint128_t, __uint128_t, uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I32_I64_I64_I128_I128_I16, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint32_t(*fn)(__uint128_t, __uint128_t, uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I32_I128_I128_I16, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F80LOADFCW] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW).fn);
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4).fn);
|
||||
@@ -156,9 +144,6 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F64FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle, Core::OPINDEX_F64FPREM1).fn);
|
||||
Info[Core::OPINDEX_F64SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle, Core::OPINDEX_F64SCALE).fn);
|
||||
|
||||
// SSE4.2 string instructions
|
||||
Info[Core::OPINDEX_VPCMPESTRX] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX).fn);
|
||||
Info[Core::OPINDEX_VPCMPISTRX] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX).fn);
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
|
||||
@@ -317,14 +302,6 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInf
|
||||
COMMON_F64_OP(FPREM)
|
||||
COMMON_F64_OP(SCALE)
|
||||
|
||||
// SSE4.2 Fallbacks
|
||||
case IR::OP_VPCMPESTRX:
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX);
|
||||
return true;
|
||||
case IR::OP_VPCMPISTRX:
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX);
|
||||
return true;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -2,6 +2,11 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "InterpreterDefines.h"
|
||||
#include "InterpreterOps.h"
|
||||
#include "F80Ops.h"
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -108,6 +113,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
|
||||
// Branch ops
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
REGISTER_OP(JUMP, Jump);
|
||||
@@ -118,12 +124,10 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(VDUPFROMGPR, VDupFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
@@ -150,12 +154,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
REGISTER_OP(MEMSET, MemSet);
|
||||
REGISTER_OP(MEMCPY, MemCpy);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINECLEAN, CacheLineClean);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
// Misc ops
|
||||
@@ -222,8 +221,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VZIP2, VZip);
|
||||
REGISTER_OP(VUNZIP, VUnZip);
|
||||
REGISTER_OP(VUNZIP2, VUnZip);
|
||||
REGISTER_OP(VTRN, VTrn);
|
||||
REGISTER_OP(VTRN2, VTrn);
|
||||
REGISTER_OP(VBSL, VBSL);
|
||||
REGISTER_OP(VCMPEQ, VCMPEQ);
|
||||
REGISTER_OP(VCMPEQZ, VCMPEQZ);
|
||||
@@ -268,8 +265,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VPCMPESTRX, VPCMPESTRX);
|
||||
REGISTER_OP(VPCMPISTRX, VPCMPISTRX);
|
||||
|
||||
// Encryption ops
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
@@ -334,6 +329,7 @@ void InterpreterOps::InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::I
|
||||
|
||||
const uintptr_t ListSize = CurrentIR->GetSSACount();
|
||||
|
||||
static_assert(sizeof(FEXCore::IR::IROp_Header) == 4);
|
||||
static_assert(sizeof(FEXCore::IR::OrderedNode) == 16);
|
||||
|
||||
auto BlockEnd = CurrentIR->GetBlocks().end();
|
||||
|
||||
@@ -36,8 +36,6 @@ namespace FEXCore::CPU {
|
||||
FABI_I64_F80_F80,
|
||||
FABI_F80_F80,
|
||||
FABI_F80_F80_F80,
|
||||
FABI_I32_I64_I64_I128_I128_I16,
|
||||
FABI_I32_I128_I128_I16,
|
||||
};
|
||||
|
||||
struct FallbackInfo {
|
||||
@@ -144,6 +142,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
DEF_OP(Jump);
|
||||
@@ -154,12 +153,10 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(VDupFromGPR);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
@@ -184,12 +181,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadVectorMasked);
|
||||
DEF_OP(VStoreVectorMasked);
|
||||
DEF_OP(MemSet);
|
||||
DEF_OP(MemCpy);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineClean);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
@@ -249,7 +241,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VSMax);
|
||||
DEF_OP(VZip);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VTrn);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
@@ -294,8 +285,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VPCMPESTRX);
|
||||
DEF_OP(VPCMPISTRX);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
|
||||
@@ -23,22 +23,6 @@ static inline void CacheLineFlush(char *Addr) {
|
||||
#endif
|
||||
}
|
||||
|
||||
static inline void CacheLineClean(char *Addr) {
|
||||
#ifdef _M_X86_64
|
||||
__asm volatile (
|
||||
"clwb (%[Addr]);"
|
||||
:: [Addr] "r" (Addr)
|
||||
: "memory");
|
||||
#elif _M_ARM_64
|
||||
__asm volatile (
|
||||
"dc cvac, %[Addr]"
|
||||
:: [Addr] "r" (Addr)
|
||||
: "memory");
|
||||
#else
|
||||
LOGMAN_THROW_A_FMT("Unsupported architecture with cacheline clean");
|
||||
#endif
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(LoadContext) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
@@ -288,366 +272,6 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadVectorMasked) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto NumElements = OpSize / ElementSize;
|
||||
|
||||
const auto *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
|
||||
const auto *Mask = GetSrc<uint8_t const*>(Data->SSAData, Op->Mask);
|
||||
|
||||
const auto SetElements = [NumElements]<typename T>(void* Dst, const T* MaskValues, const T* MemoryData) {
|
||||
const auto SignBit = 1ULL << ((sizeof(T) * 8) - 1);
|
||||
for (size_t i = 0; i < NumElements; i++) {
|
||||
if ((MaskValues[i] & SignBit) != 0) {
|
||||
std::memcpy(static_cast<uint8_t*>(Dst) + (i * sizeof(T)), MemoryData + i, sizeof(T));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
|
||||
|
||||
switch(Op->OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
|
||||
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
|
||||
memset(GDP, 0, Core::CPUState::XMM_AVX_REG_SIZE);
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
SetElements(GDP, Mask, MemData);
|
||||
return;
|
||||
}
|
||||
case 2: {
|
||||
SetElements(GDP,
|
||||
reinterpret_cast<const uint16_t*>(Mask),
|
||||
reinterpret_cast<const uint16_t*>(MemData));
|
||||
return;
|
||||
}
|
||||
case 4: {
|
||||
SetElements(GDP,
|
||||
reinterpret_cast<const uint32_t*>(Mask),
|
||||
reinterpret_cast<const uint32_t*>(MemData));
|
||||
return;
|
||||
}
|
||||
case 8: {
|
||||
SetElements(GDP,
|
||||
reinterpret_cast<const uint64_t*>(Mask),
|
||||
reinterpret_cast<const uint64_t*>(MemData));
|
||||
return;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VLoadVectorMasked element size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VStoreVectorMasked) {
|
||||
const auto Op = IROp->C<IR::IROp_VStoreVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto NumElements = OpSize / ElementSize;
|
||||
|
||||
auto *Dst = *GetSrc<uint8_t**>(Data->SSAData, Op->Addr);
|
||||
const auto *RegData = GetSrc<uint8_t const*>(Data->SSAData, Op->Data);
|
||||
const auto *Mask = GetSrc<uint8_t const*>(Data->SSAData, Op->Mask);
|
||||
|
||||
const auto SetElements = [NumElements]<typename T>(void* Dst, const T* MaskValues, const T* DataVals) {
|
||||
const auto SignBit = 1ULL << ((sizeof(T) * 8) - 1);
|
||||
for (size_t i = 0; i < NumElements; i++) {
|
||||
if ((MaskValues[i] & SignBit) != 0) {
|
||||
std::memcpy(static_cast<uint8_t*>(Dst) + (i * sizeof(T)), DataVals + i, sizeof(T));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
|
||||
|
||||
switch(Op->OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: Dst += Offset; break;
|
||||
case IR::MEM_OFFSET_UXTW.Val: Dst += (uint32_t)Offset; break;
|
||||
case IR::MEM_OFFSET_SXTW.Val: Dst += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
SetElements(Dst, Mask, RegData);
|
||||
return;
|
||||
}
|
||||
case 2: {
|
||||
SetElements(Dst,
|
||||
reinterpret_cast<const uint16_t*>(Mask),
|
||||
reinterpret_cast<const uint16_t*>(RegData));
|
||||
return;
|
||||
}
|
||||
case 4: {
|
||||
SetElements(Dst,
|
||||
reinterpret_cast<const uint32_t*>(Mask),
|
||||
reinterpret_cast<const uint32_t*>(RegData));
|
||||
return;
|
||||
}
|
||||
case 8: {
|
||||
SetElements(Dst,
|
||||
reinterpret_cast<const uint64_t*>(Mask),
|
||||
reinterpret_cast<const uint64_t*>(RegData));
|
||||
return;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VStoreVectorMasked element size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(MemSet) {
|
||||
const auto Op = IROp->C<IR::IROp_MemSet>();
|
||||
const int32_t Size = Op->Size;
|
||||
|
||||
char *MemData = *GetSrc<char **>(Data->SSAData, Op->Addr);
|
||||
uint64_t MemPrefix{};
|
||||
if (!Op->Prefix.IsInvalid()) {
|
||||
MemPrefix = *GetSrc<uint64_t*>(Data->SSAData, Op->Prefix);
|
||||
}
|
||||
|
||||
const auto Value = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
const auto Length = *GetSrc<uint64_t*>(Data->SSAData, Op->Length);
|
||||
const auto Direction = *GetSrc<uint8_t*>(Data->SSAData, Op->Direction);
|
||||
|
||||
auto MemSetElements = [](auto* Memory, uint64_t Value, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
Memory[i] = Value;
|
||||
}
|
||||
};
|
||||
|
||||
auto MemSetElementsInverse = [](auto* Memory, uint64_t Value, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
Memory[-i] = Value;
|
||||
}
|
||||
};
|
||||
|
||||
if (Direction == 0) { // Forward
|
||||
if (Op->IsAtomic) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint8_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint16_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint32_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElements(reinterpret_cast<std::atomic<uint64_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElements(reinterpret_cast<uint8_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElements(reinterpret_cast<uint16_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElements(reinterpret_cast<uint32_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElements(reinterpret_cast<uint64_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
GD = reinterpret_cast<uint64_t>(MemData + (Length * Size));
|
||||
}
|
||||
else { // Backward
|
||||
if (Op->IsAtomic) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint8_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint16_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint32_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsInverse(reinterpret_cast<std::atomic<uint64_t>*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsInverse(reinterpret_cast<uint8_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsInverse(reinterpret_cast<uint16_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsInverse(reinterpret_cast<uint32_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsInverse(reinterpret_cast<uint64_t*>(MemData + MemPrefix), Value, Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
GD = reinterpret_cast<uint64_t>(MemData - (Length * Size));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
const auto Op = IROp->C<IR::IROp_MemCpy>();
|
||||
const int32_t Size = Op->Size;
|
||||
|
||||
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
|
||||
char *MemDataDest = *GetSrc<char **>(Data->SSAData, Op->AddrDest);
|
||||
char *MemDataSrc = *GetSrc<char **>(Data->SSAData, Op->AddrSrc);
|
||||
|
||||
uint64_t DestPrefix{};
|
||||
uint64_t SrcPrefix{};
|
||||
if (!Op->PrefixDest.IsInvalid()) {
|
||||
DestPrefix = *GetSrc<uint64_t*>(Data->SSAData, Op->PrefixDest);
|
||||
|
||||
}
|
||||
if (!Op->PrefixSrc.IsInvalid()) {
|
||||
SrcPrefix = *GetSrc<uint64_t*>(Data->SSAData, Op->PrefixSrc);
|
||||
}
|
||||
|
||||
const auto Length = *GetSrc<uint64_t*>(Data->SSAData, Op->Length);
|
||||
const auto Direction = *GetSrc<uint8_t*>(Data->SSAData, Op->Direction);
|
||||
|
||||
auto MemSetElementsAtomic = [](auto* MemDst, auto* MemSrc, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
MemDst[i].store(MemSrc[i].load());
|
||||
}
|
||||
};
|
||||
|
||||
auto MemSetElementsAtomicInverse = [](auto* MemDst, auto* MemSrc, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
MemDst[-i].store(MemSrc[-i].load());
|
||||
}
|
||||
};
|
||||
|
||||
auto MemSetElements = [](auto* MemDst, auto* MemSrc, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
MemDst[i] = MemSrc[i];
|
||||
}
|
||||
};
|
||||
|
||||
auto MemSetElementsInverse = [](auto* MemDst, auto* MemSrc, size_t Length) {
|
||||
for (size_t i = 0; i < Length; ++i) {
|
||||
MemDst[-i] = MemSrc[-i];
|
||||
}
|
||||
};
|
||||
|
||||
if (Direction == 0) { // Forward
|
||||
if (Op->IsAtomic) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsAtomic(reinterpret_cast<std::atomic<uint8_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint8_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsAtomic(reinterpret_cast<std::atomic<uint16_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint16_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsAtomic(reinterpret_cast<std::atomic<uint32_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint32_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsAtomic(reinterpret_cast<std::atomic<uint64_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint64_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElements(reinterpret_cast<uint8_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint8_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElements(reinterpret_cast<uint16_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint16_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElements(reinterpret_cast<uint32_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint32_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElements(reinterpret_cast<uint64_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint64_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
DstPtr[0] = reinterpret_cast<uint64_t>(MemDataDest + (Length * Size));
|
||||
DstPtr[1] = reinterpret_cast<uint64_t>(MemDataSrc + (Length * Size));
|
||||
}
|
||||
else { // Backward
|
||||
if (Op->IsAtomic) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsAtomicInverse(reinterpret_cast<std::atomic<uint8_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint8_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsAtomicInverse(reinterpret_cast<std::atomic<uint16_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint16_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsAtomicInverse(reinterpret_cast<std::atomic<uint32_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint32_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsAtomicInverse(reinterpret_cast<std::atomic<uint64_t>*>(MemDataDest + DestPrefix), reinterpret_cast<std::atomic<uint64_t>*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
MemSetElementsInverse(reinterpret_cast<uint8_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint8_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 2:
|
||||
MemSetElementsInverse(reinterpret_cast<uint16_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint16_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 4:
|
||||
MemSetElementsInverse(reinterpret_cast<uint32_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint32_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
case 8:
|
||||
MemSetElementsInverse(reinterpret_cast<uint64_t*>(MemDataDest + DestPrefix), reinterpret_cast<uint64_t*>(MemDataSrc + SrcPrefix), Length);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
DstPtr[0] = reinterpret_cast<uint64_t>(MemDataDest - (Length * Size));
|
||||
DstPtr[1] = reinterpret_cast<uint64_t>(MemDataSrc - (Length * Size));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
@@ -657,15 +281,6 @@ DEF_OP(CacheLineClear) {
|
||||
CacheLineFlush(MemData);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClean) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClean>();
|
||||
|
||||
char *MemData = *GetSrc<char **>(Data->SSAData, Op->Addr);
|
||||
|
||||
// 64-byte cache line clear
|
||||
CacheLineClean(MemData);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
|
||||
@@ -8,8 +8,6 @@ $end_info$
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
|
||||
@@ -904,67 +902,6 @@ DEF_OP(VZip) {
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VTrn) {
|
||||
const auto Op = IROp->C<IR::IROp_VTrn>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
uint8_t Elements = OpSize / ElementSize;
|
||||
const uint8_t BaseOffset = IROp->Op == IR::OP_VTRN2 ? 1 : 0;
|
||||
Elements >>= 1;
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
auto *Dst_d = reinterpret_cast<uint8_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint8_t*>(Src1);
|
||||
auto *Src2_d = reinterpret_cast<uint8_t*>(Src2);
|
||||
for (unsigned i = 0; i < Elements; ++i) {
|
||||
Dst_d[i*2] = Src1_d[i*2 + BaseOffset];
|
||||
Dst_d[i*2+1] = Src2_d[i*2 + BaseOffset];
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto *Dst_d = reinterpret_cast<uint16_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint16_t*>(Src1);
|
||||
auto *Src2_d = reinterpret_cast<uint16_t*>(Src2);
|
||||
for (unsigned i = 0; i < Elements; ++i) {
|
||||
Dst_d[i*2] = Src1_d[i*2 + BaseOffset];
|
||||
Dst_d[i*2+1] = Src2_d[i*2 + BaseOffset];
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto *Dst_d = reinterpret_cast<uint32_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint32_t*>(Src1);
|
||||
auto *Src2_d = reinterpret_cast<uint32_t*>(Src2);
|
||||
for (unsigned i = 0; i < Elements; ++i) {
|
||||
Dst_d[i*2] = Src1_d[i*2 + BaseOffset];
|
||||
Dst_d[i*2+1] = Src2_d[i*2 + BaseOffset];
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto *Dst_d = reinterpret_cast<uint64_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint64_t*>(Src1);
|
||||
auto *Src2_d = reinterpret_cast<uint64_t*>(Src2);
|
||||
for (unsigned i = 0; i < Elements; ++i) {
|
||||
Dst_d[i*2] = Src1_d[i*2 + BaseOffset];
|
||||
Dst_d[i*2+1] = Src2_d[i*2 + BaseOffset];
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUnZip) {
|
||||
const auto Op = IROp->C<IR::IROp_VUnZip>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -1027,9 +964,7 @@ DEF_OP(VUnZip) {
|
||||
}
|
||||
|
||||
DEF_OP(VBSL) {
|
||||
const auto Op = IROp->C<IR::IROp_VBSL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VBSL>();
|
||||
const auto Src1 = *GetSrc<InterpVector256*>(Data->SSAData, Op->VectorMask);
|
||||
const auto Src2 = *GetSrc<InterpVector256*>(Data->SSAData, Op->VectorTrue);
|
||||
const auto Src3 = *GetSrc<InterpVector256*>(Data->SSAData, Op->VectorFalse);
|
||||
@@ -1039,8 +974,7 @@ DEF_OP(VBSL) {
|
||||
.Upper = (Src2.Upper & Src1.Upper) | (Src3.Upper & ~Src1.Upper),
|
||||
};
|
||||
|
||||
memset(GDP, 0, sizeof(InterpVector256));
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(VCMPEQ) {
|
||||
@@ -2278,34 +2212,6 @@ DEF_OP(VRev64) {
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VPCMPESTRX) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto RAX = *GetSrc<uint64_t*>(Data->SSAData, Op->RAX);
|
||||
const auto RDX = *GetSrc<uint64_t*>(Data->SSAData, Op->RDX);
|
||||
const auto LHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->LHS);
|
||||
const auto RHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->RHS);
|
||||
|
||||
const auto Result = OpHandlers<IR::OP_VPCMPESTRX>::handle(RAX, RDX, LHS, RHS, Control);
|
||||
|
||||
memset(GDP, 0, sizeof(uint64_t));
|
||||
memcpy(GDP, &Result, sizeof(Result));
|
||||
}
|
||||
|
||||
DEF_OP(VPCMPISTRX) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
|
||||
const auto LHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->LHS);
|
||||
const auto RHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->RHS);
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto Result = OpHandlers<IR::OP_VPCMPISTRX>::handle(LHS, RHS, Control);
|
||||
|
||||
memset(GDP, 0, sizeof(uint64_t));
|
||||
memcpy(GDP, &Result, sizeof(Result));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+66
-13
@@ -262,13 +262,13 @@ DEF_OP(MulH) {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 4) {
|
||||
sxtw(TMP1, Src1.W());
|
||||
sxtw(TMP2, Src2.W());
|
||||
sxtw(TMP1, Src1);
|
||||
sxtw(TMP2, Src2);
|
||||
mul(ARMEmitter::Size::i32Bit, Dst, TMP1, TMP2);
|
||||
ubfx(ARMEmitter::Size::i32Bit, Dst, Dst, 32, 32);
|
||||
}
|
||||
else {
|
||||
smulh(Dst.X(), Src1.X(), Src2.X());
|
||||
smulh(Dst, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -289,7 +289,7 @@ DEF_OP(UMulH) {
|
||||
ubfx(ARMEmitter::Size::i64Bit, Dst, Dst, 32, 32);
|
||||
}
|
||||
else {
|
||||
umulh(Dst.X(), Src1.X(), Src2.X());
|
||||
umulh(Dst, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -477,9 +477,9 @@ DEF_OP(PDep) {
|
||||
const auto IndexReg = TMP4.R();
|
||||
const auto ZeroReg = ARMEmitter::Reg::zr;
|
||||
|
||||
const auto InputReg = StaticRegisters[0];
|
||||
const auto MaskReg = StaticRegisters[1];
|
||||
const auto DestReg = StaticRegisters[2];
|
||||
const auto InputReg = SRA64[0];
|
||||
const auto MaskReg = SRA64[1];
|
||||
const auto DestReg = SRA64[2];
|
||||
|
||||
const auto SpillCode = 1U << InputReg.Idx() |
|
||||
1U << MaskReg.Idx() |
|
||||
@@ -494,7 +494,7 @@ DEF_OP(PDep) {
|
||||
// We sadly need to spill regs for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(TMP1, false, SpillCode);
|
||||
SpillStaticRegs(false, SpillCode);
|
||||
|
||||
|
||||
mov(EmitSize, InputReg, Input);
|
||||
@@ -558,7 +558,7 @@ DEF_OP(PExt) {
|
||||
// We sadly need to spill a reg for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(TMP2, false, 1U << Mask.Idx());
|
||||
SpillStaticRegs(false, 1U << Mask.Idx());
|
||||
mov(EmitSize, Mask, ZeroReg);
|
||||
|
||||
// Main loop
|
||||
@@ -610,7 +610,7 @@ DEF_OP(LDiv) {
|
||||
case 4: {
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
sxtw(TMP2, Divisor.W());
|
||||
sxtw(TMP2, Divisor);
|
||||
sdiv(EmitSize, Dst, TMP1, TMP2);
|
||||
break;
|
||||
}
|
||||
@@ -744,7 +744,7 @@ DEF_OP(LRem) {
|
||||
case 4: {
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
sxtw(TMP3, Divisor.W());
|
||||
sxtw(TMP3, Divisor);
|
||||
sdiv(EmitSize, TMP2, TMP1, TMP3);
|
||||
msub(EmitSize, Dst, TMP2, TMP3, TMP1);
|
||||
break;
|
||||
@@ -1173,8 +1173,8 @@ DEF_OP(VExtractToGPR) {
|
||||
// Inverting our dedicated predicate for 128-bit operations selects
|
||||
// all of the top lanes. We can then compact those into a temporary.
|
||||
const auto CompactPred = ARMEmitter::PReg::p0;
|
||||
not_(CompactPred, PRED_TMP_32B.Zeroing(), PRED_TMP_16B);
|
||||
compact(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), CompactPred, Vector.Z());
|
||||
not_(CompactPred, PRED_TMP_32B, PRED_TMP_16B);
|
||||
compact(ARMEmitter::SubRegSize::i64Bit, VTMP1, CompactPred, Vector);
|
||||
|
||||
// Sanitize the zero-based index to work on the now-moved
|
||||
// upper half of the vector.
|
||||
@@ -1274,4 +1274,57 @@ DEF_OP(FCmp) {
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
void Arm64JITCore::RegisterALUHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
REGISTER_OP(CONSTANT, Constant);
|
||||
REGISTER_OP(ENTRYPOINTOFFSET, EntrypointOffset);
|
||||
REGISTER_OP(INLINECONSTANT, InlineConstant);
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
REGISTER_OP(UDIV, UDiv);
|
||||
REGISTER_OP(REM, Rem);
|
||||
REGISTER_OP(UREM, URem);
|
||||
REGISTER_OP(MULH, MulH);
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
REGISTER_OP(LSHL, Lshl);
|
||||
REGISTER_OP(LSHR, Lshr);
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
REGISTER_OP(LUREM, LURem);
|
||||
REGISTER_OP(NOT, Not);
|
||||
REGISTER_OP(POPCOUNT, Popcount);
|
||||
REGISTER_OP(FINDLSB, FindLSB);
|
||||
REGISTER_OP(FINDMSB, FindMSB);
|
||||
REGISTER_OP(FINDTRAILINGZEROS, FindTrailingZeros);
|
||||
REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes);
|
||||
REGISTER_OP(REV, Rev);
|
||||
REGISTER_OP(BFI, Bfi);
|
||||
REGISTER_OP(BFE, Bfe);
|
||||
REGISTER_OP(SBFE, Sbfe);
|
||||
REGISTER_OP(SELECT, Select);
|
||||
REGISTER_OP(VEXTRACTTOGPR, VExtractToGPR);
|
||||
REGISTER_OP(FLOAT_TOGPR_ZS, Float_ToGPR_ZS);
|
||||
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
|
||||
REGISTER_OP(FCMP, FCmp);
|
||||
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
}
|
||||
@@ -27,7 +27,7 @@ void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
@@ -58,7 +58,7 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
|
||||
|
||||
Bind(&Lit.Loc);
|
||||
dc64(Lit.Lit);
|
||||
@@ -70,7 +70,7 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
|
||||
@@ -438,5 +438,23 @@ DEF_OP(AtomicFetchNeg) {
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterAtomicHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(CASPAIR, CASPair);
|
||||
REGISTER_OP(CAS, CAS);
|
||||
REGISTER_OP(ATOMICADD, AtomicAdd);
|
||||
REGISTER_OP(ATOMICSUB, AtomicSub);
|
||||
REGISTER_OP(ATOMICAND, AtomicAnd);
|
||||
REGISTER_OP(ATOMICOR, AtomicOr);
|
||||
REGISTER_OP(ATOMICXOR, AtomicXor);
|
||||
REGISTER_OP(ATOMICSWAP, AtomicSwap);
|
||||
REGISTER_OP(ATOMICFETCHADD, AtomicFetchAdd);
|
||||
REGISTER_OP(ATOMICFETCHSUB, AtomicFetchSub);
|
||||
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
|
||||
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
|
||||
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+53
-66
@@ -20,9 +20,19 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
|
||||
// Now branch to our signal return helper
|
||||
// This can't be a direct branch since the code needs to live at a constant location
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler));
|
||||
br(ARMEmitter::Reg::r0);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
// spill back to CTX
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
@@ -167,22 +177,13 @@ DEF_OP(Syscall) {
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) == FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
|
||||
SpillStaticRegs();
|
||||
}
|
||||
else {
|
||||
// Need to spill all caller saved registers still
|
||||
GPRSpillMask = CALLER_GPR_MASK;
|
||||
FPRSpillMask = CALLER_FPR_MASK;
|
||||
SpillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
|
||||
}
|
||||
|
||||
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
@@ -205,22 +206,21 @@ DEF_OP(Syscall) {
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY &&
|
||||
(Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
FillStaticRegs();
|
||||
}
|
||||
else {
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask);
|
||||
FillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
|
||||
}
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Move result to its destination register
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -248,9 +248,9 @@ DEF_OP(InlineSyscall) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
if (Reg == ARMEmitter::Reg::r8 ||
|
||||
Reg == ARMEmitter::Reg::r4 ||
|
||||
Reg == ARMEmitter::Reg::r5) {
|
||||
if (Reg.Idx() == ARMEmitter::Reg::r8.Idx() ||
|
||||
Reg.Idx() == ARMEmitter::Reg::r4.Idx() ||
|
||||
Reg.Idx() == ARMEmitter::Reg::r5.Idx()) {
|
||||
|
||||
SpillMask |= (1U << Reg.Idx());
|
||||
Intersects = true;
|
||||
@@ -260,7 +260,7 @@ DEF_OP(InlineSyscall) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -281,13 +281,13 @@ DEF_OP(InlineSyscall) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg == ARMEmitter::Reg::r8) {
|
||||
if (Reg.Idx() == FEXCore::ARMEmitter::Reg::r8.Idx()) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI]));
|
||||
}
|
||||
else if (Reg == ARMEmitter::Reg::r4) {
|
||||
else if (Reg.Idx() == FEXCore::ARMEmitter::Reg::r4.Idx()) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX]));
|
||||
}
|
||||
else if (Reg == ARMEmitter::Reg::r5) {
|
||||
else if (Reg.Idx() == FEXCore::ARMEmitter::Reg::r5.Idx()) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX]));
|
||||
}
|
||||
else {
|
||||
@@ -328,13 +328,13 @@ DEF_OP(Thunk) {
|
||||
// X0: CTX
|
||||
// X1: Args (from guest stack)
|
||||
|
||||
SpillStaticRegs(TMP1); // spill to ctx before ra64 spill
|
||||
SpillStaticRegs(); // spill to ctx before ra64 spill
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr.ID()));
|
||||
|
||||
auto thunkFn = static_cast<Context::ContextImpl*>(ThreadState->CTX)->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
@@ -403,12 +403,12 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// X1: RIP
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
@@ -424,7 +424,7 @@ DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
@@ -450,34 +450,21 @@ DEF_OP(CPUID) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = XCR Function
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0
|
||||
// Results want to be in a i32v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, ARMEmitter::Reg::r0);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r0, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
REGISTER_OP(JUMP, Jump);
|
||||
REGISTER_OP(CONDJUMP, CondJump);
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -55,7 +55,7 @@ DEF_OP(VInsGPR) {
|
||||
// Move the upper lane down for the insertion.
|
||||
const auto CompactPred = ARMEmitter::PReg::p0;
|
||||
not_(CompactPred, PRED_TMP_32B.Zeroing(), PRED_TMP_16B);
|
||||
compact(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), CompactPred, DestVector.Z());
|
||||
compact(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), CompactPred, DestVector);
|
||||
}
|
||||
|
||||
// Put data in place for destructive SPLICE below.
|
||||
@@ -108,32 +108,6 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VDupFromGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VDupFromGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1,
|
||||
"Unexpected {} element size: {}", __func__, ElementSize);
|
||||
|
||||
const auto SubEmitSize =
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
dup(SubEmitSize, Dst.Z(), Src);
|
||||
} else {
|
||||
dup(SubEmitSize, Dst.Q(), Src);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
@@ -225,7 +199,7 @@ DEF_OP(Vector_FToZS) {
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
fcvtzs(Dst, SubEmitSize, Mask.Merging(), Vector, SubEmitSize);
|
||||
} else {
|
||||
fcvtzs(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
@@ -248,8 +222,8 @@ DEF_OP(Vector_FToS) {
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
frinti(SubEmitSize, Dst.Z(), Mask.Merging(), Vector.Z());
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Dst.Z(), SubEmitSize);
|
||||
frinti(SubEmitSize, Dst, Mask.Merging(), Vector);
|
||||
fcvtzs(Dst, SubEmitSize, Mask.Merging(), Dst, SubEmitSize);
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -302,12 +276,12 @@ DEF_OP(Vector_FToF) {
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Mask, Vector.Z());
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst, Mask, Vector);
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Vector.Z());
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst, Mask, Vector);
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
@@ -391,5 +365,18 @@ DEF_OP(Vector_FToI) {
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterConversionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
|
||||
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
|
||||
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
|
||||
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17,73 +17,37 @@ DEF_OP(AESImc) {
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
eor(VTMP2.Q(), VTMP2.Q(), VTMP2.Q());
|
||||
mov(VTMP1.Q(), State.Q());
|
||||
mov(VTMP1.Q(), GetVReg(Op->State.ID()).Q());
|
||||
aese(VTMP1, VTMP2);
|
||||
aesmc(VTMP1, VTMP1);
|
||||
eor(Dst.Q(), VTMP1.Q(), Key.Q());
|
||||
eor(GetVReg(Node).Q(), VTMP1.Q(), GetVReg(Op->Key.ID()).Q());
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
eor(VTMP2.Q(), VTMP2.Q(), VTMP2.Q());
|
||||
mov(VTMP1.Q(), State.Q());
|
||||
mov(VTMP1.Q(), GetVReg(Op->State.ID()).Q());
|
||||
aese(VTMP1, VTMP2);
|
||||
eor(Dst.Q(), VTMP1.Q(), Key.Q());
|
||||
eor(GetVReg(Node).Q(), VTMP1.Q(), GetVReg(Op->Key.ID()).Q());
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
eor(VTMP2.Q(), VTMP2.Q(), VTMP2.Q());
|
||||
mov(VTMP1.Q(), State.Q());
|
||||
mov(VTMP1.Q(), GetVReg(Op->State.ID()).Q());
|
||||
aesd(VTMP1, VTMP2);
|
||||
aesimc(VTMP1, VTMP1);
|
||||
eor(Dst.Q(), VTMP1.Q(), Key.Q());
|
||||
eor(GetVReg(Node).Q(), VTMP1.Q(), GetVReg(Op->Key.ID()).Q());
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
eor(VTMP2.Q(), VTMP2.Q(), VTMP2.Q());
|
||||
mov(VTMP1.Q(), State.Q());
|
||||
mov(VTMP1.Q(), GetVReg(Op->State.ID()).Q());
|
||||
aesd(VTMP1, VTMP2);
|
||||
eor(Dst.Q(), VTMP1.Q(), Key.Q());
|
||||
eor(GetVReg(Node).Q(), VTMP1.Q(), GetVReg(Op->Key.ID()).Q());
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
@@ -137,22 +101,18 @@ DEF_OP(CRC32) {
|
||||
crc32cw(Dst.W(), Src1.W(), Src2.W());
|
||||
break;
|
||||
case 8:
|
||||
crc32cx(Dst.X(), Src1.X(), Src2.X());
|
||||
crc32cx(Dst, Src1, Src2);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src1 = GetVReg(Op->Src1.ID());
|
||||
auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
@@ -176,4 +136,16 @@ DEF_OP(PCLMUL) {
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -14,5 +14,10 @@ DEF_OP(GetHostFlag) {
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterFlagHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(GETHOSTFLAG, GetHostFlag);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+91
-431
@@ -14,6 +14,8 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
@@ -23,6 +25,7 @@ $end_info$
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
@@ -30,6 +33,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <sys/mman.h>
|
||||
#include <stdio.h>
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
@@ -83,7 +87,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -103,7 +107,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F80_F32:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
@@ -127,7 +131,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F80_F64:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -153,13 +157,13 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
|
||||
case FABI_F80_I16:
|
||||
case FABI_F80_I32: {
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
|
||||
uxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
|
||||
}
|
||||
else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
|
||||
@@ -183,7 +187,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F32_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -209,7 +213,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -235,7 +239,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -259,7 +263,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -285,7 +289,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -306,11 +310,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
FillStaticRegs();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
sxth(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
uxth(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -335,7 +339,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -360,7 +364,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -388,7 +392,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -415,7 +419,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -445,75 +449,6 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I64_I64_I128_I128_I16: {
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
mov(ARMEmitter::XReg::x0, SrcRAX.X());
|
||||
mov(ARMEmitter::XReg::x1, SrcRDX.X());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r4, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r5, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r6, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x7, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r7);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r7);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
FillStaticRegs();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(Dst.W(), ARMEmitter::WReg::w0);
|
||||
break;
|
||||
}
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Control = Op->Control;
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r4, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r5);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
FillStaticRegs();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(Dst.W(), ARMEmitter::WReg::w0);
|
||||
break;
|
||||
}
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -549,7 +484,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
|
||||
FEXCore::ARMEmitter::ForwardLabel l_BranchHost;
|
||||
emit.ldr(FEXCore::ARMEmitter::XReg::x0, &l_BranchHost);
|
||||
@@ -563,7 +498,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
@@ -574,7 +509,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx, 0)
|
||||
, HostSupportsSVE{ctx->HostFeatures.SupportsAVX}
|
||||
@@ -582,20 +517,39 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
RAPass->AllocateRegisterSet(RegisterClasses);
|
||||
uint32_t NumUsedGPRs = NumGPRs;
|
||||
uint32_t NumUsedGPRPairs = NumGPRPairs;
|
||||
uint32_t UsedRegisterCount = RegisterCount;
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, GeneralPairRegisters.size());
|
||||
RAPass->AllocateRegisterSet(UsedRegisterCount, RegisterClasses);
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, NumUsedGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, SRA64.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, NumFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, SRAFPR.size() );
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, NumUsedGPRPairs);
|
||||
RAPass->AddRegisters(FEXCore::IR::ComplexClass, 1);
|
||||
|
||||
for (uint32_t i = 0; i < GeneralPairRegisters.size(); ++i) {
|
||||
for (uint32_t i = 0; i < NumUsedGPRPairs; ++i) {
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2, FEXCore::IR::GPRPairClass, i);
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2 + 1, FEXCore::IR::GPRPairClass, i);
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < FEXCore::IR::IROps::OP_LAST + 1; ++i) {
|
||||
OpHandlers[i] = &Arm64JITCore::Op_Unhandled;
|
||||
}
|
||||
|
||||
RegisterALUHandlers();
|
||||
RegisterAtomicHandlers();
|
||||
RegisterBranchHandlers();
|
||||
RegisterConversionHandlers();
|
||||
RegisterFlagHandlers();
|
||||
RegisterMemoryHandlers();
|
||||
RegisterMiscHandlers();
|
||||
RegisterMoveHandlers();
|
||||
RegisterVectorHandlers();
|
||||
RegisterEncryptionHandlers();
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
|
||||
@@ -604,7 +558,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
@@ -612,14 +566,9 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::Context::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
|
||||
|
||||
// Fill in the fallback handlers
|
||||
@@ -636,25 +585,23 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (CTX->Dispatcher->GetConfig().StaticRegisterAllocation) {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegisterSRA;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegisterSRA;
|
||||
}
|
||||
else {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegister;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegister;
|
||||
}
|
||||
void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
if (ParanoidTSO()) {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
|
||||
}
|
||||
else {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
|
||||
}
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Thread->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -724,7 +671,7 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
@@ -737,21 +684,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Put the code header at the start of the data block.
|
||||
ARMEmitter::BackwardLabel JITCodeHeaderLabel{};
|
||||
Bind(&JITCodeHeaderLabel);
|
||||
JITCodeHeader *CodeHeader = GetCursorAddress<JITCodeHeader *>();
|
||||
CursorIncrement(sizeof(JITCodeHeader));
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
@@ -761,6 +693,14 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Entry);
|
||||
#endif
|
||||
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
// AAPCS64
|
||||
// r30 = LR
|
||||
// r29 = FP
|
||||
@@ -781,15 +721,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
// X1-X3 = Temp
|
||||
// X4-r18 = RA
|
||||
|
||||
CodeData.BlockEntry = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Get the address of the JITCodeHeader and store in to the core state.
|
||||
// Two instruction cost, each 1 cycle.
|
||||
adr(TMP1, &JITCodeHeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
GuestEntry = GetCursorAddress<uint8_t *>();
|
||||
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(CodeData.BlockEntry, Entry);
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
CursorIncrement(GDBSize);
|
||||
}
|
||||
|
||||
@@ -834,257 +769,15 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
#define REGISTER_OP(op, x) case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
// ALU ops
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
REGISTER_OP(CONSTANT, Constant);
|
||||
REGISTER_OP(ENTRYPOINTOFFSET, EntrypointOffset);
|
||||
REGISTER_OP(INLINECONSTANT, InlineConstant);
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
REGISTER_OP(UDIV, UDiv);
|
||||
REGISTER_OP(REM, Rem);
|
||||
REGISTER_OP(UREM, URem);
|
||||
REGISTER_OP(MULH, MulH);
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
REGISTER_OP(LSHL, Lshl);
|
||||
REGISTER_OP(LSHR, Lshr);
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
REGISTER_OP(LUREM, LURem);
|
||||
REGISTER_OP(NOT, Not);
|
||||
REGISTER_OP(POPCOUNT, Popcount);
|
||||
REGISTER_OP(FINDLSB, FindLSB);
|
||||
REGISTER_OP(FINDMSB, FindMSB);
|
||||
REGISTER_OP(FINDTRAILINGZEROS, FindTrailingZeros);
|
||||
REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes);
|
||||
REGISTER_OP(REV, Rev);
|
||||
REGISTER_OP(BFI, Bfi);
|
||||
REGISTER_OP(BFE, Bfe);
|
||||
REGISTER_OP(SBFE, Sbfe);
|
||||
REGISTER_OP(SELECT, Select);
|
||||
REGISTER_OP(VEXTRACTTOGPR, VExtractToGPR);
|
||||
REGISTER_OP(FLOAT_TOGPR_ZS, Float_ToGPR_ZS);
|
||||
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
|
||||
REGISTER_OP(FCMP, FCmp);
|
||||
|
||||
// Atomic ops
|
||||
REGISTER_OP(CASPAIR, CASPair);
|
||||
REGISTER_OP(CAS, CAS);
|
||||
REGISTER_OP(ATOMICADD, AtomicAdd);
|
||||
REGISTER_OP(ATOMICSUB, AtomicSub);
|
||||
REGISTER_OP(ATOMICAND, AtomicAnd);
|
||||
REGISTER_OP(ATOMICOR, AtomicOr);
|
||||
REGISTER_OP(ATOMICXOR, AtomicXor);
|
||||
REGISTER_OP(ATOMICSWAP, AtomicSwap);
|
||||
REGISTER_OP(ATOMICFETCHADD, AtomicFetchAdd);
|
||||
REGISTER_OP(ATOMICFETCHSUB, AtomicFetchSub);
|
||||
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
|
||||
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
|
||||
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
|
||||
// Branch ops
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
REGISTER_OP(JUMP, Jump);
|
||||
REGISTER_OP(CONDJUMP, CondJump);
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(VDUPFROMGPR, VDupFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
|
||||
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
|
||||
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
|
||||
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
|
||||
|
||||
// Encryption ops
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
|
||||
// Flag ops
|
||||
REGISTER_OP(GETHOSTFLAG, GetHostFlag);
|
||||
|
||||
// Memory ops
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP_RT(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP_RT(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
REGISTER_OP(FILLREGISTER, FillRegister);
|
||||
REGISTER_OP(LOADFLAG, LoadFlag);
|
||||
REGISTER_OP(STOREFLAG, StoreFlag);
|
||||
REGISTER_OP(LOADMEM, LoadMem);
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP_RT(LOADMEMTSO, LoadMemTSO);
|
||||
REGISTER_OP_RT(STOREMEMTSO, StoreMemTSO);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
|
||||
REGISTER_OP(MEMSET, MemSet);
|
||||
REGISTER_OP(MEMCPY, MemCpy);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINECLEAN, CacheLineClean);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
// Misc ops
|
||||
REGISTER_OP(DUMMY, NoOp);
|
||||
REGISTER_OP(IRHEADER, NoOp);
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
REGISTER_OP(PHIVALUE, NoOp);
|
||||
REGISTER_OP(PRINT, Print);
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
|
||||
// Vector ops
|
||||
REGISTER_OP(VECTORZERO, VectorZero);
|
||||
REGISTER_OP(VECTORIMM, VectorImm);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
REGISTER_OP(VAND, VAnd);
|
||||
REGISTER_OP(VBIC, VBic);
|
||||
REGISTER_OP(VOR, VOr);
|
||||
REGISTER_OP(VXOR, VXor);
|
||||
REGISTER_OP(VADD, VAdd);
|
||||
REGISTER_OP(VSUB, VSub);
|
||||
REGISTER_OP(VUQADD, VUQAdd);
|
||||
REGISTER_OP(VUQSUB, VUQSub);
|
||||
REGISTER_OP(VSQADD, VSQAdd);
|
||||
REGISTER_OP(VSQSUB, VSQSub);
|
||||
REGISTER_OP(VADDP, VAddP);
|
||||
REGISTER_OP(VADDV, VAddV);
|
||||
REGISTER_OP(VUMINV, VUMinV);
|
||||
REGISTER_OP(VURAVG, VURAvg);
|
||||
REGISTER_OP(VABS, VAbs);
|
||||
REGISTER_OP(VPOPCOUNT, VPopcount);
|
||||
REGISTER_OP(VFADD, VFAdd);
|
||||
REGISTER_OP(VFADDP, VFAddP);
|
||||
REGISTER_OP(VFSUB, VFSub);
|
||||
REGISTER_OP(VFMUL, VFMul);
|
||||
REGISTER_OP(VFDIV, VFDiv);
|
||||
REGISTER_OP(VFMIN, VFMin);
|
||||
REGISTER_OP(VFMAX, VFMax);
|
||||
REGISTER_OP(VFRECP, VFRecp);
|
||||
REGISTER_OP(VFSQRT, VFSqrt);
|
||||
REGISTER_OP(VFRSQRT, VFRSqrt);
|
||||
REGISTER_OP(VNEG, VNeg);
|
||||
REGISTER_OP(VFNEG, VFNeg);
|
||||
REGISTER_OP(VNOT, VNot);
|
||||
REGISTER_OP(VUMIN, VUMin);
|
||||
REGISTER_OP(VSMIN, VSMin);
|
||||
REGISTER_OP(VUMAX, VUMax);
|
||||
REGISTER_OP(VSMAX, VSMax);
|
||||
REGISTER_OP(VZIP, VZip);
|
||||
REGISTER_OP(VZIP2, VZip2);
|
||||
REGISTER_OP(VUNZIP, VUnZip);
|
||||
REGISTER_OP(VUNZIP2, VUnZip2);
|
||||
REGISTER_OP(VTRN, VTrn);
|
||||
REGISTER_OP(VTRN2, VTrn2);
|
||||
REGISTER_OP(VBSL, VBSL);
|
||||
REGISTER_OP(VCMPEQ, VCMPEQ);
|
||||
REGISTER_OP(VCMPEQZ, VCMPEQZ);
|
||||
REGISTER_OP(VCMPGT, VCMPGT);
|
||||
REGISTER_OP(VCMPGTZ, VCMPGTZ);
|
||||
REGISTER_OP(VCMPLTZ, VCMPLTZ);
|
||||
REGISTER_OP(VFCMPEQ, VFCMPEQ);
|
||||
REGISTER_OP(VFCMPNEQ, VFCMPNEQ);
|
||||
REGISTER_OP(VFCMPLT, VFCMPLT);
|
||||
REGISTER_OP(VFCMPGT, VFCMPGT);
|
||||
REGISTER_OP(VFCMPLE, VFCMPLE);
|
||||
REGISTER_OP(VFCMPORD, VFCMPORD);
|
||||
REGISTER_OP(VFCMPUNO, VFCMPUNO);
|
||||
REGISTER_OP(VUSHL, VUShl);
|
||||
REGISTER_OP(VUSHR, VUShr);
|
||||
REGISTER_OP(VSSHR, VSShr);
|
||||
REGISTER_OP(VUSHLS, VUShlS);
|
||||
REGISTER_OP(VUSHRS, VUShrS);
|
||||
REGISTER_OP(VSSHRS, VSShrS);
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VUSHRI, VUShrI);
|
||||
REGISTER_OP(VSSHRI, VSShrI);
|
||||
REGISTER_OP(VSHLI, VShlI);
|
||||
REGISTER_OP(VUSHRNI, VUShrNI);
|
||||
REGISTER_OP(VUSHRNI2, VUShrNI2);
|
||||
REGISTER_OP(VSXTL, VSXTL);
|
||||
REGISTER_OP(VSXTL2, VSXTL2);
|
||||
REGISTER_OP(VUXTL, VUXTL);
|
||||
REGISTER_OP(VUXTL2, VUXTL2);
|
||||
REGISTER_OP(VSQXTN, VSQXTN);
|
||||
REGISTER_OP(VSQXTN2, VSQXTN2);
|
||||
REGISTER_OP(VSQXTUN, VSQXTUN);
|
||||
REGISTER_OP(VSQXTUN2, VSQXTUN2);
|
||||
REGISTER_OP(VUMUL, VMul);
|
||||
REGISTER_OP(VSMUL, VMul);
|
||||
REGISTER_OP(VUMULL, VUMull);
|
||||
REGISTER_OP(VSMULL, VSMull);
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
#undef REGISTER_OP
|
||||
|
||||
default:
|
||||
Op_Unhandled(IROp, ID);
|
||||
break;
|
||||
}
|
||||
// Execute handler
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
(this->*Handler)(IROp, ID);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockEntry),
|
||||
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
|
||||
static_cast<uint32_t>(GetCursorAddress<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
}
|
||||
@@ -1097,45 +790,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
// Add the JitCodeTail
|
||||
auto JITBlockTailLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITRIPEntries = GetCursorAddress<JITRIPReconstructEntries*>();
|
||||
|
||||
CursorIncrement(sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
ClearICache(CodeData.BlockBegin, CodeData.Size);
|
||||
auto CodeEnd = GetCursorAddress<uint8_t *>();
|
||||
ClearICache(GuestEntry, CodeEnd - GuestEntry);
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
@@ -1143,13 +799,13 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
#endif
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = CodeData.Size;
|
||||
DebugData->HostCodeSize = CodeEnd - GuestEntry;
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
|
||||
this->IR = nullptr;
|
||||
|
||||
return CodeData;
|
||||
return GuestEntry;
|
||||
}
|
||||
|
||||
void Arm64JITCore::ResetStack() {
|
||||
@@ -1168,8 +824,12 @@ void Arm64JITCore::ResetStack() {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return fextl::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
Arm64JITCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
|
||||
+54
-37
@@ -6,6 +6,7 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
@@ -13,18 +14,15 @@ $end_info$
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
@@ -33,13 +31,13 @@ namespace FEXCore::Core {
|
||||
namespace FEXCore::CPU {
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
public:
|
||||
explicit Arm64JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]] fextl::string GetName() override { return "JIT"; }
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
@@ -50,6 +48,8 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
@@ -57,12 +57,32 @@ private:
|
||||
const bool HostSupportsSVE{};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
uint64_t Entry;
|
||||
CPUBackend::CompiledCode CodeData{};
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
std::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
* @{ */
|
||||
constexpr static uint32_t NumGPRs = RA64.size();
|
||||
constexpr static uint32_t NumFPRs = RAFPR.size();
|
||||
constexpr static uint32_t NumGPRPairs = RA64Pair.size();
|
||||
constexpr static uint32_t NumCalleeGPRs = 10;
|
||||
constexpr static uint32_t NumCalleeGPRPairs = 5;
|
||||
constexpr static uint32_t RegisterCount = NumGPRs + NumFPRs + NumGPRPairs;
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
constexpr static uint64_t FPRBase = (1ULL << 32);
|
||||
constexpr static uint64_t GPRPairBase = (2ULL << 32);
|
||||
|
||||
/** @} */
|
||||
|
||||
constexpr static uint8_t RA_32 = 0;
|
||||
constexpr static uint8_t RA_64 = 1;
|
||||
constexpr static uint8_t RA_FPR = 2;
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
@@ -70,9 +90,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return StaticRegisters[Reg.Reg];
|
||||
return SRA64[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return GeneralRegisters[Reg.Reg];
|
||||
return RA64[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -84,9 +104,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return GeneralFPRegisters[Reg.Reg];
|
||||
return RAFPR[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -97,7 +117,7 @@ private:
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return GeneralPairRegisters[Reg.Reg];
|
||||
return RA64Pair[Reg.Reg];
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
@@ -202,7 +222,7 @@ private:
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
@@ -210,16 +230,23 @@ private:
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store register style.
|
||||
OpType RT_LoadRegister;
|
||||
OpType RT_StoreRegister;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
void RegisterALUHandlers();
|
||||
void RegisterAtomicHandlers();
|
||||
void RegisterBranchHandlers();
|
||||
void RegisterConversionHandlers();
|
||||
void RegisterFlagHandlers();
|
||||
void RegisterMemoryHandlers();
|
||||
void RegisterMiscHandlers();
|
||||
void RegisterMoveHandlers();
|
||||
void RegisterVectorHandlers();
|
||||
void RegisterEncryptionHandlers();
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
@@ -297,6 +324,7 @@ private:
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
DEF_OP(Jump);
|
||||
@@ -307,12 +335,10 @@ private:
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(VDupFromGPR);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
@@ -329,8 +355,6 @@ private:
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadRegisterSRA);
|
||||
DEF_OP(StoreRegisterSRA);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
@@ -341,14 +365,9 @@ private:
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(LoadMemTSO);
|
||||
DEF_OP(StoreMemTSO);
|
||||
DEF_OP(VLoadVectorMasked);
|
||||
DEF_OP(VStoreVectorMasked);
|
||||
DEF_OP(MemSet);
|
||||
DEF_OP(MemCpy);
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineClean);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
@@ -409,8 +428,6 @@ private:
|
||||
DEF_OP(VZip2);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VUnZip2);
|
||||
DEF_OP(VTrn);
|
||||
DEF_OP(VTrn2);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
|
||||
+96
-924
File diff suppressed because it is too large.
Load diff
+28
-17
@@ -4,23 +4,18 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <syscall.h>
|
||||
#endif
|
||||
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, GetCursorAddress<uint8_t*>() - CodeData.BlockBegin});
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, GetCursorAddress<uint8_t*>() - GuestEntry});
|
||||
}
|
||||
|
||||
DEF_OP(Fence) {
|
||||
@@ -60,15 +55,15 @@ DEF_OP(Break) {
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case Core::FAULT_SIGILL:
|
||||
case SIGILL:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL));
|
||||
br(TMP1);
|
||||
break;
|
||||
case Core::FAULT_SIGTRAP:
|
||||
case SIGTRAP:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
break;
|
||||
case Core::FAULT_SIGSEGV:
|
||||
case SIGSEGV:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV));
|
||||
br(TMP1);
|
||||
break;
|
||||
@@ -144,7 +139,7 @@ DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
SpillStaticRegs();
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
|
||||
@@ -162,7 +157,6 @@ DEF_OP(Print) {
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
DEF_OP(ProcessorID) {
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
@@ -170,7 +164,7 @@ DEF_OP(ProcessorID) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -213,11 +207,6 @@ DEF_OP(ProcessorID) {
|
||||
// Node is in w1
|
||||
orr(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ShiftType::LSL, 12);
|
||||
}
|
||||
#else
|
||||
DEF_OP(ProcessorID) {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#endif
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
@@ -242,5 +231,27 @@ DEF_OP(Yield) {
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(DUMMY, NoOp);
|
||||
REGISTER_OP(IRHEADER, NoOp);
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
REGISTER_OP(PHIVALUE, NoOp);
|
||||
REGISTER_OP(PRINT, Print);
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,5 +42,11 @@ DEF_OP(CreateElementPair) {
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMoveHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+272
-351
File diff suppressed because it is too large.
Load diff
+6
-5
@@ -1,10 +1,9 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
@@ -14,12 +13,14 @@ struct InternalThreadState;
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetX86JITBackendFeatures();
|
||||
|
||||
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -5,7 +5,6 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -13,6 +12,7 @@ $end_info$
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
|
||||
@@ -5,7 +5,6 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -13,6 +12,7 @@ $end_info$
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
+13
-45
@@ -7,7 +7,6 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
@@ -26,10 +25,20 @@ $end_info$
|
||||
#include <stdint.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize); // + 8 to consume return address
|
||||
}
|
||||
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)]);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
@@ -146,7 +155,6 @@ DEF_OP(Syscall) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
// XXX: This is very terrible, but I don't care for right now
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
@@ -187,11 +195,7 @@ DEF_OP(Syscall) {
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
@@ -207,7 +211,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
mov(rdi, GetSrc<RA_64>(Op->ArgPtr.ID()));
|
||||
|
||||
auto thunkFn = static_cast<Context::ContextImpl*>(ThreadState->CTX)->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
|
||||
mov(rax, reinterpret_cast<uintptr_t>(thunkFn));
|
||||
call(rax);
|
||||
@@ -307,45 +311,10 @@ DEF_OP(CPUID) {
|
||||
mov(Dst.second, rdx);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
push(Reg);
|
||||
|
||||
// CPUID ABI
|
||||
// this: rdi
|
||||
// Function: rsi
|
||||
//
|
||||
// Result: RAX, RDX. 4xi32
|
||||
|
||||
// rsi can be in the source registers, so copy argument to edx first
|
||||
mov (esi, GetSrc<RA_32>(Op->Function.ID()));
|
||||
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)]);
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
mov(Dst.first.cvt32(), eax);
|
||||
mov(Dst.second, rax);
|
||||
shr(Dst.second, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
REGISTER_OP(JUMP, Jump);
|
||||
@@ -355,7 +324,6 @@ void X86JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -5,12 +5,13 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -110,53 +111,6 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VDupFromGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VDupFromGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Src = GetSrc<RA_64>(Op->Src.ID()).cvt64();
|
||||
|
||||
vmovq(Dst, Src);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
if (Is256Bit) {
|
||||
vpbroadcastb(ToYMM(Dst), Dst);
|
||||
} else {
|
||||
vpbroadcastb(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
case 2:
|
||||
if (Is256Bit) {
|
||||
vpbroadcastw(ToYMM(Dst), Dst);
|
||||
} else {
|
||||
vpbroadcastw(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
case 4:
|
||||
if (Is256Bit) {
|
||||
vpbroadcastd(ToYMM(Dst), Dst);
|
||||
} else {
|
||||
vpbroadcastd(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
if (Is256Bit) {
|
||||
vpbroadcastq(ToYMM(Dst), Dst);
|
||||
} else {
|
||||
vpbroadcastq(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled element size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
@@ -403,7 +357,6 @@ void X86JITCore::RegisterConversionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(VDUPFROMGPR, VDupFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
|
||||
@@ -5,11 +5,12 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
@@ -20,67 +21,23 @@ DEF_OP(AESImc) {
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Key = GetSrc(Op->Key.ID());
|
||||
const auto State = GetSrc(Op->State.ID());
|
||||
|
||||
if (Is256Bit) {
|
||||
vaesenc(ToYMM(Dst), ToYMM(State), ToYMM(Key));
|
||||
} else {
|
||||
vaesenc(Dst, State, Key);
|
||||
}
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
vaesenc(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Key = GetSrc(Op->Key.ID());
|
||||
const auto State = GetSrc(Op->State.ID());
|
||||
|
||||
if (Is256Bit) {
|
||||
vaesenclast(ToYMM(Dst), ToYMM(State), ToYMM(Key));
|
||||
} else {
|
||||
vaesenclast(Dst, State, Key);
|
||||
}
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
vaesenclast(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Key = GetSrc(Op->Key.ID());
|
||||
const auto State = GetSrc(Op->State.ID());
|
||||
|
||||
if (Is256Bit) {
|
||||
vaesdec(ToYMM(Dst), ToYMM(State), ToYMM(Key));
|
||||
} else {
|
||||
vaesdec(Dst, State, Key);
|
||||
}
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
vaesdec(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Key = GetSrc(Op->Key.ID());
|
||||
const auto State = GetSrc(Op->State.ID());
|
||||
|
||||
if (Is256Bit) {
|
||||
vaesdeclast(ToYMM(Dst), ToYMM(State), ToYMM(Key));
|
||||
} else {
|
||||
vaesdeclast(Dst, State, Key);
|
||||
}
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
vaesdeclast(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
@@ -119,24 +76,18 @@ DEF_OP(CRC32) {
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Src1 = GetSrc(Op->Src1.ID());
|
||||
const auto Src2 = GetSrc(Op->Src2.ID());
|
||||
auto Dst = GetDst(Node);
|
||||
auto Src1 = GetSrc(Op->Src1.ID());
|
||||
auto Src2 = GetSrc(Op->Src2.ID());
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
case 0b00000001:
|
||||
case 0b00010000:
|
||||
case 0b00010001:
|
||||
if (Is256Bit) {
|
||||
vpclmulqdq(ToYMM(Dst), ToYMM(Src1), ToYMM(Src2), Op->Selector);
|
||||
} else {
|
||||
vpclmulqdq(Dst, Src1, Src2, Op->Selector);
|
||||
}
|
||||
vpclmulqdq(Dst, Src1, Src2, Op->Selector);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
|
||||
@@ -5,12 +5,12 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
|
||||
+31
-137
@@ -19,6 +19,7 @@ $end_info$
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
@@ -27,7 +28,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
@@ -35,9 +35,12 @@ $end_info$
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <signal.h>
|
||||
#include <sys/mman.h>
|
||||
#include <tuple>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
// #define DEBUG_RA 1
|
||||
// #define DEBUG_CYCLES
|
||||
@@ -144,12 +147,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_F80_I32: {
|
||||
PushRegs();
|
||||
|
||||
if (Info.ABI == FABI_F80_I16) {
|
||||
movsx(rdi, GetSrc<RA_32>(IROp->Args[0].ID()).cvt16());
|
||||
}
|
||||
else {
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
@@ -225,7 +223,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
PopRegs();
|
||||
|
||||
movsx(GetDst<RA_64>(Node), ax);
|
||||
movzx(GetDst<RA_64>(Node), ax);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
@@ -304,63 +302,6 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I32_I64_I64_I128_I128_I16: {
|
||||
PushRegs();
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto LHS = GetSrc(Op->LHS.ID());
|
||||
const auto RHS = GetSrc(Op->RHS.ID());
|
||||
const auto SrcRAX = GetSrc<RA_64>(Op->RAX.ID());
|
||||
const auto SrcRDX = GetSrc<RA_64>(Op->RDX.ID());
|
||||
|
||||
mov(rdi, SrcRAX);
|
||||
mov(rsi, SrcRDX);
|
||||
|
||||
movq(rdx, LHS);
|
||||
pextrq(rcx, LHS, 1);
|
||||
|
||||
movq(r8, RHS);
|
||||
pextrq(r9, RHS, 1);
|
||||
|
||||
sub(rsp, 16);
|
||||
mov(dword [rsp], Control);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
add(rsp, 16);
|
||||
PopRegs();
|
||||
|
||||
mov(GetDst<RA_32>(Node), rax);
|
||||
break;
|
||||
}
|
||||
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
PushRegs();
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
|
||||
const auto LHS = GetSrc(Op->LHS.ID());
|
||||
const auto RHS = GetSrc(Op->RHS.ID());
|
||||
const auto Control = Op->Control;
|
||||
|
||||
movq(rdi, LHS);
|
||||
pextrq(rsi, LHS, 1);
|
||||
|
||||
movq(rdx, RHS);
|
||||
pextrq(rcx, RHS, 1);
|
||||
|
||||
mov(r8, Control);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
mov(GetDst<RA_32>(Node), rax);
|
||||
break;
|
||||
}
|
||||
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -384,7 +325,7 @@ static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame,
|
||||
}
|
||||
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
@@ -396,14 +337,14 @@ static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame,
|
||||
void X86JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
X86JITCore::X86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, CodeGenerator(0, this, nullptr) // this is not used here
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
RAPass->AllocateRegisterSet(RegisterClasses);
|
||||
RAPass->AllocateRegisterSet(RegisterCount, RegisterClasses);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, NumGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, NumXMMs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, NumGPRPairs);
|
||||
@@ -433,7 +374,7 @@ X86JITCore::X86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::Intern
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
@@ -441,14 +382,9 @@ X86JITCore::X86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::Intern
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<X86JITCore_ExitFunctionLink>);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::Context::ThreadExitFunctionLink<X86JITCore_ExitFunctionLink>);
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
|
||||
@@ -458,6 +394,12 @@ X86JITCore::X86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::Intern
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void X86JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
}
|
||||
|
||||
X86JITCore::~X86JITCore() {
|
||||
|
||||
}
|
||||
@@ -640,7 +582,7 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
|
||||
return { &CodeGenerator::sete , &CodeGenerator::cmove , &CodeGenerator::je };
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("x86::CompileCode");
|
||||
JumpTargets.clear();
|
||||
@@ -656,27 +598,12 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
CodeData.BlockBegin = getCurr<uint8_t*>();
|
||||
|
||||
// Put the code header at the start of the data block.
|
||||
Label JITCodeHeaderLabel{};
|
||||
L(JITCodeHeaderLabel);
|
||||
|
||||
JITCodeHeader *CodeHeader = getCurr<JITCodeHeader *>();
|
||||
setSize(getSize() + sizeof(JITCodeHeader));
|
||||
|
||||
CodeData.BlockEntry = getCurr<uint8_t*>();
|
||||
|
||||
// Get the address of the JITCodeHeader and store in to the core state.
|
||||
// Only two instructions, so very low overhead.
|
||||
lea(TMP1, ptr [rip + JITCodeHeaderLabel]);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader)], TMP1);
|
||||
|
||||
GuestEntry = getCurr<uint8_t*>();
|
||||
CursorEntry = getSize();
|
||||
this->IR = IR;
|
||||
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(CodeData.BlockBegin, Entry);
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
setSize(getSize() + GDBSize);
|
||||
}
|
||||
|
||||
@@ -759,7 +686,7 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
if (IROp->Op != IR::OP_BEGINBLOCK &&
|
||||
IROp->Op != IR::OP_CONDJUMP &&
|
||||
IROp->Op != IR::OP_JUMP) {
|
||||
fextl::stringstream Inst;
|
||||
std::stringstream Inst;
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
|
||||
if (IROp->HasDest) {
|
||||
@@ -799,7 +726,7 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockBegin),
|
||||
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
|
||||
static_cast<uint32_t>(getCurr<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
}
|
||||
@@ -812,62 +739,29 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
// Add the JitCodeTail
|
||||
auto JITBlockTailLocation = getCurr<uint8_t *>();
|
||||
auto JITBlockTail = getCurr<JITCodeTail*>();
|
||||
setSize(getSize() + sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = getCurr<uint8_t *>();
|
||||
auto JITRIPEntries = getCurr<JITRIPReconstructEntries*>();
|
||||
|
||||
setSize(getSize() + sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = getCurr<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
void *GuestExit = getCurr<void*>();
|
||||
this->IR = nullptr;
|
||||
|
||||
ready();
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = CodeData.Size;
|
||||
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(GuestExit) - reinterpret_cast<uintptr_t>(GuestEntry);
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
|
||||
return CodeData;
|
||||
return GuestEntry;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return fextl::make_unique<X86JITCore>(ctx, Thread);
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetX86JITBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
}
|
||||
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
X86JITCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
}
|
||||
+18
-22
@@ -9,20 +9,18 @@ $end_info$
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#define XBYAK64
|
||||
#include <xbyak/xbyak.h>
|
||||
#include <xbyak/xbyak_util.h>
|
||||
|
||||
using namespace Xbyak;
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <tuple>
|
||||
@@ -53,13 +51,13 @@ const std::array<Xbyak::Xmm, 11> RAXMM_x = { xmm1, xmm2, xmm3, xmm4, xmm5, xmm6
|
||||
|
||||
class X86JITCore final : public CPUBackend, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
explicit X86JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
explicit X86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~X86JITCore() override;
|
||||
|
||||
[[nodiscard]] fextl::string GetName() override { return "JIT"; }
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
@@ -70,6 +68,8 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
|
||||
private:
|
||||
@@ -123,7 +123,7 @@ private:
|
||||
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
@@ -135,12 +135,11 @@ private:
|
||||
/** @} */
|
||||
|
||||
Label* PendingTargetLabel{};
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
uint64_t Entry;
|
||||
CPUBackend::CompiledCode CodeData{};
|
||||
|
||||
fextl::unordered_map<IR::NodeID, Label> JumpTargets;
|
||||
std::unordered_map<IR::NodeID, Label> JumpTargets;
|
||||
Xbyak::util::Cpu Features{};
|
||||
|
||||
bool MemoryDebug = false;
|
||||
@@ -151,6 +150,7 @@ private:
|
||||
constexpr static uint32_t NumGPRs = RA64.size(); // 4 is the minimum required for GPR ops
|
||||
constexpr static uint32_t NumXMMs = RAXMM.size();
|
||||
constexpr static uint32_t NumGPRPairs = RA64Pair.size();
|
||||
constexpr static uint32_t RegisterCount = NumGPRs + NumXMMs + NumGPRPairs;
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
@@ -205,6 +205,10 @@ private:
|
||||
void EmitDetectionString();
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using SetCC = void (X86JITCore::*)(const Operand& op);
|
||||
using CMovCC = void (X86JITCore::*)(const Reg& reg, const Operand& op);
|
||||
@@ -304,6 +308,7 @@ private:
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
DEF_OP(Jump);
|
||||
@@ -313,12 +318,10 @@ private:
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(VDupFromGPR);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_UToF);
|
||||
@@ -344,12 +347,7 @@ private:
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(VLoadVectorMasked);
|
||||
DEF_OP(VStoreVectorMasked);
|
||||
DEF_OP(MemSet);
|
||||
DEF_OP(MemCpy);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineClean);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
@@ -410,8 +408,6 @@ private:
|
||||
DEF_OP(VZip2);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VUnZip2);
|
||||
DEF_OP(VTrn);
|
||||
DEF_OP(VTrn2);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
|
||||
+3
-315
@@ -6,7 +6,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -14,6 +14,7 @@ $end_info$
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -765,320 +766,12 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadVectorMasked) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Mask = GetSrc(Op->Mask.ID());
|
||||
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4: {
|
||||
if (Is256Bit) {
|
||||
vmaskmovps(ToYMM(Dst), ToYMM(Mask), yword [MemPtr]);
|
||||
} else {
|
||||
vmaskmovps(Dst, Mask, xword [MemPtr]);
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 8: {
|
||||
if (Is256Bit) {
|
||||
vmaskmovpd(ToYMM(Dst), ToYMM(Mask), yword [MemPtr]);
|
||||
} else {
|
||||
vmaskmovpd(Dst, Mask, xword [MemPtr]);
|
||||
}
|
||||
return;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VLoadVectorMasked element size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
DEF_OP(VStoreVectorMasked) {
|
||||
const auto Op = IROp->C<IR::IROp_VStoreVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Data = GetDst(Op->Data.ID());
|
||||
const auto Mask = GetSrc(Op->Mask.ID());
|
||||
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4: {
|
||||
if (Is256Bit) {
|
||||
vmaskmovps(yword [MemPtr], ToYMM(Mask), ToYMM(Data));
|
||||
} else {
|
||||
vmaskmovps(xword [MemPtr], Mask, Data);
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 8: {
|
||||
if (Is256Bit) {
|
||||
vmaskmovpd(yword [MemPtr], ToYMM(Mask), ToYMM(Data));
|
||||
} else {
|
||||
vmaskmovpd(xword [MemPtr], Mask, Data);
|
||||
}
|
||||
return;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled VStoreVectorMasked element size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(MemSet) {
|
||||
const auto Op = IROp->C<IR::IROp_MemSet>();
|
||||
|
||||
const int32_t Size = Op->Size;
|
||||
const auto MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto Value = GetSrc<RA_64>(Op->Value.ID());
|
||||
const auto Length = GetSrc<RA_64>(Op->Length.ID());
|
||||
const auto Direction = GetSrc<RA_64>(Op->Direction.ID());
|
||||
const auto Dst = GetSrc<RA_64>(Node);
|
||||
|
||||
// If Direction == 0 then:
|
||||
// MemReg is incremented (by size)
|
||||
// else:
|
||||
// MemReg is decremented (by size)
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
// TMP1 = rax
|
||||
// TMP2 = rcx
|
||||
// TMP4 = rdi
|
||||
// That leaves us with TMP3 and TMP5
|
||||
mov(rax, Value);
|
||||
mov(rcx, Length);
|
||||
mov(rdi, MemReg);
|
||||
|
||||
if (!Op->Prefix.IsInvalid()) {
|
||||
add(rdi, GetSrc<RA_64>(Op->Prefix.ID()));
|
||||
}
|
||||
|
||||
{
|
||||
mov(TMP3, Length);
|
||||
auto CalculateDest = [&]() {
|
||||
mov(Dst, MemReg);
|
||||
switch (Size) {
|
||||
case 1:
|
||||
break;
|
||||
case 2:
|
||||
shl(TMP3, 1);
|
||||
break;
|
||||
case 4:
|
||||
shl(TMP3, 2);
|
||||
break;
|
||||
case 8:
|
||||
shl(TMP3, 3);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
Label AfterDir;
|
||||
Label BackwardDir;
|
||||
|
||||
cmp(Direction, 0);
|
||||
jne(BackwardDir);
|
||||
// Incrementing DF flag.
|
||||
cld();
|
||||
CalculateDest();
|
||||
add(Dst, TMP3);
|
||||
jmp(AfterDir);
|
||||
|
||||
L(BackwardDir);
|
||||
// Decrementing DF flag.
|
||||
std();
|
||||
CalculateDest();
|
||||
sub(Dst, TMP3);
|
||||
|
||||
L(AfterDir);
|
||||
}
|
||||
|
||||
switch (Size) {
|
||||
case 1:
|
||||
rep(); stosb();
|
||||
break;
|
||||
case 2:
|
||||
rep(); stosw();
|
||||
break;
|
||||
case 4:
|
||||
rep(); stosd();
|
||||
break;
|
||||
case 8:
|
||||
rep(); stosq();
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
// Ensure we set DF back to zero. Required by the ABI.
|
||||
cld();
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
const auto Op = IROp->C<IR::IROp_MemCpy>();
|
||||
|
||||
const int32_t Size = Op->Size;
|
||||
const auto MemRegDest = GetSrc<RA_64>(Op->AddrDest.ID());
|
||||
const auto MemRegSrc = GetSrc<RA_64>(Op->AddrSrc.ID());
|
||||
|
||||
const auto Length = GetSrc<RA_64>(Op->Length.ID());
|
||||
const auto Direction = GetSrc<RA_64>(Op->Direction.ID());
|
||||
|
||||
// If Direction == 0 then:
|
||||
// MemRegDest is incremented (by size)
|
||||
// MemRegSrc is incremented (by size)
|
||||
// else:
|
||||
// MemRegDest is decremented (by size)
|
||||
// MemRegSrc is decremented (by size)
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
// TMP1 = Length
|
||||
// TMP2 = Dest
|
||||
// TMP3 = Src
|
||||
// TMP4 = Temp value
|
||||
mov(TMP1, Length);
|
||||
mov(TMP2, MemRegDest);
|
||||
mov(TMP3, MemRegSrc);
|
||||
if (!Op->PrefixDest.IsInvalid()) {
|
||||
add(TMP2, GetSrc<RA_64>(Op->PrefixDest.ID()));
|
||||
}
|
||||
if (!Op->PrefixSrc.IsInvalid()) {
|
||||
add(TMP3, GetSrc<RA_64>(Op->PrefixSrc.ID()));
|
||||
}
|
||||
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
Label Done;
|
||||
Label BackwardImpl;
|
||||
cmp(Direction, 0);
|
||||
jne(BackwardImpl);
|
||||
|
||||
// Emit forward direction memcpy then backward direction memcpy.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
Label DoneInternal;
|
||||
Label AgainInternal;
|
||||
|
||||
L(AgainInternal);
|
||||
cmp(TMP1, 0);
|
||||
je(DoneInternal);
|
||||
|
||||
{
|
||||
switch (Size) {
|
||||
case 1:
|
||||
movzx(TMP4, byte [TMP3]);
|
||||
mov(byte [TMP2], TMP4.cvt8());
|
||||
break;
|
||||
case 2:
|
||||
movzx(TMP4, word [TMP3]);
|
||||
mov(word [TMP2], TMP4.cvt16());
|
||||
break;
|
||||
case 4:
|
||||
mov(TMP4.cvt32(), dword [TMP3]);
|
||||
mov(dword [TMP2], TMP4.cvt32());
|
||||
break;
|
||||
case 8:
|
||||
mov(TMP4, qword [TMP3]);
|
||||
mov(qword [TMP2], TMP4);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Direction == 1) {
|
||||
// Incrementing pointers
|
||||
add(TMP2, Size);
|
||||
add(TMP3, Size);
|
||||
}
|
||||
else {
|
||||
// Decrementing pointers
|
||||
sub(TMP2, Size);
|
||||
sub(TMP3, Size);
|
||||
}
|
||||
|
||||
// Decrement counter by one
|
||||
sub(TMP1, 1);
|
||||
|
||||
jmp(AgainInternal);
|
||||
L(DoneInternal);
|
||||
|
||||
// Pointer math using source pointers and length.
|
||||
mov(TMP3, Length);
|
||||
switch (Size) {
|
||||
case 1:
|
||||
break;
|
||||
case 2:
|
||||
shl(TMP3, 1);
|
||||
break;
|
||||
case 4:
|
||||
shl(TMP3, 2);
|
||||
break;
|
||||
case 8:
|
||||
shl(TMP3, 3);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size);
|
||||
break;
|
||||
}
|
||||
|
||||
// Needs to use temporaries just in case of overwrite
|
||||
mov(TMP1, MemRegDest);
|
||||
mov(TMP2, MemRegSrc);
|
||||
|
||||
mov(Dst.first, TMP1);
|
||||
mov(Dst.second, TMP2);
|
||||
|
||||
if (Direction == 1) {
|
||||
// Incrementing pointers
|
||||
add(Dst.first, TMP3);
|
||||
add(Dst.second, TMP3);
|
||||
|
||||
jmp(Done);
|
||||
L(BackwardImpl);
|
||||
}
|
||||
else {
|
||||
// Decrementing pointers
|
||||
sub(Dst.first, TMP3);
|
||||
sub(Dst.second, TMP3);
|
||||
}
|
||||
}
|
||||
|
||||
L(Done);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (Op->Serialize) {
|
||||
clflush(ptr [MemReg]);
|
||||
}
|
||||
else {
|
||||
clflushopt(ptr [MemReg]);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClean) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClean>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
clwb(ptr [MemReg]);
|
||||
clflush(ptr [MemReg]);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
@@ -1115,12 +808,7 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP(LOADMEMTSO, LoadMem);
|
||||
REGISTER_OP(STOREMEMTSO, StoreMem);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
REGISTER_OP(MEMSET, MemSet);
|
||||
REGISTER_OP(MEMCPY, MemCpy);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINECLEAN, CacheLineClean);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
@@ -6,7 +6,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
@@ -17,6 +16,7 @@ $end_info$
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
@@ -24,7 +24,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, getCurr<uint8_t*>() - CodeData.BlockBegin});
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, getCurr<uint8_t*>() - GuestEntry});
|
||||
}
|
||||
|
||||
DEF_OP(Fence) {
|
||||
@@ -43,7 +43,6 @@ DEF_OP(Fence) {
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
|
||||
@@ -80,11 +79,6 @@ DEF_OP(Break) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
#else
|
||||
DEF_OP(Break) {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#endif
|
||||
|
||||
DEF_OP(GetRoundingMode) {
|
||||
auto Dst = GetDst<RA_32>(Node);
|
||||
|
||||
@@ -5,7 +5,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
|
||||
+6
-284
@@ -5,13 +5,14 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -1943,7 +1944,7 @@ DEF_OP(VUnZip2) {
|
||||
}
|
||||
case 8: {
|
||||
if (Is256Bit) {
|
||||
vshufpd(ToYMM(Dst), ToYMM(VectorLower), ToYMM(VectorUpper), 0b11'11);
|
||||
vshufpd(ToYMM(Dst), ToYMM(VectorLower), ToYMM(VectorUpper), 0b1'1);
|
||||
vpermq(ToYMM(Dst), ToYMM(Dst), 0b11'01'10'00);
|
||||
} else {
|
||||
vshufpd(Dst, VectorLower, VectorUpper, 0b1'1);
|
||||
@@ -1957,191 +1958,6 @@ DEF_OP(VUnZip2) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VTrn) {
|
||||
const auto Op = IROp->C<IR::IROp_VTrn>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto VectorLower = GetSrc(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetSrc(Op->VectorUpper.ID());
|
||||
|
||||
const auto LoadPshufbReg = [&](Xbyak::Xmm reg, uint64_t lower) {
|
||||
mov(rax, lower);
|
||||
mov(rcx, 0x80'80'80'80'80'80'80'80);
|
||||
vmovq(reg, rax);
|
||||
pinsrq(reg, rcx, 1);
|
||||
};
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
LoadPshufbReg(xmm15, 0x0E'0C'0A'08'06'04'02'00);
|
||||
|
||||
if (Is256Bit) {
|
||||
vinserti128(ymm15, ymm15, xmm15, 1);
|
||||
|
||||
vpshufb(ymm14, ToYMM(VectorLower), ymm15);
|
||||
vpshufb(ymm13, ToYMM(VectorUpper), ymm15);
|
||||
|
||||
vpunpcklbw(ToYMM(Dst), ymm14, ymm13);
|
||||
} else {
|
||||
vpshufb(xmm14, VectorLower, xmm15);
|
||||
vpshufb(xmm13, VectorUpper, xmm15);
|
||||
vpunpcklbw(Dst, xmm14, xmm13);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
LoadPshufbReg(xmm15, 0x0D'0C'09'08'05'04'01'00);
|
||||
|
||||
if (Is256Bit) {
|
||||
vinserti128(ymm15, ymm15, xmm15, 1);
|
||||
|
||||
vpshufb(ymm14, ToYMM(VectorLower), ymm15);
|
||||
vpshufb(ymm13, ToYMM(VectorUpper), ymm15);
|
||||
|
||||
vpunpcklwd(ToYMM(Dst), ymm14, ymm13);
|
||||
} else {
|
||||
vpshufb(xmm14, VectorLower, xmm15);
|
||||
vpshufb(xmm13, VectorUpper, xmm15);
|
||||
vpunpcklwd(Dst, xmm14, xmm13);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
LoadPshufbReg(xmm15, 0x0B'0A'09'08'03'02'01'00);
|
||||
|
||||
if (Is256Bit) {
|
||||
vinserti128(ymm15, ymm15, xmm15, 1);
|
||||
|
||||
vpshufb(ymm14, ToYMM(VectorLower), ymm15);
|
||||
vpshufb(ymm13, ToYMM(VectorUpper), ymm15);
|
||||
|
||||
vpunpckldq(ToYMM(Dst), ymm14, ymm13);
|
||||
} else {
|
||||
vpshufb(xmm14, VectorLower, xmm15);
|
||||
vpshufb(xmm13, VectorUpper, xmm15);
|
||||
vpunpckldq(Dst, xmm14, xmm13);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
LoadPshufbReg(xmm15, 0x07'06'05'04'03'02'01'00);
|
||||
|
||||
if (Is256Bit) {
|
||||
vinserti128(ymm15, ymm15, xmm15, 1);
|
||||
|
||||
vpshufb(ymm14, ToYMM(VectorLower), ymm15);
|
||||
vpshufb(ymm13, ToYMM(VectorUpper), ymm15);
|
||||
|
||||
vpunpcklqdq(ToYMM(Dst), ymm14, ymm13);
|
||||
} else {
|
||||
vpshufb(xmm14, VectorLower, xmm15);
|
||||
vpshufb(xmm13, VectorUpper, xmm15);
|
||||
vpunpcklqdq(Dst, xmm14, xmm13);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VTrn2) {
|
||||
const auto Op = IROp->C<IR::IROp_VTrn2>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto VectorLower = GetSrc(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetSrc(Op->VectorUpper.ID());
|
||||
|
||||
const auto LoadPshufbReg = [&](Xbyak::Xmm reg, uint64_t lower) {
|
||||
mov(rax, lower);
|
||||
mov(rcx, 0x80'80'80'80'80'80'80'80);
|
||||
vmovq(reg, rax);
|
||||
pinsrq(reg, rcx, 1);
|
||||
};
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
LoadPshufbReg(xmm15, 0x0F'0D'0B'09'07'05'03'01);
|
||||
|
||||
if (Is256Bit) {
|
||||
vinserti128(ymm15, ymm15, xmm15, 1);
|
||||
|
||||
vpshufb(ymm14, ToYMM(VectorLower), ymm15);
|
||||
vpshufb(ymm13, ToYMM(VectorUpper), ymm15);
|
||||
|
||||
vpunpcklbw(ToYMM(Dst), ymm14, ymm13);
|
||||
} else {
|
||||
vpshufb(xmm14, VectorLower, xmm15);
|
||||
vpshufb(xmm13, VectorUpper, xmm15);
|
||||
vpunpcklbw(Dst, xmm14, xmm13);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
LoadPshufbReg(xmm15, 0x0F'0E'0B'0A'07'06'03'02);
|
||||
|
||||
if (Is256Bit) {
|
||||
vinserti128(ymm15, ymm15, xmm15, 1);
|
||||
|
||||
vpshufb(ymm14, ToYMM(VectorLower), ymm15);
|
||||
vpshufb(ymm13, ToYMM(VectorUpper), ymm15);
|
||||
|
||||
vpunpcklwd(ToYMM(Dst), ymm14, ymm13);
|
||||
} else {
|
||||
vpshufb(xmm14, VectorLower, xmm15);
|
||||
vpshufb(xmm13, VectorUpper, xmm15);
|
||||
vpunpcklwd(Dst, xmm14, xmm13);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
LoadPshufbReg(xmm15, 0x0F'0E'0D'0C'07'06'05'04);
|
||||
|
||||
if (Is256Bit) {
|
||||
vinserti128(ymm15, ymm15, xmm15, 1);
|
||||
|
||||
vpshufb(ymm14, ToYMM(VectorLower), ymm15);
|
||||
vpshufb(ymm13, ToYMM(VectorUpper), ymm15);
|
||||
|
||||
vpunpckldq(ToYMM(Dst), ymm14, ymm13);
|
||||
} else {
|
||||
vpshufb(xmm14, VectorLower, xmm15);
|
||||
vpshufb(xmm13, VectorUpper, xmm15);
|
||||
vpunpckldq(Dst, xmm14, xmm13);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
LoadPshufbReg(xmm15, 0x0F'0E'0D'0C'0B'0A'09'08);
|
||||
|
||||
if (Is256Bit) {
|
||||
vinserti128(ymm15, ymm15, xmm15, 1);
|
||||
|
||||
vpshufb(ymm14, ToYMM(VectorLower), ymm15);
|
||||
vpshufb(ymm13, ToYMM(VectorUpper), ymm15);
|
||||
|
||||
vpunpcklqdq(ToYMM(Dst), ymm14, ymm13);
|
||||
} else {
|
||||
vpshufb(xmm14, VectorLower, xmm15);
|
||||
vpshufb(xmm13, VectorUpper, xmm15);
|
||||
vpunpcklqdq(Dst, xmm14, xmm13);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VBSL) {
|
||||
const auto Op = IROp->C<IR::IROp_VBSL>();
|
||||
@@ -2725,90 +2541,15 @@ DEF_OP(VFCMPUNO) {
|
||||
}
|
||||
|
||||
DEF_OP(VUShl) {
|
||||
const auto Op = IROp->C<IR::IROp_VUShl>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 4 || ElementSize == 8,
|
||||
"VUShl only supports 32-bit and 64-bit elements");
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto ShiftVector = GetSrc(Op->ShiftVector.ID());
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
if (Is256Bit) {
|
||||
vpsllvd(ToYMM(Dst), ToYMM(Vector), ToYMM(ShiftVector));
|
||||
} else {
|
||||
vpsllvd(Dst, Vector, ShiftVector);
|
||||
}
|
||||
return;
|
||||
case 8:
|
||||
if (Is256Bit) {
|
||||
vpsllvq(ToYMM(Dst), ToYMM(Vector), ToYMM(ShiftVector));
|
||||
} else {
|
||||
vpsllvq(Dst, Vector, ShiftVector);
|
||||
}
|
||||
return;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(VUShr) {
|
||||
const auto Op = IROp->C<IR::IROp_VUShr>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 4 || ElementSize == 8,
|
||||
"VUShr only supports 32-bit and 64-bit elements");
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto ShiftVector = GetSrc(Op->ShiftVector.ID());
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
if (Is256Bit) {
|
||||
vpsrlvd(ToYMM(Dst), ToYMM(Vector), ToYMM(ShiftVector));
|
||||
} else {
|
||||
vpsrlvd(Dst, Vector, ShiftVector);
|
||||
}
|
||||
return;
|
||||
case 8:
|
||||
if (Is256Bit) {
|
||||
vpsrlvq(ToYMM(Dst), ToYMM(Vector), ToYMM(ShiftVector));
|
||||
} else {
|
||||
vpsrlvq(Dst, Vector, ShiftVector);
|
||||
}
|
||||
return;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
return;
|
||||
}
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(VSShr) {
|
||||
const auto Op = IROp->C<IR::IROp_VSShr>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 4, "VSShr only supports 32-bit elements");
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto ShiftVector = GetSrc(Op->ShiftVector.ID());
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (Is256Bit) {
|
||||
vpsravd(ToYMM(Dst), ToYMM(Vector), ToYMM(ShiftVector));
|
||||
} else {
|
||||
vpsravd(Dst, Vector, ShiftVector);
|
||||
}
|
||||
LOGMAN_MSG_A_FMT("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(VUShlS) {
|
||||
@@ -3391,23 +3132,6 @@ DEF_OP(VShlI) {
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
const auto Mask = 0xFFU >> BitShift;
|
||||
|
||||
mov(rax, Mask);
|
||||
vmovq(xmm15, rax);
|
||||
|
||||
if (Is256Bit) {
|
||||
vpsllw(ToYMM(Dst), ToYMM(Vector), BitShift);
|
||||
vpbroadcastb(ymm15, xmm15);
|
||||
vpand(ToYMM(Dst), ToYMM(Dst), ymm15);
|
||||
} else {
|
||||
vpsllw(Dst, Vector, BitShift);
|
||||
vpbroadcastb(xmm15, xmm15);
|
||||
vpand(Dst, Dst, ymm15);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
if (Is256Bit) {
|
||||
vpsllw(ToYMM(Dst), ToYMM(Vector), BitShift);
|
||||
@@ -4565,8 +4289,6 @@ void X86JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VZIP2, VZip2);
|
||||
REGISTER_OP(VUNZIP, VUnZip);
|
||||
REGISTER_OP(VUNZIP2, VUnZip2);
|
||||
REGISTER_OP(VTRN, VTrn);
|
||||
REGISTER_OP(VTRN2, VTrn2);
|
||||
REGISTER_OP(VBSL, VBSL);
|
||||
REGISTER_OP(VCMPEQ, VCMPEQ);
|
||||
REGISTER_OP(VCMPEQZ, VCMPEQZ);
|
||||
|
||||
+12
-10
@@ -11,15 +11,15 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include <sys/mman.h>
|
||||
|
||||
namespace FEXCore {
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl *CTX)
|
||||
: BlockLinks_mbr { fextl::pmr::get_default_resource() }
|
||||
, ctx {CTX} {
|
||||
LookupCache::LookupCache(FEXCore::Context::Context *CTX)
|
||||
: ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
|
||||
// Setup our PMR map.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
BlockLinks = BlockLinks_pma.new_object<BlockLinksMapType>();
|
||||
|
||||
// Block cache ends up looking like this
|
||||
// PageMemoryMap[VirtualMemoryRegion >> 12]
|
||||
@@ -33,7 +33,7 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl *CTX)
|
||||
// Allocate a region of memory that we can use to back our block pointers
|
||||
// We need one pointer per page of virtual memory
|
||||
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize));
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, TotalCacheSize, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
|
||||
// Allocate our memory backing our pages
|
||||
// We need 32KB per guest page (One pointer per byte)
|
||||
@@ -52,7 +52,7 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl *CTX)
|
||||
|
||||
LookupCache::~LookupCache() {
|
||||
const size_t TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
|
||||
// No need to free BlockLinks map.
|
||||
// These will get freed when their memory allocators are deallocated.
|
||||
@@ -62,7 +62,7 @@ void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE);
|
||||
madvise(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, MADV_DONTNEED);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
@@ -70,9 +70,11 @@ void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
|
||||
madvise(reinterpret_cast<void*>(PagePointer), TotalCacheSize, MADV_DONTNEED);
|
||||
// Clear the BlockLinks allocator which frees the BlockLinks map implicitly.
|
||||
BlockLinks_mbr.release();
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
BlockLinks = BlockLinks_pma.new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
BlockList.clear();
|
||||
}
|
||||
|
||||
+15
-11
@@ -1,28 +1,30 @@
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
#include <memory_resource>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#include <mutex>
|
||||
#include <tsl/robin_map.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
class LookupCache {
|
||||
public:
|
||||
|
||||
struct LookupCacheEntry {
|
||||
uintptr_t HostCode;
|
||||
uintptr_t GuestCode;
|
||||
};
|
||||
|
||||
LookupCache(FEXCore::Context::ContextImpl *CTX);
|
||||
LookupCache(FEXCore::Context::Context *CTX);
|
||||
~LookupCache();
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
@@ -67,7 +69,7 @@ public:
|
||||
return 0;
|
||||
}
|
||||
|
||||
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
||||
std::map<uint64_t, std::vector<uint64_t>> CodePages;
|
||||
|
||||
// Appends Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
@@ -168,6 +170,8 @@ public:
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
@@ -243,10 +247,10 @@ private:
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, std::function<void()>>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
std::pmr::polymorphic_allocator<std::byte> BlockLinks_pma {&BlockLinks_mbr};
|
||||
BlockLinksMapType *BlockLinks;
|
||||
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
tsl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
size_t TotalCacheSize;
|
||||
|
||||
@@ -256,7 +260,7 @@ private:
|
||||
|
||||
size_t AllocateOffset {};
|
||||
|
||||
FEXCore::Context::ContextImpl *ctx;
|
||||
FEXCore::Context::Context *ctx;
|
||||
uint64_t VirtualMemSize{};
|
||||
};
|
||||
}
|
||||
+12
-21
@@ -1,16 +1,12 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// If any of the config options mismatch on load then the cache won't be used
|
||||
// Any of these will result in codegen changes
|
||||
struct
|
||||
FEX_PACKED
|
||||
CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
struct CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
uint64_t Cookie{};
|
||||
|
||||
// Instructions per block configuration
|
||||
@@ -20,45 +16,41 @@ namespace FEXCore::CodeSerialize {
|
||||
unsigned Arch : 4;
|
||||
|
||||
// Multiblock enabled
|
||||
unsigned MultiBlock : 1;
|
||||
|
||||
// Hardware TSO enabled
|
||||
unsigned HardwareTSOEnabled : 1;
|
||||
bool MultiBlock : 1;
|
||||
|
||||
// TSO enabled
|
||||
unsigned TSOEnabled : 1;
|
||||
bool TSOEnabled : 1;
|
||||
|
||||
// ABI local flag unsafe optimization
|
||||
unsigned ABILocalFlags : 1;
|
||||
bool ABILocalFlags : 1;
|
||||
|
||||
// ABI no PF unsafe optimization
|
||||
unsigned ABINoPF : 1;
|
||||
bool ABINoPF : 1;
|
||||
|
||||
// Static register allocation enabled
|
||||
unsigned SRA : 1;
|
||||
bool SRA : 1;
|
||||
|
||||
// Paranoid TSO mode enabled
|
||||
unsigned ParanoidTSO : 1;
|
||||
bool ParanoidTSO : 1;
|
||||
|
||||
// Guest code execution mode (We don't support live mode switch)
|
||||
unsigned Is64BitMode : 1;
|
||||
bool Is64BitMode : 1;
|
||||
|
||||
// SMC checks style
|
||||
unsigned SMCChecks : 2;
|
||||
|
||||
// x87 reduced precision
|
||||
unsigned x87ReducedPrecision : 1;
|
||||
bool x87ReducedPrecision : 1;
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 17;
|
||||
unsigned _Pad : 18;
|
||||
|
||||
bool operator==(CodeObjectSerializationConfig const &other) const {
|
||||
return Cookie == other.Cookie &&
|
||||
MaxInstPerBlock == other.MaxInstPerBlock &&
|
||||
Arch == other.Arch &&
|
||||
MultiBlock == other.MultiBlock &&
|
||||
HardwareTSOEnabled == other.HardwareTSOEnabled &&
|
||||
TSOEnabled == other.TSOEnabled &&
|
||||
ABILocalFlags == other.ABILocalFlags &&
|
||||
ABINoPF == other.ABINoPF &&
|
||||
@@ -75,7 +67,6 @@ namespace FEXCore::CodeSerialize {
|
||||
Hash <<= 32; Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1; Hash |= other.Arch;
|
||||
Hash <<= 1; Hash |= other.MultiBlock;
|
||||
Hash <<= 1; Hash |= other.HardwareTSOEnabled;
|
||||
Hash <<= 1; Hash |= other.TSOEnabled;
|
||||
Hash <<= 1; Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1; Hash |= other.ABINoPF;
|
||||
@@ -88,6 +79,6 @@ namespace FEXCore::CodeSerialize {
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is generated!");
|
||||
}
|
||||
@@ -2,24 +2,25 @@
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <filesystem>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <sys/uio.h>
|
||||
#include <sys/mman.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename) {
|
||||
#ifndef _WIN32
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
// This function adds a named region *JOB* to our named region handler
|
||||
// This needs to be as fast as possible to keep out of the way of the JIT
|
||||
|
||||
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
|
||||
auto BaseFilename = std::filesystem::path(filename).filename().string();
|
||||
|
||||
if (!BaseFilename.empty()) {
|
||||
// Create a new entry that once set up will be put in to our section object map
|
||||
auto Entry = fextl::make_unique<CodeRegionEntry>(
|
||||
auto Entry = std::make_unique<CodeRegionEntry>(
|
||||
Base,
|
||||
Size,
|
||||
Offset,
|
||||
@@ -76,14 +77,12 @@ namespace FEXCore::CodeSerialize {
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
#ifndef _WIN32
|
||||
// Removing a named region through the job system
|
||||
// We need to find the entry that we are deleting first
|
||||
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
std::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
@@ -120,10 +119,9 @@ namespace FEXCore::CodeSerialize {
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(std::unique_ptr<SerializationJobData> Data) {
|
||||
// XXX: Actually add serialization job
|
||||
}
|
||||
}
|
||||
Loaded 100 of 673 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user