Compare commits

..
1 Commits
Author SHA1 Message Date
Ryan Houdek 33fe6813fc Docs: Update for release FEX-2104 2021-04-02 11:35:29 -07:00
569 changed files with 19427 additions and 61891 deletions

No files matched your search

+1 -14
View File
@@ -13,14 +13,13 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
jobs:
build:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2], [self-hosted, ARMv8.4]]
arch: [[self-hosted, x64], [self-hosted, ARMv8.0], [self-hosted, ARMv8.2]]
fail-fast: false
steps:
@@ -117,18 +116,6 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC64.log || true
- name: gcc target tests 32
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the gvisor tests
run: cmake --build . --config $BUILD_TYPE --target gcc_target_tests_32
- name: GCC32 Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_GCC32.log || true
- name: Struct verifier tests
working-directory: ${{runner.workspace}}/build
shell: bash
-16
View File
@@ -30,19 +30,3 @@
shallow = true
path = External/fex-gcc-target-tests-bins
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
[submodule "External/jemalloc"]
path = External/jemalloc
url = https://github.com/FEX-Emu/jemalloc.git
[submodule "External/fmt"]
path = External/fmt
url = https://github.com/fmtlib/fmt.git
[submodule "External/drm-headers"]
path = External/drm-headers
url = https://github.com/FEX-Emu/drm-headers.git
[submodule "External/xxhash"]
path = External/xxhash
url = https://github.com/FEX-Emu/xxHash.git
[submodule "External/Vulkan-Docs"]
shallow = true
path = External/Vulkan-Docs
url = https://github.com/KhronosGroup/Vulkan-Docs.git
+50 -308
View File
@@ -14,10 +14,6 @@ option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
option(ENABLE_WERROR "Enables -Werror" FALSE)
option(ENABLE_STATIC_PIE "Enables static-pie build" FALSE)
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
@@ -48,6 +44,38 @@ else()
set(CMAKE_INTERPROCEDURAL_OPTIMIZATION FALSE)
endif()
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
message(STATUS "CCache enabled")
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
endif()
if (ENABLE_XRAY)
add_compile_options(-fxray-instrument)
link_libraries(-fxray-instrument)
endif()
if (ENABLE_LLD)
link_libraries(-fuse-ld=lld)
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address)
link_libraries(-fno-omit-frame-pointer -fsanitize=address)
endif()
if (ENABLE_TSAN)
add_compile_options(-fno-omit-frame-pointer -fsanitize=thread)
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
endif()
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
@@ -67,186 +95,6 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
add_definitions(-D_M_ARM_64=1)
endif()
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
message(STATUS "CCache enabled")
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
endif()
if (ENABLE_XRAY)
add_compile_options(-fxray-instrument)
link_libraries(-fxray-instrument)
endif()
if (ENABLE_COMPILE_TIME_TRACE)
add_compile_options(-ftime-trace)
link_libraries(-ftime-trace)
endif()
set (PTHREAD_LIB pthread)
if (ENABLE_LLD)
set (LD_OVERRIDE "-fuse-ld=lld")
link_libraries(${LD_OVERRIDE})
endif()
if (NOT ENABLE_OFFLINE_TELEMETRY)
# Disable FEX offline telemetry entirely if asked
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
endif()
if (ENABLE_STATIC_PIE)
if (_M_ARM_64 AND ENABLE_LLD)
message (FATAL_ERROR "Static linking does not currently work with AArch64+LLD. Use GNU ld for now.")
endif()
file(WRITE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt.c
"int main(int argc, char* argv[])
{
return 0;
}")
# Compile the test application with our LD_OVERRIDE and static-pie options
try_compile(
COMPILE_RESULT
${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp
${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt.c
COMPILE_DEFINITIONS "-fPIE ${LD_OVERRIDE}"
LINK_LIBRARIES "-static-pie ${LD_OVERRIDE}"
COPY_FILE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt
)
if (${COMPILE_RESULT})
# Read the symbols from the elf
execute_process(COMMAND
readelf -s ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt
OUTPUT_FILE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/plt_out.txt
OUTPUT_VARIABLE PLT_SYMBOLS)
# Pull out the __rela_iplt_{start,end} symbols if they exist
execute_process(COMMAND
"grep" "__rela_iplt" ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/plt_out.txt
OUTPUT_VARIABLE PLT_SYMBOLS)
set (SYMBOLS_FINE TRUE)
set (HAS_IPLT -1)
# Check if we have any symbols in our grep output
# The symbols must either not exist at all OR the symbols are zero
if (PLT_SYMBOLS)
string(FIND ${PLT_SYMBOLS} "__rela_iplt_start" HAS_IPLT)
endif()
if (NOT HAS_IPLT EQUAL -1)
# We have some symbols from readelf. Let's parse the results to check if they are zero
# Format: '35: 0000000000000000 0 NOTYPE LOCAL HIDDEN UND __rela_iplt_start'
string(REPLACE "\n" ";" SYMBOL_LIST ${PLT_SYMBOLS})
foreach (SYMBOL ${SYMBOL_LIST})
# strip any leading and trailing whitespace
string (STRIP ${SYMBOL} SYMBOL)
# Convert string to a list
string(REPLACE " " ";" SYMBOL_VALUES ${SYMBOL}})
# Pull out the address argument
list(GET SYMBOL_VALUES 1 OFFSET)
# Check against integer zero
if (NOT ${OFFSET} EQUAL 0)
# Symbol wasn't zero, this now fails
set (SYMBOLS_FINE FALSE)
endif()
endforeach()
endif()
if (SYMBOLS_FINE)
# We can now exnable static-pie
set (STATIC_PIE_OPTIONS "-static-pie")
# Pthreads has an issue with exposing symbols
# We need to make some concessions to the pthread gods
if (ENABLE_LLD)
set (PTHREAD_LIB
-Wl,--undefined-glob=pthread_*
-Wl,--undefined=__cxa_finalize
-Wl,--undefined=_pthread_cleanup_push_defer
-Wl,--undefined=_pthread_cleanup_pop_restore
-Wl,--undefined=__pthread_cleanup_upto
pthread)
else()
set (PTHREAD_LIB
-Wl,--undefined=pthread_join
-Wl,--undefined=pthread_attr_getdetachstate
-Wl,--undefined=pthread_sigmask
-Wl,--undefined=pthread_mutex_lock
-Wl,--undefined=pthread_cond_init
-Wl,--undefined=pthread_attr_init
-Wl,--undefined=pthread_mutex_unlock
-Wl,--undefined=pthread_mutexattr_destroy
-Wl,--undefined=pthread_detach
-Wl,--undefined=pthread_mutex_init
-Wl,--undefined=pthread_getattr_np
-Wl,--undefined=pthread_cond_timedwait
-Wl,--undefined=pthread_attr_destroy
-Wl,--undefined=pthread_mutexattr_settype
-Wl,--undefined=pthread_rwlock_unlock
-Wl,--undefined=pthread_rwlock_wrlock
-Wl,--undefined=pthread_setspecific
-Wl,--undefined=pthread_create
-Wl,--undefined=pthread_cond_clockwait
-Wl,--undefined=pthread_key_create
-Wl,--undefined=pthread_rwlock_rdlock
-Wl,--undefined=pthread_setname_np
-Wl,--undefined=pthread_cond_signal
-Wl,--undefined=pthread_mutexattr_init
-Wl,--undefined=pthread_attr_setstack
-Wl,--undefined=pthread_self
-Wl,--undefined=pthread_getaffinity_np
-Wl,--undefined=pthread_cond_wait
-Wl,--undefined=pthread_mutex_trylock
-Wl,--undefined=pthread_cond_broadcast
-Wl,--undefined=pthread_cond_destroy
-Wl,--undefined=pthread_getspecific
-Wl,--undefined=pthread_key_delete
-Wl,--undefined=pthread_once
-Wl,--undefined=__cxa_finalize
-Wl,--undefined=_pthread_cleanup_push_defer
-Wl,--undefined=_pthread_cleanup_pop_restore
-Wl,--undefined=__pthread_cleanup_upto
pthread)
endif()
else()
message (FATAL_ERROR "Application has __rela_iplt_{start,end} symbols. Which means static-pie can't be enabled")
endif()
else()
message (FATAL_ERROR "Couldn't compile static-pie test. Static-pie can't be enabled! Is your glibc compiled without static-pie?")
endif()
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
link_libraries(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
endif()
if (ENABLE_TSAN)
add_compile_options(-fno-omit-frame-pointer -fsanitize=thread)
link_libraries(-fno-omit-frame-pointer -fsanitize=thread)
endif()
if (ENABLE_JEMALLOC)
add_definitions(-DENABLE_JEMALLOC=1)
add_subdirectory(External/jemalloc/)
include_directories(External/jemalloc/pregen/include/)
else()
message (STATUS
" jemalloc disabled!\n"
" This is not a recommended configuration!\n"
" This will very explicitly break 32-bit application execution!\n"
" Use at your own risk!")
endif()
set (CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
add_subdirectory(External/vixl/)
include_directories(External/vixl/src/)
@@ -255,15 +103,7 @@ if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
pkg_check_modules(XXHASH libxxhash>=0.8.0 QUIET)
if (NOT XXHASH_FOUND)
message(STATUS "xxHash not found. Using Externals")
add_subdirectory(External/xxhash/)
include_directories(External/xxhash/)
endif()
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
@@ -271,8 +111,6 @@ add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
add_subdirectory(External/cpp-optparse/)
include_directories(External/cpp-optparse/)
add_subdirectory(External/fmt/)
add_subdirectory(External/imgui/)
include_directories(External/imgui/)
@@ -306,6 +144,11 @@ if(ENUM_ENUM_WARNING)
add_compile_options(-Wno-deprecated-enum-enum-conversion)
endif()
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
if(COMPILER_SUPPORTS_MARCH_NATIVE)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -march=native")
endif()
if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
add_compile_options(-Werror)
if (NOT ENABLE_STRICT_WERROR)
@@ -315,32 +158,18 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
endif()
if(_M_ARM_64)
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=native")
endif()
else()
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
# Manually detect newer CPU revisions until clang and llvm fixes their bug
# This script will either provide a supported CPU or 'native'
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
OUTPUT_VARIABLE AARCH64_CPU)
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
# Manually detect newer CPU revisions until clang and llvm fixes their bug
# This script will either provide a supported CPU or 'native'
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo"
OUTPUT_VARIABLE AARCH64_CPU)
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=${AARCH64_CPU}")
endif()
endif()
else()
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
if(COMPILER_SUPPORTS_MARCH_NATIVE)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -march=native")
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=${AARCH64_CPU}")
endif()
endif()
@@ -407,7 +236,7 @@ add_compile_options(-Wall)
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/include/Config.h.in
${CMAKE_BINARY_DIR}/generated/ConfigDefines.h)
${CMAKE_BINARY_DIR}/generated/Config.h)
if (BUILD_TESTS)
include(CTest)
@@ -416,17 +245,9 @@ if (BUILD_TESTS)
endif()
add_subdirectory(External/FEXCore)
# Binfmt_misc files must be installed prior to Source/ installs
add_subdirectory(Data/binfmts/)
add_subdirectory(Source/)
add_subdirectory(Data/AppConfig/)
# Install the ThunksDB file
install(
FILES ${CMAKE_CURRENT_SOURCE_DIR}/Data/ThunksDB.json
DESTINATION ${DATA_DIRECTORY}/)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
@@ -438,10 +259,7 @@ if (BUILD_THUNKS)
PREFIX host-libs
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/HostLibs"
BINARY_DIR "Host"
CMAKE_ARGS
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DVULKAN_XML=${CMAKE_SOURCE_DIR}/External/Vulkan-Docs/xml/vk.xml"
CMAKE_ARGS "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
)
@@ -459,13 +277,7 @@ if (BUILD_THUNKS)
PREFIX guest-libs
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
BINARY_DIR "Guest"
CMAKE_ARGS
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DX86_C_COMPILER:STRING=${X86_C_COMPILER}"
"-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
"-DVULKAN_XML=${CMAKE_SOURCE_DIR}/External/Vulkan-Docs/xml/vk.xml"
CMAKE_ARGS "-DX86_C_COMPILER:STRING=${X86_C_COMPILER}" "-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}" "-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
)
@@ -479,73 +291,3 @@ if (BUILD_THUNKS)
DEPENDS guest-libs
)
endif()
set(FEX_VERSION_MAJOR "0")
set(FEX_VERSION_MINOR "0")
set(FEX_VERSION_PATCH "0")
find_package(Git)
if (GIT_FOUND)
execute_process(
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
RESULT_VARIABLE GIT_ERROR
ERROR_QUIET
OUTPUT_STRIP_TRAILING_WHITESPACE
)
if (NOT ${GIT_ERROR} EQUAL 0)
# Likely built in a way that doesn't have tags
# Setup a version tag that is unknown
set(GIT_DESCRIBE_STRING "FEX-0000")
endif()
# Change something like `FEX-2106.1-76-<hash>` in to a list
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
# Extract the `2106.1` element
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
# Change `2106.1` in to a list
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
# Calculate list size
list(LENGTH DESCRIBE_LIST LIST_SIZE)
# Pull out the major version
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
# Minor version only exists if there is a .1 at the end
# eg: 2106 versus 2106.1
if (LIST_SIZE GREATER 1)
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
endif()
endif()
# Package creation
set (CPACK_GENERATOR "DEB")
if (ENABLE_STATIC_PIE)
set (CPACK_PACKAGE_NAME fex-emu-static)
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu")
else()
set (CPACK_PACKAGE_NAME fex-emu)
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu-static")
endif()
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.org>")
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/CPack/Description.txt")
# Debian defines
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
"${CMAKE_CURRENT_SOURCE_DIR}/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/CPack/triggers")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
# binfmt_misc conflicts with qemu-user-static
# We also only install binfmt_misc on aarch64 hosts
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
endif()
include (CPack)
-3
View File
@@ -1,3 +0,0 @@
x86 and x86-64 Linux emulator
FEX is very much work in progress, so expect things to change.
-18
View File
@@ -1,18 +0,0 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Setup binfmt_misc
update-binfmts --import FEX-x86
update-binfmts --import FEX-x86_64
}
# Install FEXInterpreter hardlink
# Needs to be done before setting up binfmt_misc
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
-17
View File
@@ -1,17 +0,0 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Uninstall
update-binfmts --unimport FEX-x86
update-binfmts --unimport FEX-x86_64
}
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
# Remove FEXInterpreter hardlink
unlink /usr/bin/FEXInterpreter
-1
View File
@@ -1 +0,0 @@
activate-noawait ldconfig
-349
View File
@@ -1,349 +0,0 @@
{
"DB": {
"GL": {
"Library" : "libGL-guest.so",
"Depends": [
"X11"
],
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libGL.so",
"/usr/lib/x86_64-linux-gnu/libGL.so.1",
"/usr/lib/x86_64-linux-gnu/libGL.so.1.2.0",
"/usr/lib/x86_64-linux-gnu/libGL.so.1.7.0",
"/usr/local/lib/x86_64-linux-gnu/libGL.so",
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1",
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.2.0",
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.7.0",
"/lib/x86_64-linux-gnu/libGL.so",
"/lib/x86_64-linux-gnu/libGL.so.1",
"/lib/x86_64-linux-gnu/libGL.so.1.2.0",
"/lib/x86_64-linux-gnu/libGL.so.1.7.0"
]
},
"GLESv2": {
"Library": "libGLESv2-guest.so",
"Depends": [
"X11"
],
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libGLESv2.so",
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2",
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so",
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2",
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
"/lib/x86_64-linux-gnu/libGLESv2.so",
"/lib/x86_64-linux-gnu/libGLESv2.so.2",
"/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0"
]
},
"X11": {
"Library": "libX11-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libX11.so",
"/usr/lib/x86_64-linux-gnu/libX11.so.6",
"/usr/lib/x86_64-linux-gnu/libX11.so.6.4.0",
"/usr/local/lib/x86_64-linux-gnu/libX11.so",
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6",
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6.4.0",
"/lib/x86_64-linux-gnu/libX11.so",
"/lib/x86_64-linux-gnu/libX11.so.6",
"/lib/x86_64-linux-gnu/libX11.so.6.4.0"
]
},
"Vulkan-radeon": {
"Library": "libvulkan_radeon-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libvulkan_radeon.so",
"/usr/local/lib/x86_64-linux-gnu/libvulkan_radeon.so",
"/lib/x86_64-linux-gnu/libvulkan_radeon.so"
],
"Comment": [
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
]
},
"Vulkan-lavapipe": {
"Library": "libvulkan_lvp-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libvulkan_lvp.so",
"/usr/local/lib/x86_64-linux-gnu/libvulkan_lvp.so",
"/lib/x86_64-linux-gnu/libvulkan_lvp.so"
]
},
"Vulkan-freedreno": {
"Library": "libvulkan_freedreno-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
"/usr/local/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
"/lib/x86_64-linux-gnu/libvulkan_freedreno.so"
]
},
"Vulkan-intel": {
"Library": "libvulkan_intel-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libvulkan_intel.so",
"/usr/local/lib/x86_64-linux-gnu/libvulkan_intel.so",
"/lib/x86_64-linux-gnu/libvulkan_intel.so"
]
},
"Vulkan-panfrost": {
"Library": "libvulkan_panfrost-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
"/usr/local/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
"/lib/x86_64-linux-gnu/libvulkan_panfrost.so"
]
},
"Vulkan-nvidia": {
"Library": "libvulkan_nvidia-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
"/usr/local/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
"/lib/x86_64-linux-gnu/libGLX_nvidia.so.0"
],
"Comment": [
"Not currently wired up"
]
},
"Vulkan-virtio": {
"Library": "libvulkan_virtio-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libvulkan_virtio.so",
"/usr/local/lib/x86_64-linux-gnu/libvulkan_virtio.so",
"/lib/x86_64-linux-gnu/libvulkan_virtio.so"
]
},
"xcb": {
"Library": "libxcb-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxcb.so",
"/usr/lib/x86_64-linux-gnu/libxcb.so.1",
"/usr/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb.so",
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1",
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
"/lib/x86_64-linux-gnu/libxcb.so",
"/lib/x86_64-linux-gnu/libxcb.so.1",
"/lib/x86_64-linux-gnu/libxcb.so.1.1.0"
]
},
"xcb-dri2": {
"Library": "libxcb_dri2-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so",
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so",
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
"/lib/x86_64-linux-gnu/libxcb-dri2.so",
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
]
},
"xcb-dri3": {
"Library": "libxcb_dri3-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so",
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so",
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
"/lib/x86_64-linux-gnu/libxcb-dri3.so",
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
]
},
"xcb-xfixes": {
"Library": "libxcb_xfixes-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so",
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so",
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
"/lib/x86_64-linux-gnu/libxcb-xfixes.so",
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
]
},
"xcb-shm": {
"Library": "libxcb_shm-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so",
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0",
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so",
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
"/lib/x86_64-linux-gnu/libxcb-shm.so",
"/lib/x86_64-linux-gnu/libxcb-shm.so.0",
"/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
]
},
"xcb-sync": {
"Library": "libxcb_sync-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so",
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1",
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so",
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1",
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
"/lib/x86_64-linux-gnu/libxcb-sync.so",
"/lib/x86_64-linux-gnu/libxcb-sync.so.1",
"/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
]
},
"xcb-randr": {
"Library": "libxcb_randr-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so",
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0",
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so",
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
"/lib/x86_64-linux-gnu/libxcb-randr.so",
"/lib/x86_64-linux-gnu/libxcb-randr.so.0",
"/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
]
},
"xcb-present": {
"Library": "libxcb_present-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxcb-present.so",
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0",
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so",
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
"/lib/x86_64-linux-gnu/libxcb-present.so",
"/lib/x86_64-linux-gnu/libxcb-present.so.0",
"/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0"
]
},
"xcb-glx": {
"Library": "libxcb_glx-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so",
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0",
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so",
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0",
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
"/lib/x86_64-linux-gnu/libxcb-glx.so",
"/lib/x86_64-linux-gnu/libxcb-glx.so.0",
"/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
]
},
"xshmfence": {
"Library": "libshmfence-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libxshmfence.so",
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1",
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so",
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1",
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
"/lib/x86_64-linux-gnu/libxshmfence.so",
"/lib/x86_64-linux-gnu/libxshmfence.so.1",
"/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0"
]
},
"drm": {
"Library": "libdrm-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libdrm.so",
"/usr/lib/x86_64-linux-gnu/libdrm.so.2",
"/usr/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
"/usr/local/lib/x86_64-linux-gnu/libdrm.so",
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2",
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
"/lib/x86_64-linux-gnu/libdrm.so",
"/lib/x86_64-linux-gnu/libdrm.so.2",
"/lib/x86_64-linux-gnu/libdrm.so.2.4.0"
]
},
"asound": {
"Library": "libasound-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libasound.so",
"/usr/lib/x86_64-linux-gnu/libasound.so.2",
"/usr/lib/x86_64-linux-gnu/libasound.so.2.0.0",
"/usr/local/lib/x86_64-linux-gnu/libasound.so",
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2",
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2.0.0",
"/lib/x86_64-linux-gnu/libasound.so",
"/lib/x86_64-linux-gnu/libasound.so.2",
"/lib/x86_64-linux-gnu/libasound.so.2.0.0"
]
},
"Xrender": {
"Library": "libXrender-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libXrender.so",
"/usr/lib/x86_64-linux-gnu/libXrender.so.1",
"/usr/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
"/usr/local/lib/x86_64-linux-gnu/libXrender.so",
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1",
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
"/lib/x86_64-linux-gnu/libXrender.so",
"/lib/x86_64-linux-gnu/libXrender.so.1",
"/lib/x86_64-linux-gnu/libXrender.so.1.3.0"
]
},
"Xext": {
"Library": "libXext-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libXext.so",
"/usr/lib/x86_64-linux-gnu/libXext.so.6",
"/usr/lib/x86_64-linux-gnu/libXext.so.6.4.0",
"/usr/local/lib/x86_64-linux-gnu/libXext.so",
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6",
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6.4.0",
"/lib/x86_64-linux-gnu/libXext.so",
"/lib/x86_64-linux-gnu/libXext.so.6",
"/lib/x86_64-linux-gnu/libXext.so.6.4.0"
]
},
"Xfixes": {
"Library": "libXfixes-guest.so",
"Overlay": [
"/usr/lib/x86_64-linux-gnu/libXfixes.so",
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3",
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so",
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3",
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
"/lib/x86_64-linux-gnu/libXfixes.so",
"/lib/x86_64-linux-gnu/libXfixes.so.3",
"/lib/x86_64-linux-gnu/libXfixes.so.3.1.0"
]
},
"":{}
}
}
-17
View File
@@ -1,17 +0,0 @@
function(GenBinFmt Name)
# Get the filename only component
get_filename_component(FMT_NAME ${Name} NAME_WE)
# Configure it
configure_file(
${Name}
${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME})
# Then install the configured binfmt
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
endfunction()
GenBinFmt(FEX-x86.in)
GenBinFmt(FEX-x86_64.in)
-8
View File
@@ -1,8 +0,0 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
credentials yes
fix_binary yes
preserve yes
-8
View File
@@ -1,8 +0,0 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
credentials yes
fix_binary yes
preserve yes
+2 -2
View File
@@ -3,7 +3,7 @@ FROM ubuntu:20.04 as builder
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
clang-10 llvm-10 nasm ninja-build \
clang-10 llvm-10 nasm ninja-build libnuma-dev \
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
python3 linux-headers-generic
@@ -23,7 +23,7 @@ FROM ubuntu:20.04
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y \
libcap-dev libglfw3-dev libepoxy-dev
libnuma-dev libcap-dev libglfw3-dev libepoxy-dev
COPY --from=builder /opt/FEX/build/Bin/* /usr/bin/
+1
View File
@@ -16,6 +16,7 @@ endif()
set(ENABLE_JIT_X86_64 ${_M_X86_64} CACHE BOOL "Enable the x86_64 JIT")
set(ENABLE_JIT_ARM64 ${_M_ARM_64} CACHE BOOL "Enable the ARM64 JIT")
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
option(ENABLE_JITSYMBOLS "Enable visibility of JITSymbols in profiling tools" FALSE)
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
+5 -65
View File
@@ -98,19 +98,14 @@ def print_man_option(short, long, desc, default):
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
def print_man_env_option(name, desc, default, no_json_key):
output_man.write("\\fBFEX_{0}\\fR\n".format(name.upper()))
def print_man_env_option(name, desc, default):
output_man.write("\\fBFEX_{0}\\fR\n".format(name))
# Print description
for line in desc:
output_man.write(".Pp\n")
output_man.write("{0}\n".format(line))
if (not no_json_key):
output_man.write(".Pp\n")
output_man.write("\\fBJSON key:\\fR '{0}'\n".format(name))
output_man.write(".Pp\n\n")
output_man.write(".Pp\n")
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
@@ -159,48 +154,12 @@ def print_man_environment(options):
# Wrap the string argument in quotes
default = "'" + default + "'"
print_man_env_option(
op_key,
op_key.upper(),
op_vals["Desc"],
default,
False
default
)
print_man_environment_tail()
output_man.write(".El\n")
def print_man_environment_tail():
# Additional environment variables that live outside of the normal loop
print_man_env_option(
"FEX_APP_CONFIG_LOCATION",
[
"Allows the user to override where FEX looks for configuration files",
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
"This will override the full path",
],
"''", True)
print_man_env_option(
"FEX_APP_CONFIG",
[
"Allows the user to override where FEX looks for only the application config file",
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/Config.json",
"This will override this file location",
"One must be careful with this option as it will override any applications that load with execve as well"
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
],
"''", True)
print_man_env_option(
"FEX_APP_DATA_LOCATION",
[
"Allows the user to override where FEX looks for data files",
"By default FEX will look in {$HOME, $XDG_DATA_HOME}/.fex-emu/",
"This will override the full path",
"This is the folder where FEX stores generated files like IR cache"
],
"''", True)
def print_man_header():
header ='''.Dd {0}
.Dt FEX
@@ -374,7 +333,7 @@ def print_parse_argloader_options(options):
conversion_func = "std::to_string"
if ("ArgumentHandler" in op_vals):
NeedsString = True
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
conversion_func = "FEX::Handler::{0}".format(op_vals["ArgumentHandler"])
if (value_type == "str"):
NeedsString = True
conversion_func = ""
@@ -396,21 +355,6 @@ def print_parse_argloader_options(options):
output_argloader.write("#endif\n")
def print_parse_envloader_options(options):
output_argloader.write("#ifdef ENVLOADER\n")
output_argloader.write("#undef ENVLOADER\n")
output_argloader.write("if (false) {}\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
if ("ArgumentHandler" in op_vals):
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
output_argloader.write("Value = {0}(Value);\n".format(conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def check_for_duplicate_options(options):
short_map = []
long_map = []
@@ -485,8 +429,4 @@ output_man.close()
output_argloader = open(output_argumentloader_filename, "w")
print_argloader_options(options);
print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
output_argloader.close()
+28 -48
View File
@@ -7,7 +7,7 @@ def print_enums(ops, defines):
output_file.write("enum IROps : uint8_t {\n")
for op_key, op_vals in ops.items():
output_file.write("\tOP_%s,\n" % op_key.upper())
output_file.write("\t\tOP_%s,\n" % op_key.upper())
output_file.write("};\n")
@@ -20,10 +20,7 @@ def print_ir_structs(ops, defines):
# Print out defines here
for op_val in defines:
if op_val:
output_file.write("\t%s;\n" % op_val)
else:
output_file.write("\n")
output_file.write("\t%s;\n" % op_val)
output_file.write("// Default structs\n")
output_file.write("struct __attribute__((packed)) IROp_Header {\n")
@@ -84,21 +81,11 @@ def print_ir_structs(ops, defines):
output_file.write("\tstatic constexpr IROps OPCODE = OP_%s;\n" % op_key.upper())
if (SSAArgs > 0):
# Add helpers for accessing SSA arguments, given how frequently they're accessed
output_file.write("\n")
output_file.write("\t[[nodiscard]] OrderedNodeWrapper& Args(size_t Index) {\n")
output_file.write("\t\treturn Header.Args[Index];\n")
output_file.write("\t}\n")
output_file.write("\t[[nodiscard]] const OrderedNodeWrapper& Args(size_t Index) const {\n")
output_file.write("\t\treturn Header.Args[Index];\n")
output_file.write("\t}\n")
output_file.write("};\n")
# Add a static assert that the IR ops must be pod
output_file.write("static_assert(std::is_trivial_v<IROp_%s>);\n" % op_key)
output_file.write("static_assert(std::is_standard_layout_v<IROp_%s>);\n\n" % op_key)
output_file.write("static_assert(std::is_trivial<IROp_%s>::value);\n\n" % op_key)
output_file.write("static_assert(std::is_standard_layout<IROp_%s>::value);\n\n" % op_key)
output_file.write("#undef IROP_STRUCTS\n")
output_file.write("#endif\n\n")
@@ -119,12 +106,12 @@ def print_ir_sizes(ops, defines):
output_file.write("// Make sure our array maps directly to the IROps enum\n")
output_file.write("static_assert(IRSizes[IROps::OP_LAST] == -1ULL);\n\n")
output_file.write("[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }\n\n")
output_file.write("[[maybe_unused]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }\n\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] std::string_view const& GetName(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetArgs(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
output_file.write("std::string_view const& GetName(IROps Op);\n")
output_file.write("uint8_t GetArgs(IROps Op);\n")
output_file.write("FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
output_file.write("bool HasSideEffects(IROps Op);\n")
output_file.write("#undef IROP_SIZES\n")
output_file.write("#endif\n\n")
@@ -283,15 +270,14 @@ def print_ir_allocator_helpers(ops, defines):
output_file.write("\t\t\n")
output_file.write("\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n")
output_file.write("\t\toperator OrderedNode *() { return Node; }\n")
output_file.write("\t\toperator const OrderedNode *() const { return Node; }\n")
output_file.write("\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n")
output_file.write("\t\toperator OpNodeWrapper () { return Node->Header.Value; }\n")
output_file.write("\t};\n")
output_file.write("\ttemplate <class T>\n")
output_file.write("\tusing IRPair = Wrapper<T>;\n\n")
output_file.write("\tIRPair<IROp_Header> AllocateRawOp(size_t HeaderSize) {\n")
output_file.write("\t\tauto Op = reinterpret_cast<IROp_Header*>(DualListData.DataAllocate(HeaderSize));\n")
output_file.write("\t\tauto Op = reinterpret_cast<IROp_Header*>(Data.Allocate(HeaderSize));\n")
output_file.write("\t\tmemset(Op, 0, HeaderSize);\n")
output_file.write("\t\tOp->Op = IROps::OP_DUMMY;\n")
output_file.write("\t\treturn IRPair<IROp_Header>{Op, CreateNode(Op)};\n")
@@ -300,7 +286,7 @@ def print_ir_allocator_helpers(ops, defines):
output_file.write("\ttemplate<class T, IROps T2>\n")
output_file.write("\tT *AllocateOrphanOp() {\n")
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
output_file.write("\t\tauto Op = reinterpret_cast<T*>(Data.Allocate(Size));\n")
output_file.write("\t\tmemset(Op, 0, Size);\n")
output_file.write("\t\tOp->Header.Op = T2;\n")
output_file.write("\t\treturn Op;\n")
@@ -309,25 +295,25 @@ def print_ir_allocator_helpers(ops, defines):
output_file.write("\ttemplate<class T, IROps T2>\n")
output_file.write("\tIRPair<T> AllocateOp() {\n")
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
output_file.write("\t\tauto Op = reinterpret_cast<T*>(Data.Allocate(Size));\n")
output_file.write("\t\tmemset(Op, 0, Size);\n")
output_file.write("\t\tOp->Header.Op = T2;\n")
output_file.write("\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n")
output_file.write("\t}\n\n")
output_file.write("\tuint8_t GetOpSize(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\tuint8_t GetOpSize(OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(Data.Begin());\n")
output_file.write("\t\treturn HeaderOp->Size;\n")
output_file.write("\t}\n\n")
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\t\tLOGMAN_THROW_A_FMT(HeaderOp->HasDest, \"Op {} has no dest\\n\", GetName(HeaderOp->Op));\n")
output_file.write("\tuint8_t GetOpElements(OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(Data.Begin());\n")
output_file.write("\t\tLogMan::Throw::A(HeaderOp->HasDest, \"Op %s has no dest\\n\", GetName(HeaderOp->Op));\n")
output_file.write("\t\treturn HeaderOp->Size / HeaderOp->ElementSize;\n")
output_file.write("\t}\n\n")
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\tbool OpHasDest(OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(Data.Begin());\n")
output_file.write("\t\treturn HeaderOp->HasDest;\n")
output_file.write("\t}\n\n")
@@ -401,14 +387,11 @@ def print_ir_allocator_helpers(ops, defines):
output_file.write(") {\n")
output_file.write("\t\tauto Op = AllocateOp<IROp_%s, IROps::OP_%s>();\n" % (op_key, op_key.upper()))
if (SSAArgs != 0):
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
for i in range(0, SSAArgs):
output_file.write("\t\tOp.first->Header.Args[%d] = ssa%d->Wrapped(ListDataBegin);\n" % (i, i))
output_file.write("\t\tOp.first->Header.NumArgs = %d;\n" % (SSAArgs))
if (SSAArgs != 0):
for i in range(0, SSAArgs):
output_file.write("\t\tOp.first->Header.Args[%d] = ssa%d->Wrapped(ListData.Begin());\n" % (i, i))
output_file.write("\t\tssa%d->AddUse();\n" % (i))
if (HasArgs):
@@ -416,6 +399,11 @@ def print_ir_allocator_helpers(ops, defines):
data_name = op_vals["Args"][i]
output_file.write("\t\tOp.first->%s = %s;\n" % (data_name, data_name))
if (HasFixedDestSize):
output_file.write("\t\tOp.first->Header.Size = %d;\n" % FixedDestSize)
if (HasDestSize):
output_file.write("\t\tOp.first->Header.Size = %s;\n" % DestSize)
if (HasDest):
# We can only infer a size if we have arguments
if not (HasFixedDestSize or HasDestSize):
@@ -424,18 +412,10 @@ def print_ir_allocator_helpers(ops, defines):
if (SSAArgs != 0):
for i in range(0, SSAArgs):
output_file.write("\t\tuint8_t Size%d = GetOpSize(ssa%s);\n" % (i, i))
for i in range(0, SSAArgs):
output_file.write("\t\tInferSize = std::max(InferSize, Size%d);\n" % (i))
output_file.write("\t\tOp.first->Header.Size = InferSize;\n")
output_file.write("\t\tOp.first->Header.NumArgs = %d;\n" % (SSAArgs))
if (HasFixedDestSize):
output_file.write("\t\tOp.first->Header.Size = %d;\n" % FixedDestSize)
if (HasDestSize):
output_file.write("\t\tOp.first->Header.Size = %s;\n" % DestSize)
output_file.write("\t\tOp.first->Header.ElementSize = Op.first->Header.Size / (%s);\n" % NumElements)
if (HasDest):
@@ -519,7 +499,7 @@ def print_ir_parser_allocator_helpers(ops, defines):
if (SSAArgs != 0):
for i in range(0, SSAArgs):
output_file.write("\t\tOp.first->Header.Args[%d] = ssa%d->Wrapped(DualListData.ListBegin());\n" % (i, i))
output_file.write("\t\tOp.first->Header.Args[%d] = ssa%d->Wrapped(ListData.Begin());\n" % (i, i))
output_file.write("\t\tssa%d->AddUse();\n" % (i))
if (HasArgs):
+12 -68
View File
@@ -3,6 +3,7 @@ set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
set (SRCS
Common/Paths.cpp
Common/JitSymbols.cpp
Common/NetStream.cpp
Common/SoftFloat-3e/extF80_add.c
Common/SoftFloat-3e/extF80_div.c
Common/SoftFloat-3e/extF80_sub.c
@@ -80,12 +81,7 @@ set (SRCS
Interface/Core/Frontend.cpp
Interface/Core/GdbServer.cpp
Interface/Core/HostFeatures.cpp
Interface/Core/OpcodeDispatcher/Crypto.cpp
Interface/Core/OpcodeDispatcher/Flags.cpp
Interface/Core/OpcodeDispatcher/Vector.cpp
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/SignalDelegator.cpp
Interface/Core/X86Tables.cpp
Interface/Core/X86DebugInfo.cpp
Interface/Core/X86HelperGen.cpp
@@ -96,17 +92,6 @@ set (SRCS
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
Interface/Core/Interpreter/InterpreterCore.cpp
Interface/Core/Interpreter/InterpreterOps.cpp
Interface/Core/Interpreter/ALUOps.cpp
Interface/Core/Interpreter/AtomicOps.cpp
Interface/Core/Interpreter/BranchOps.cpp
Interface/Core/Interpreter/ConversionOps.cpp
Interface/Core/Interpreter/EncryptionOps.cpp
Interface/Core/Interpreter/F80Ops.cpp
Interface/Core/Interpreter/FlagOps.cpp
Interface/Core/Interpreter/MemoryOps.cpp
Interface/Core/Interpreter/MiscOps.cpp
Interface/Core/Interpreter/MoveOps.cpp
Interface/Core/Interpreter/VectorOps.cpp
Interface/Core/X86Tables/BaseTables.cpp
Interface/Core/X86Tables/DDDTables.cpp
Interface/Core/X86Tables/EVEXTables.cpp
@@ -129,8 +114,6 @@ set (SRCS
Interface/IR/Passes/DeadContextStoreElimination.cpp
Interface/IR/Passes/IRCompaction.cpp
Interface/IR/Passes/IRValidation.cpp
Interface/IR/Passes/RAValidation.cpp
Interface/IR/Passes/LongDivideRemovalPass.cpp
Interface/IR/Passes/ValueDominanceValidation.cpp
Interface/IR/Passes/PhiValidation.cpp
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
@@ -138,13 +121,9 @@ set (SRCS
Interface/IR/Passes/StaticRegisterAllocationPass.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/SyscallOptimization.cpp
Utils/Allocator.cpp
Utils/Allocator/64BitAllocator.cpp
Utils/FileLoading.cpp
Utils/ForcedAssert.cpp
Utils/ELFLoader.cpp
Utils/ELFSymbolDatabase.cpp
Utils/LogManager.cpp
Utils/NetStream.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
)
@@ -153,7 +132,7 @@ if(_M_ARM_64)
Interface/Core/ArchHelpers/Arm64.cpp)
endif()
set(DEFINES -DTHREAD_LOCAL=_Thread_local)
set(DEFINES )
if (_M_X86_64)
list(APPEND DEFINES -D_M_X86_64=1)
@@ -195,16 +174,10 @@ if (ENABLE_JIT_ARM64)
Interface/Core/JIT/Arm64/VectorOps.cpp)
endif()
set (LIBS vixl dl fmt::fmt xxhash tiny-json)
if (ENABLE_JEMALLOC)
list (APPEND LIBS FEX_jemalloc)
if (ENABLE_JITSYMBOLS)
list(APPEND DEFINES -DENABLE_JITSYMBOLS=1)
endif()
# Generate config
configure_file(
${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
${CMAKE_BINARY_DIR}/generated/Config/Config.json)
# Generate IR include file
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
@@ -247,9 +220,8 @@ add_custom_target(IR_INC
set(OUTPUT_CONFIG_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/Config")
set(OUTPUT_CONFIG_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigValues.inl")
set(OUTPUT_CONFIG_OPTION_NAME "${OUTPUT_CONFIG_FOLDER}/ConfigOptions.inl")
set(INPUT_CONFIG_NAME "${CMAKE_BINARY_DIR}/generated/Config/Config.json")
set(INPUT_CONFIG_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json")
set(OUTPUT_MAN_NAME "${CMAKE_BINARY_DIR}/generated/FEX.1")
set(OUTPUT_MAN_NAME_COMPRESS "${CMAKE_BINARY_DIR}/generated/FEX.1.gz")
add_custom_target(CREATE_CONFIG_FOLDER ALL
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_CONFIG_FOLDER}")
@@ -265,12 +237,6 @@ add_custom_command(
"${OUTPUT_CONFIG_OPTION_NAME}"
)
add_custom_command(
OUTPUT "${OUTPUT_MAN_NAME_COMPRESS}"
DEPENDS "${OUTPUT_MAN_NAME}"
COMMAND "gzip" "-kf9n" "${OUTPUT_MAN_NAME}"
)
set_source_files_properties(${OUTPUT_CONFIG_NAME} PROPERTIES
GENERATED TRUE)
set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
@@ -278,18 +244,15 @@ set_source_files_properties(${OUTPUT_CONFIG_OPTION_NAME} PROPERTIES
set_source_files_properties(${OUTPUT_MAN_NAME} PROPERTIES
GENERATED TRUE)
set_source_files_properties(${OUTPUT_MAN_NAME_COMPRESS} PROPERTIES
GENERATED TRUE)
# Create the target
add_custom_target(CONFIG_INC
DEPENDS "${OUTPUT_CONFIG_NAME}"
DEPENDS "${OUTPUT_CONFIG_OPTION_NAME}"
DEPENDS "${OUTPUT_MAN_NAME}"
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
DEPENDS "${OUTPUT_MAN_NAME}")
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
# Install the man page
install(FILES ${OUTPUT_MAN_NAME} DESTINATION ${MAN_DIR}/man1)
# Add in diagnostic colours if the option is available.
# Ninja code generator will kill colours if this isn't here
@@ -301,11 +264,8 @@ function(AddObject Name Type)
add_dependencies(${Name} IR_INC)
add_dependencies(${Name} CONFIG_INC)
target_link_libraries(${Name} ${LIBS})
target_link_libraries(${Name} pthread rt vixl ${LINUX_LIBS} dl)
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
set_target_properties(${Name} PROPERTIES VISIBILITY_INLINES_HIDDEN TRUE)
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
@@ -320,12 +280,9 @@ function(AddObject Name Type)
target_compile_options(${Name}
PRIVATE
-Wall
-Werror=cast-qual
-Werror=ignored-qualifiers
-Werror=implicit-fallthrough
-Wno-trigraphs
-ffunction-sections
)
if (GCC_COLOR)
@@ -342,25 +299,12 @@ endfunction()
function(AddLibrary Name Type)
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
target_link_libraries(${Name} ${LIBS})
target_link_libraries(${Name} pthread rt vixl ${LINUX_LIBS} dl)
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
set_target_properties(${Name} PROPERTIES VISIBILITY_INLINES_HIDDEN TRUE)
set_target_properties(${Name} PROPERTIES VERSION ${FEXCore_VERSION} SOVERSION ${FEXCore_VERSION})
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
target_include_directories(${Name} PUBLIC "${PROJECT_SOURCE_DIR}/include/")
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
target_link_options(${Name}
PRIVATE
"LINKER:--gc-sections"
"LINKER:--strip-all"
"LINKER:--as-needed"
)
endif()
endfunction()
AddObject(${PROJECT_NAME}_object OBJECT)
+13 -18
View File
@@ -1,16 +1,12 @@
#pragma once
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/MathUtils.h>
#include "Common/MathUtils.h"
#include <FEXCore/Utils/LogManager.h>
#include <cstdint>
#include <cstdlib>
#include <cstring>
#include <stdint.h>
#include <stdlib.h>
#include <type_traits>
namespace FEXCore {
template<typename T>
struct BitSet final {
using ElementType = T;
@@ -20,16 +16,16 @@ struct BitSet final {
ElementType *Memory;
void Allocate(size_t Elements) {
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
LogMan::Throw::A((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(malloc(AllocateSize));
}
void Realloc(size_t Elements) {
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
LogMan::Throw::A((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(realloc(Memory, AllocateSize));
}
void Free() {
FEXCore::Allocator::free(Memory);
free(Memory);
Memory = nullptr;
}
bool Get(T Element) {
@@ -64,8 +60,8 @@ struct BitSetView final {
ElementType *Memory;
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0,
"Bitset view offset needs to be aligned to size of backing element");
LogMan::Throw::A((ElementOffset % MinimumSize) == 0,
"Bitset view offset needs to be aligned to size of backing element");
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
}
@@ -90,12 +86,11 @@ struct BitSetView final {
bool operator[](T Element) {
return Get(Element);
}
};
static_assert(sizeof(BitSet<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
static_assert(std::is_trivially_copyable_v<BitSet<uint32_t>>, "Needs to trivially copyable");
static_assert(std::is_trivially_copyable<BitSet<uint32_t>>::value, "Needs to trivially copyable");
static_assert(sizeof(BitSetView<uint32_t>) == sizeof(uintptr_t), "Needs to just be a pointer");
static_assert(std::is_trivially_copyable_v<BitSetView<uint32_t>>, "Needs to trivially copyable");
} // namespace FEXCore
static_assert(std::is_trivially_copyable<BitSetView<uint32_t>>::value, "Needs to trivially copyable");
+20 -28
View File
@@ -1,53 +1,45 @@
#include "Common/JitSymbols.h"
#include <string>
#include <sstream>
#include <unistd.h>
#include <fmt/format.h>
namespace FEXCore {
JITSymbols::JITSymbols() : fp{nullptr, std::fclose} {
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
JITSymbols::JITSymbols() {
std::stringstream PerfMap;
PerfMap << "/tmp/perf-" << getpid() << ".map";
fp.reset(fopen(PerfMap.c_str(), "wb"));
fp = fopen(PerfMap.str().c_str(), "wb");
if (fp) {
// Disable buffering on this file
setvbuf(fp.get(), nullptr, _IONBF, 0);
setvbuf(fp, nullptr, _IONBF, 0);
}
}
JITSymbols::~JITSymbols() = default;
JITSymbols::~JITSymbols() {
if (fp) {
fclose(fp);
}
}
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
void JITSymbols::Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
fmt::print(fp.get(), "{:x} {:x} JIT_0x{:x}_{:x}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
std::stringstream String;
String << std::hex << HostAddr << " " << CodeSize << " " << "JIT_0x" << GuestAddr << "_" << HostAddr << std::endl;
fwrite(String.str().c_str(), 1, String.str().size(), fp);
}
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
void JITSymbols::Register(void *HostAddr, uint32_t CodeSize, std::string const &Name) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
fmt::print(fp.get(), "{:x} {:x} {}_{:x}\n", HostAddr, CodeSize, Name, HostAddr);
std::stringstream String;
String << std::hex << HostAddr << " " << CodeSize << " " << Name << "_" << HostAddr << std::endl;
fwrite(String.str().c_str(), 1, String.str().size(), fp);
}
}
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
fmt::print(fp.get(), "{:x} {:x} {}\n", HostAddr, CodeSize, Name);
}
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
if (!fp) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
fmt::print(fp.get(), "{:x} {:x} FEXJIT\n", HostAddr, CodeSize);
}
} // namespace FEXCore
+4 -11
View File
@@ -1,24 +1,17 @@
#pragma once
#include <cstdint>
#include <cstdio>
#include <memory>
#include <string_view>
#include <string>
namespace FEXCore {
class JITSymbols final {
public:
JITSymbols();
~JITSymbols();
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
void Register(void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(void *HostAddr, uint32_t CodeSize, std::string const &Name);
private:
using FILEPtr = std::unique_ptr<FILE, decltype(&std::fclose)>;
FILEPtr fp;
FILE* fp{};
};
}
+13
View File
@@ -0,0 +1,13 @@
#pragma once
#include <stdint.h>
static inline uint64_t AlignUp(uint64_t value, uint64_t size) {
return value + (size - value % size) % size;
};
static inline uint64_t AlignDown(uint64_t value, uint64_t size) {
return value - value % size;
};
@@ -1,16 +1,19 @@
#include <FEXCore/Utils/NetStream.h>
#include "NetStream.h"
#include <cstring>
#include <sys/types.h>
#include <sys/socket.h>
#include <stdio.h>
#include <unistd.h>
namespace FEXCore::Utils {
int NetStream::NetBuf::flushBuffer(const char *buffer, size_t size) {
size_t total = 0;
// Send data
while (total < size) {
size_t sent = send(socket, (const void*)(buffer + total), size - total, MSG_NOSIGNAL);
size_t sent = send(socket, (const void*)(buffer + total), size - total, 0);
if (sent == -1) {
// lets just assume all errors are end of file.
return -1;
@@ -26,7 +29,7 @@ std::streamsize NetStream::NetBuf::xsputn(const char* buffer, std::streamsize si
// Check if the string fits neatly in our buffer
if (size <= buf_remaining) {
::memcpy(pptr(), buffer, size);
std::memcpy(pptr(), buffer, size);
pbump(size);
return size;
}
@@ -81,4 +84,3 @@ NetStream::~NetStream() {
NetStream::NetBuf::~NetBuf() {
close(socket);
}
}
@@ -1,14 +1,10 @@
#pragma once
#include <FEXCore/Utils/CompilerDefs.h>
#include <array>
#include <iostream>
#include <iterator>
#include <string.h>
namespace FEXCore::Utils {
class FEX_DEFAULT_VISIBILITY NetStream : public std::iostream {
class NetStream : public std::iostream {
public:
NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
virtual ~NetStream();
@@ -43,4 +39,3 @@ private:
std::array<char, 1500> input_buffer; // enough for a typical packet
};
};
}
+12 -53
View File
@@ -3,48 +3,13 @@
#include <cstdlib>
#include <filesystem>
#include <memory>
#include <pwd.h>
#include <system_error>
#include <unistd.h>
#include <sys/stat.h>
namespace FEXCore::Paths {
std::unique_ptr<std::string> CachePath;
std::unique_ptr<std::string> EntryCache;
char const* FindUserHomeThroughUID() {
auto passwd = getpwuid(geteuid());
if (passwd) {
return passwd->pw_dir;
}
return nullptr;
}
const char *GetHomeDirectory() {
char const *HomeDir = getenv("HOME");
// Try to get home directory from uid
if (!HomeDir) {
HomeDir = FindUserHomeThroughUID();
}
// try the PWD
if (!HomeDir) {
HomeDir = getenv("PWD");
}
// Still doesn't exit? You get local
if (!HomeDir) {
HomeDir = ".";
}
return HomeDir;
}
std::string CachePath;
std::string EntryCache;
void InitializePaths() {
CachePath = std::make_unique<std::string>();
EntryCache = std::make_unique<std::string>();
char const *HomeDir = getenv("HOME");
if (!HomeDir) {
@@ -57,35 +22,29 @@ namespace FEXCore::Paths {
char *XDGDataDir = getenv("XDG_DATA_DIR");
if (XDGDataDir) {
*CachePath = XDGDataDir;
CachePath = XDGDataDir;
}
else {
if (HomeDir) {
*CachePath = HomeDir;
CachePath = HomeDir;
}
}
*CachePath += "/.fex-emu/";
*EntryCache = *CachePath + "/EntryCache/";
CachePath += "/.fex-emu/";
EntryCache = CachePath + "/EntryCache/";
std::error_code ec{};
// Ensure the folder structure is created for our Data
if (!std::filesystem::exists(*EntryCache, ec) &&
!std::filesystem::create_directories(*EntryCache, ec)) {
LogMan::Msg::DFmt("Couldn't create EntryCache directory: '{}'", *EntryCache);
if (!std::filesystem::exists(EntryCache) &&
!std::filesystem::create_directories(EntryCache)) {
LogMan::Msg::D("Couldn't create EntryCache directory: '%s'", EntryCache.c_str());
}
}
void ShutdownPaths() {
CachePath.reset();
EntryCache.reset();
}
std::string GetCachePath() {
return *CachePath;
return CachePath;
}
std::string GetEntryCachePath() {
return *EntryCache;
return EntryCache;
}
}
-4
View File
@@ -3,10 +3,6 @@
namespace FEXCore::Paths {
void InitializePaths();
void ShutdownPaths();
const char *GetHomeDirectory();
std::string GetCachePath();
std::string GetEntryCachePath();
}
+18 -46
View File
@@ -1,6 +1,4 @@
#pragma once
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <cmath>
@@ -16,19 +14,9 @@ extern "C" {
struct X80SoftFloat {
#ifdef _M_X86_64
// Define this to push some operations to x87
// Only useful to see if precision loss is killing something
// #define DEBUG_X86_FLOAT
#ifdef DEBUG_X86_FLOAT
#define BIGFLOAT long double
#define BIGFLOATSIZE 10
#else
#define BIGFLOAT __float128
#define BIGFLOATSIZE 16
#endif
#elif defined(_M_ARM_64)
#define BIGFLOAT long double
#define BIGFLOATSIZE 16
#else
#error No 128bit float for this target!
#endif
@@ -111,7 +99,7 @@ struct X80SoftFloat {
}
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
WARN_ONCE("x87: Application used FSCALE which may have accuracy problems");
X80SoftFloat Int = FRNDINT(rhs);
BIGFLOAT Src2_d = Int;
Src2_d = exp2l(Src2_d);
@@ -121,7 +109,7 @@ struct X80SoftFloat {
}
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
WARN_ONCE("x87: Application used F2XM1 which may have accuracy problems");
BIGFLOAT Src1_d = lhs;
BIGFLOAT Result = exp2l(Src1_d);
Result -= 1.0;
@@ -129,7 +117,7 @@ struct X80SoftFloat {
}
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
WARN_ONCE("x87: Application used FYL2X which may have accuracy problems");
BIGFLOAT Src1_d = lhs;
BIGFLOAT Src2_d = rhs;
BIGFLOAT Tmp = Src2_d * log2l(Src1_d);
@@ -137,7 +125,7 @@ struct X80SoftFloat {
}
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
WARN_ONCE("x87: Application used FATAN which may have accuracy problems");
BIGFLOAT Src1_d = lhs;
BIGFLOAT Src2_d = rhs;
BIGFLOAT Tmp = atan2l(Src1_d, Src2_d);
@@ -145,21 +133,21 @@ struct X80SoftFloat {
}
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
WARN_ONCE("x87: Application used FTAN which may have accuracy problems");
BIGFLOAT Src_d = lhs;
Src_d = tanl(Src_d);
return Src_d;
}
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
WARN_ONCE("x87: Application used FSIN which may have accuracy problems");
BIGFLOAT Src_d = lhs;
Src_d = sinl(Src_d);
return Src_d;
}
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
WARN_ONCE("x87: Application used FCOS which may have accuracy problems");
BIGFLOAT Src_d = lhs;
Src_d = cosl(Src_d);
return Src_d;
@@ -170,24 +158,18 @@ struct X80SoftFloat {
}
operator float() const {
const float32_t Result = extF80_to_f32(*this);
return FEXCore::BitCast<float>(Result);
float32_t Result = extF80_to_f32(*this);
return *(float*)&Result;
}
operator double() const {
const float64_t Result = extF80_to_f64(*this);
return FEXCore::BitCast<double>(Result);
float64_t Result = extF80_to_f64(*this);
return *(double*)&Result;
}
operator BIGFLOAT() const {
#if BIGFLOATSIZE == 16
const float128_t Result = extF80_to_f128(*this);
return FEXCore::BitCast<BIGFLOAT>(Result);
#else
BIGFLOAT result{};
memcpy(&result, this, sizeof(result));
return result;
#endif
float128_t Result = extF80_to_f128(*this);
return *(BIGFLOAT*)&Result;
}
operator int16_t() const {
@@ -214,11 +196,11 @@ struct X80SoftFloat {
}
void operator=(const float rhs) {
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
*this = f32_to_extF80(*(float32_t*)&rhs);
}
void operator=(const double rhs) {
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
*this = f64_to_extF80(*(float64_t*)&rhs);
}
void operator=(const int16_t rhs) {
@@ -233,12 +215,6 @@ struct X80SoftFloat {
*this = ui64_to_extF80(rhs);
}
#if BIGFLOATSIZE == 10
void operator=(const long double rhs) {
memcpy(this, &rhs, sizeof(rhs));
}
#endif
operator void*() {
return reinterpret_cast<void*>(this);
}
@@ -250,19 +226,15 @@ struct X80SoftFloat {
}
X80SoftFloat(const float rhs) {
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
*this = f32_to_extF80(*(float32_t*)&rhs);
}
X80SoftFloat(const double rhs) {
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
*this = f64_to_extF80(*(float64_t*)&rhs);
}
X80SoftFloat(BIGFLOAT rhs) {
#if BIGFLOATSIZE == 16
*this = f128_to_extF80(FEXCore::BitCast<float128_t>(rhs));
#else
*this = FEXCore::BitCast<long double>(rhs);
#endif
*this = f128_to_extF80(*(float128_t*)&rhs);
}
X80SoftFloat(const int16_t rhs) {
+56 -393
View File
@@ -1,126 +1,43 @@
#include "Common/StringConv.h"
#include "Common/Paths.h"
#include "Utils/FileLoading.h"
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Context/Context.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
#include <assert.h>
#include <cstdlib>
#include <filesystem>
#include <fstream>
#include <functional>
#include <pwd.h>
#include <map>
#include <memory>
#include <list>
#include <optional>
#include <stddef.h>
#include <stdint.h>
#include <string>
#include <string_view>
#include <sys/sysinfo.h>
#include <system_error>
#include <type_traits>
#include <unordered_map>
#include <utility>
#include <vector>
#include <tiny-json.h>
namespace FEXCore::Context {
struct Context;
}
#include <unistd.h>
namespace FEXCore::Config {
namespace DefaultValues {
#define P(x) x
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
#include <FEXCore/Config/ConfigValues.inl>
}
namespace JSON {
struct JsonAllocator {
jsonPool_t PoolObject;
std::unique_ptr<std::list<json_t>> json_objects;
};
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
json_t* PoolInit(jsonPool_t* Pool) {
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
alloc->json_objects = std::make_unique<std::list<json_t>>();
return &*alloc->json_objects->emplace(alloc->json_objects->end());
char const* FindUserHomeThroughUID() {
auto passwd = getpwuid(geteuid());
if (passwd) {
return passwd->pw_dir;
}
return nullptr;
}
json_t* PoolAlloc(jsonPool_t* Pool) {
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
return &*alloc->json_objects->emplace(alloc->json_objects->end());
}
const char *GetHomeDirectory() {
char const *HomeDir = getenv("HOME");
static void LoadJSonConfig(const std::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
std::vector<char> Data;
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
return;
// Try to get home directory from uid
if (!HomeDir) {
HomeDir = FindUserHomeThroughUID();
}
JsonAllocator Pool {
.PoolObject = {
.init = PoolInit,
.alloc = PoolAlloc,
},
};
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
if (!json) {
LogMan::Msg::EFmt("Couldn't create json");
return;
// try the PWD
if (!HomeDir) {
HomeDir = getenv("PWD");
}
json_t const* ConfigList = json_getProperty(json, "Config");
if (!ConfigList) {
LogMan::Msg::EFmt("Couldn't get config list");
return;
// Still doesn't exit? You get local
if (!HomeDir) {
HomeDir = ".";
}
for (json_t const* ConfigItem = json_getChild(ConfigList);
ConfigItem != nullptr;
ConfigItem = json_getSibling(ConfigItem)) {
const char* ConfigName = json_getName(ConfigItem);
const char* ConfigString = json_getValue(ConfigItem);
if (!ConfigName) {
LogMan::Msg::EFmt("Couldn't get config name");
return;
}
if (!ConfigString) {
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
return;
}
Func(ConfigName, ConfigString);
}
}
}
std::string GetDataDirectory() {
std::string DataDir{};
char const *HomeDir = Paths::GetHomeDirectory();
char const *DataXDG = getenv("XDG_DATA_HOME");
char const *DataOverride = getenv("FEX_APP_DATA_LOCATION");
if (DataOverride) {
// Data override will override the complete directory
DataDir = DataOverride;
}
else {
DataDir = DataXDG ?: HomeDir;
DataDir += "/.fex-emu/";
}
return DataDir;
return HomeDir;
}
std::string GetConfigDirectory(bool Global) {
@@ -129,22 +46,15 @@ namespace JSON {
ConfigDir = GLOBAL_DATA_DIRECTORY;
}
else {
char const *HomeDir = Paths::GetHomeDirectory();
char const *HomeDir = GetHomeDirectory();
char const *ConfigXDG = getenv("XDG_CONFIG_HOME");
char const *ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
if (ConfigOverride) {
// Config override completely overrides the config directory
ConfigDir = ConfigOverride;
}
else {
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
ConfigDir += "/.fex-emu/";
}
ConfigDir = ConfigXDG ? ConfigXDG : HomeDir;
ConfigDir += "/.fex-emu/";
// Ensure the folder structure is created for our configuration
std::error_code ec{};
if (!std::filesystem::exists(ConfigDir, ec) &&
!std::filesystem::create_directories(ConfigDir, ec)) {
if (!std::filesystem::exists(ConfigDir) &&
!std::filesystem::create_directories(ConfigDir)) {
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigDir.c_str());
// Let's go local in this case
return "./";
}
@@ -154,44 +64,34 @@ namespace JSON {
}
std::string GetConfigFileLocation() {
std::string ConfigFile{};
const char *AppConfig = getenv("FEX_APP_CONFIG");
if (AppConfig) {
// App config environment variable overwrites only the config file
ConfigFile = AppConfig;
}
else {
ConfigFile = GetConfigDirectory(false) + "Config.json";
}
std::string ConfigFile = GetConfigDirectory(false) + "Config.json";
return ConfigFile;
}
std::string GetApplicationConfig(const std::string &Filename, bool Global) {
std::string GetApplicationConfig(std::string &Filename, bool Global) {
std::string ConfigFile = GetConfigDirectory(Global);
std::error_code ec{};
if (!Global &&
!std::filesystem::exists(ConfigFile, ec) &&
!std::filesystem::create_directories(ConfigFile, ec)) {
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
!std::filesystem::exists(ConfigFile) &&
!std::filesystem::create_directories(ConfigFile)) {
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigFile.c_str());
// Let's go local in this case
return "./" + Filename + ".json";
return "./";
}
ConfigFile += "AppConfig/";
// Attempt to create the local folder if it doesn't exist
if (!Global &&
!std::filesystem::exists(ConfigFile, ec) &&
!std::filesystem::create_directories(ConfigFile, ec)) {
// Let's go local in this case
return "./" + Filename + ".json";
}
ConfigFile += Filename + ".json";
ConfigFile += "AppConfig/" + Filename + ".json";
return ConfigFile;
}
std::string GetDataDirectory() {
std::string DataDir{};
char const *HomeDir = GetHomeDirectory();
char const *DataXDG = getenv("XDG_DATA_HOME");
DataDir = DataXDG ?: HomeDir;
DataDir += "/.fex-emu/";
return DataDir;
}
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
}
@@ -291,8 +191,7 @@ namespace JSON {
void MetaLayer::MergeConfigMap(const LayerOptions &Options) {
// Insert this layer's options, overlaying previous options that exist here
for (auto &it : Options) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV ||
it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV) {
MergeEnvironmentVariables(it.first, it.second);
}
else {
@@ -320,7 +219,7 @@ namespace JSON {
}
}
std::string ExpandPath(std::string const &ContainerPrefix, std::string PathName) {
std::string ExpandPath(std::string PathName) {
if (PathName.empty()) {
return {};
}
@@ -341,74 +240,10 @@ namespace JSON {
Path = std::filesystem::absolute(Path);
// Only return if it exists
std::error_code ec{};
if (std::filesystem::exists(Path, ec)) {
if (std::filesystem::exists(Path)) {
return Path;
}
}
else {
// If the containerprefix and pathname isn't empty
// Then we check if the pathname exists in our current namespace
// If the path DOESN'T exist but DOES exist with the prefix applied
// then redirect to the prefix
//
// This might not be expected behaviour for some edge cases but since
// all paths aren't mounted inside the container, then it'll be fine
//
// Main catch case for this is the default thunk install folders
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
if (!ContainerPrefix.empty() && !PathName.empty()) {
if (!std::filesystem::exists(PathName)) {
auto ContainerPath = ContainerPrefix + PathName;
if (std::filesystem::exists(ContainerPath)) {
return ContainerPath;
}
}
}
}
return {};
}
std::string ltrim(std::string String) {
size_t pos = std::string::npos;
if ((pos = String.find_first_not_of(" \t\n\r")) != std::string::npos) {
String.erase(0, pos);
}
return String;
}
std::string rtrim(std::string String) {
size_t pos = std::string::npos;
if ((pos = String.find_last_not_of(" \t\n\r")) != std::string::npos) {
String.erase(String.begin() + pos + 1, String.end());
}
return String;
}
std::string trim(std::string String) {
return rtrim(ltrim(String));
}
std::string FindContainerPrefix() {
// We only support pressure-vessel at the moment
const static std::string ContainerManager = "/run/host/container-manager";
if (std::filesystem::exists(ContainerManager)) {
std::vector<char> Manager{};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
std::string ManagerStr = Manager.data();
ManagerStr = trim(ManagerStr);
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
// We are running inside of pressure vessel
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
return "/run/host/";
}
}
}
return {};
}
@@ -424,9 +259,8 @@ namespace JSON {
}
}
std::string ContainerPrefix { FindContainerPrefix() };
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
auto NewPath = ExpandPath(ContainerPrefix, PathName);
auto ExpandPathIfExists = [](FEXCore::Config::ConfigOption Config, std::string PathName) {
auto NewPath = ExpandPath(PathName);
if (!NewPath.empty()) {
FEXCore::Config::EraseSet(Config, NewPath);
}
@@ -434,7 +268,7 @@ namespace JSON {
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
FEX_CONFIG_OPT(PathName, ROOTFS);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
auto ExpandedString = ExpandPath(PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
@@ -442,8 +276,7 @@ namespace JSON {
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
std::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
std::error_code ec{};
if (std::filesystem::exists(NamedRootFS, ec)) {
if (std::filesystem::exists(NamedRootFS)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
}
}
@@ -458,19 +291,7 @@ namespace JSON {
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
}
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
std::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
std::error_code ec{};
if (std::filesystem::exists(NamedConfig, ec)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
}
}
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKCONFIG, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
@@ -501,15 +322,11 @@ namespace JSON {
return Meta->Get(Option);
}
void Set(ConfigOption Option, std::string_view Data) {
void Set(ConfigOption Option, std::string Data) {
Meta->Set(Option, Data);
}
void Erase(ConfigOption Option) {
Meta->Erase(Option);
}
void EraseSet(ConfigOption Option, std::string_view Data) {
void EraseSet(ConfigOption Option, std::string Data) {
Meta->EraseSet(Option, Data);
}
@@ -548,17 +365,6 @@ namespace JSON {
}
}
template<>
std::string Value<std::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
}
else {
return std::string(Default);
}
}
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
@@ -583,148 +389,5 @@ namespace JSON {
*List = **Value;
}
}
template void Value<std::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, std::list<std::string> *List);
// Application loaders
class MainLoader final : public FEXCore::Config::OptionMapper {
public:
explicit MainLoader();
explicit MainLoader(std::string ConfigFile);
void Load() override;
private:
std::string Config;
};
class AppLoader final : public FEXCore::Config::OptionMapper {
public:
explicit AppLoader(const std::string& Filename, bool Global);
void Load();
private:
std::string Config;
};
class EnvLoader final : public FEXCore::Config::Layer {
public:
explicit EnvLoader(char *const _envp[]);
void Load() override;
private:
char *const *envp;
};
static const std::map<std::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
#include <FEXCore/Config/ConfigValues.inl>
}};
static const std::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
#include <FEXCore/Config/ConfigValues.inl>
}};
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
: FEXCore::Config::Layer(Layer) {
}
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
auto it = ConfigLookup.find(ConfigName);
if (it != ConfigLookup.end()) {
Set(it->second, ConfigString);
}
}
MainLoader::MainLoader()
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
, Config{FEXCore::Config::GetConfigFileLocation()} {
}
MainLoader::MainLoader(std::string ConfigFile)
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
, Config{std::move(ConfigFile)} {
}
void MainLoader::Load() {
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
MapNameToOption(Name, ConfigString);
});
}
AppLoader::AppLoader(const std::string& Filename, bool Global)
: FEXCore::Config::OptionMapper(Global ? FEXCore::Config::LayerType::LAYER_GLOBAL_APP : FEXCore::Config::LayerType::LAYER_LOCAL_APP) {
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
// Immediately load so we can reload the meta layer
Load();
}
void AppLoader::Load() {
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
MapNameToOption(Name, ConfigString);
});
}
EnvLoader::EnvLoader(char *const _envp[])
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
, envp {_envp} {
}
void EnvLoader::Load() {
std::unordered_map<std::string_view, std::string_view> EnvMap;
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
std::string_view Var(*pvar);
size_t pos = Var.rfind('=');
if (std::string::npos == pos)
continue;
std::string_view Key = Var.substr(0,pos);
std::string_view Value {Var.substr(pos+1)};
#define ENVLOADER
#include <FEXCore/Config/ConfigOptions.inl>
EnvMap[Key]=Value;
}
std::function GetVar = [=](const std::string_view id) -> std::optional<std::string_view> {
if (EnvMap.find(id) != EnvMap.end())
return EnvMap.at(id);
// If envp[] was empty, search using std::getenv()
const char* vs = std::getenv(id.data());
if (vs) {
return vs;
}
else {
return std::nullopt;
}
};
std::optional<std::string_view> Value;
for (auto &it : EnvConfigLookup) {
if ((Value = GetVar(it.first)).has_value()) {
Set(it.second, std::string(*Value));
}
}
}
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
if (File) {
return std::make_unique<FEXCore::Config::MainLoader>(*File);
}
else {
return std::make_unique<FEXCore::Config::MainLoader>();
}
}
std::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const std::string& Filename, bool Global) {
return std::make_unique<FEXCore::Config::AppLoader>(Filename, Global);
}
std::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
return std::make_unique<FEXCore::Config::EnvLoader>(_envp);
}
}
@@ -58,7 +58,7 @@
},
"ThunkHostLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks/",
"Default": "",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
@@ -66,7 +66,7 @@
},
"ThunkGuestLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks/",
"Default": "",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
@@ -77,14 +77,7 @@
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks.",
"This can be a filesystem path",
"\teg: ~/MyThunkConfig.json",
"Or this can be a named of a Thunk config file",
"If the named config file exists in the FEX data folder folder the it will use that one",
"\teg: $HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>",
"Or if you have XDG_DATA_HOME the config will search in that directory",
"\teg: $XDG_DATA_HOME/.fex-emu/ThunkConfigs/<ThunkConfig name>"
"A json file specifying where to overlay the thunks."
]
},
"Env": {
@@ -94,16 +87,6 @@
"Desc": [
"Adds an environment variable to the emulated environment."
]
},
"HostEnv": {
"Type": "strarray",
"Default": "",
"ShortArg": "H",
"Desc": [
"Adds an environment variable to the host environment.",
"This can be useful for setting environment variables that thunks can pick up.",
"Typically isn't necessary since the guest libc isn't thunked. But is possible."
]
}
},
"Debug": {
@@ -146,50 +129,6 @@
"Desc": [
"Disables optimizations passes for debugging."
]
},
"SRA": {
"Type": "bool",
"Default": "true",
"Desc": [
"Set to false to disable Static Register Allocation"
]
},
"Force32BitAllocator": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces use of the 32-bit allocator on 32-bit applications",
"Used to work around ulimit problems of CI runner",
"Potentially useful for debugging memory problems",
"32-bit allocator is always used if your host kernel is older than 4.17"
]
},
"GlobalJITNaming": {
"Type": "bool",
"Default": "false",
"Desc": [
"Uses JITSymbols to name all JIT state as one symbol",
"Useful for querying how much time is spent inside of the JIT",
"Profiling tools will show JIT time as FEXJIT"
]
},
"LibraryJITNaming": {
"Type": "bool",
"Default": "false",
"Desc": [
"Uses JITSymbols to name JIT symbols grouped by library",
"Useful for querying how much time is spent in each guest library",
"Can be used to help guide thunk generation"
]
},
"BlockJITNaming": {
"Type": "bool",
"Default": "false",
"Desc": [
"Uses JITSymbols to name JIT symbols",
"Useful for determining hot blocks of code",
"Has some file writing overhead per JIT block"
]
}
},
"Logging": {
@@ -201,18 +140,9 @@
"Disables logging"
]
},
"OutputSocket": {
"Type": "str",
"Default": "",
"Desc": [
"Socket to connect to",
"eg: localhost:8087",
"If set will override the OutputLog location"
]
},
"OutputLog": {
"Type": "str",
"Default": "stderr",
"Default": "stdout",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
@@ -258,22 +188,6 @@
"Removes the calculation of the parity flag from GPR instructions.",
"Assuming no uses rely on it"
]
},
"ParanoidTSO": {
"Type": "bool",
"Default": "false",
"Desc": [
"Makes TSO operations even more strict.",
"Forces vector loadstores to also become atomic."
]
},
"StallProcess": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces a process to stall out on initialization",
"Useful for a process that keeps restarting and doesn't work"
]
}
},
"Misc": {
@@ -285,14 +199,6 @@
"Captures both the loaded executable and libraries it loads."
]
},
"AOTIRGenerate": {
"Type": "bool",
"Default": "false",
"Desc": [
"Scans file for executable code and generates an AOT IR cache.",
"Does not run the executable."
]
},
"AOTIRLoad": {
"Type": "bool",
"Default": "false",
+16 -44
View File
@@ -2,20 +2,10 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/Core.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Core/SignalDelegator.h>
#include "FEXCore/Debug/InternalThreadState.h"
#include <string.h>
#include <utility>
namespace FEXCore::HLE {
class SyscallVisitor;
}
#include <FEXCore/Debug/X86Tables.h>
namespace FEXCore::Context {
void InitializeStaticTables(OperatingMode Mode) {
@@ -24,10 +14,6 @@ namespace FEXCore::Context {
IR::InstallOpcodeHandlers(Mode);
}
void ShutdownStaticTables() {
FEXCore::Paths::ShutdownPaths();
}
FEXCore::Context::Context *CreateNewContext() {
return new FEXCore::Context::Context{};
}
@@ -43,15 +29,16 @@ namespace FEXCore::Context {
delete CTX;
}
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
bool InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
return CTX->InitCore(Loader);
}
void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler) {
CTX->CustomExitHandler = std::move(handler);
void SetExitHandler(FEXCore::Context::Context *CTX,
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> handler) {
CTX->CustomExitHandler = handler;
}
ExitHandler GetExitHandler(FEXCore::Context::Context *CTX) {
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> GetExitHandler(FEXCore::Context::Context *CTX) {
return CTX->CustomExitHandler;
}
@@ -63,9 +50,6 @@ namespace FEXCore::Context {
CTX->Step();
}
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
Thread->CTX->CompileBlock(Thread->CurrentFrame, GuestRIP);
}
FEXCore::Context::ExitReason RunUntilExit(FEXCore::Context::Context *CTX) {
return CTX->RunUntilExit();
@@ -110,26 +94,22 @@ namespace FEXCore::Context {
void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t Syscall, [[maybe_unused]] FEXCore::HLE::SyscallVisitor *Visitor) {
}
void HandleCallback(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
CTX->HandleCallback(Thread, RIP);
void HandleCallback(FEXCore::Context::Context *CTX, uint64_t RIP) {
CTX->HandleCallback(RIP);
}
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
CTX->RegisterHostSignalHandler(Signal, Func, Required);
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
CTX->RegisterHostSignalHandler(Signal, Func);
}
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
CTX->RegisterFrontendHostSignalHandler(Signal, Func, Required);
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
CTX->RegisterFrontendHostSignalHandler(Signal, Func);
}
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
return CTX->CreateThread(NewThreadState, ParentTID);
}
void ExecutionThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
return CTX->ExecutionThread(Thread);
}
void InitializeThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread) {
return CTX->InitializeThread(Thread);
}
@@ -162,20 +142,12 @@ namespace FEXCore::Context {
return CTX->CPUID.RunFunction(Function, Leaf);
}
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader) {
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::istream>(const std::string&)> CacheReader) {
CTX->AOTIRLoader = CacheReader;
}
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
CTX->AOTIRWriter = CacheWriter;
}
void FinalizeAOTIRCache(FEXCore::Context::Context *CTX) {
CTX->FinalizeAOTIRCache();
}
void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
CTX->WriteFilesWithCode(Writer);
bool WriteAOTIR(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
return CTX->WriteAOTIRCache(CacheWriter);
}
void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name) {
+49 -184
View File
@@ -1,49 +1,44 @@
#pragma once
#include "Common/JitSymbols.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/HostFeatures.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Utils/Event.h>
#include <stdint.h>
#include <atomic>
#include <condition_variable>
#include <functional>
#include <istream>
#include <map>
#include <memory>
#include <mutex>
#include <shared_mutex>
#include <stddef.h>
#include <string>
#include <optional>
#include <ostream>
#include <set>
#include <unordered_map>
#include <queue>
#include <vector>
namespace FEXCore {
class CodeLoader;
class ThunkHandler;
class BlockSamplingData;
class GdbServer;
class SiganlDelegator;
namespace CPU {
class Arm64JITCore;
class X86JITCore;
}
namespace HLE {
struct SyscallArguments;
class SyscallHandler;
}
}
namespace FEXCore::IR {
class RegisterAllocationPass;
class RegisterAllocationData;
class IRListView;
namespace Validation {
@@ -57,38 +52,6 @@ namespace FEXCore::Context {
MODE_SINGLESTEP = 1,
};
struct AOTIRInlineEntry {
uint64_t GuestHash;
uint64_t GuestLength;
/* RAData followed by IRData */
uint8_t InlineData[0];
IR::RegisterAllocationData *GetRAData();
IR::IRListView *GetIRData();
};
struct AOTIRInlineIndexEntry {
uint64_t GuestStart;
uint64_t DataOffset;
};
struct AOTIRInlineIndex {
uint64_t Count;
uint64_t DataBase;
AOTIRInlineIndexEntry Entries[0];
AOTIRInlineEntry *Find(uint64_t GuestStart);
AOTIRInlineEntry *GetInlineEntry(uint64_t DataOffset);
};
struct AOTIRCaptureCacheEntry {
std::unique_ptr<std::ostream> Stream;
std::map<uint64_t, uint64_t> Index;
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData);
};
struct Context {
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
@@ -115,23 +78,16 @@ namespace FEXCore::Context {
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(ABINoPF, ABINOPF);
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(Core, CORE);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
FEX_CONFIG_OPT(DumpIR, DUMPIR);
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
} Config;
using IntCallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
using IntCallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
IntCallbackReturn InterpreterCallbackReturn;
FEXCore::HostFeatures HostFeatures;
@@ -154,34 +110,30 @@ namespace FEXCore::Context {
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
CustomCPUFactoryType CustomCPUFactory;
FEXCore::Context::ExitHandler CustomExitHandler;
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> CustomExitHandler;
struct AOTIRCacheEntry {
AOTIRInlineIndex *Array;
void *mapping;
size_t size;
uint64_t start;
uint64_t len;
uint64_t crc;
IR::IRListView *IR;
IR::RegisterAllocationData *RAData;
};
std::unordered_map<std::string, AOTIRCacheEntry> AOTIRCache;
std::function<int(const std::string&)> AOTIRLoader;
std::function<std::unique_ptr<std::ostream>(const std::string&)> AOTIRWriter;
std::unordered_map<std::string, AOTIRCaptureCacheEntry> AOTIRCaptureCache;
std::function<std::unique_ptr<std::istream>(const std::string&)> AOTIRLoader;
std::unordered_map<std::string, std::map<uint64_t, AOTIRCacheEntry>> AOTIRCache;
struct AddrToFileEntry {
uint64_t Start;
uint64_t Len;
uint64_t Offset;
std::string fileid;
std::string filename;
void *CachedFileEntry;
bool ContainsCode;
};
using AddrToFileMapType = std::map<uint64_t, AddrToFileEntry>;
AddrToFileMapType AddrToFile;
std::map<std::string, std::string> FilesWithCode;
std::map<uint64_t, AddrToFileEntry> AddrToFile;
AddrToFileMapType::iterator FindAddrForFile(uint64_t Entry, uint64_t Length);
#ifdef BLOCKSTATS
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
#endif
@@ -192,9 +144,9 @@ namespace FEXCore::Context {
Context();
~Context();
FEXCore::Core::InternalThreadState* InitCore(FEXCore::CodeLoader *Loader);
bool InitCore(FEXCore::CodeLoader *Loader);
FEXCore::Context::ExitReason RunUntilExit();
int GetProgramStatus() const;
int GetProgramStatus();
bool IsPaused() const { return !Running; }
void Pause();
void Run();
@@ -205,12 +157,12 @@ namespace FEXCore::Context {
void StopThread(FEXCore::Core::InternalThreadState *Thread);
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
bool GetGdbServerStatus() { return (bool)DebugServer; }
void StartGdbServer();
void StopGdbServer();
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
void HandleCallback(uint64_t RIP);
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
static void RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
@@ -226,113 +178,37 @@ namespace FEXCore::Context {
bool GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data);
bool FindHostCodeForRIP(uint64_t RIP, uint8_t **Code);
struct GenerateIRResult {
FEXCore::IR::IRListView* IRList;
// User's responsibility to deallocate this.
FEXCore::IR::RegisterAllocationData* RAData;
uint64_t TotalInstructions;
uint64_t TotalInstructionsLength;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
// XXX:
// bool FindIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir);
// void SetIRForRIP(uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir);
void LoadEntryList();
struct CompileCodeResult {
void* CompiledCode;
FEXCore::IR::IRListView* IRData;
FEXCore::Core::DebugData* DebugData;
// User's responsibility to deallocate this.
FEXCore::IR::RegisterAllocationData* RAData;
bool GeneratedIR;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
std::tuple<FEXCore::IR::IRListView *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t, uint64_t, uint64_t> GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
std::tuple<void *, FEXCore::IR::IRListView *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool, uint64_t, uint64_t> CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
// same as CompileBlock, but aborts on failure
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
bool LoadAOTIRCache(int streamfd);
void FinalizeAOTIRCache();
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer);
/**
* @brief Initializes the JIT compilers for the thread
*
* @param State The internal FEX thread state object
* @param CompileThread Is this for the compile service or not?
*
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
* This is exposed because the CompileService needs to initialize compilers while copying data from
* the paired InternalThreadState that it is compiling code for
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
bool LoadAOTIRCache(std::istream &stream);
bool WriteAOTIRCache(std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
// Used for thread creation from syscalls
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread
*
* @param NewThreadState The initial thread state to setup for our state
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* OS thread Creation:
* - Thread = CreateThread(NewState, PPID);
* - InitializeThread(Thread);
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* State, bool CompileThread);
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
/**
* @brief Initializes the TLS data for a thread
*
* @param Thread The internal FEX thread state object
*/
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
*
* @param Thread The internal FEX thread state object
*
* The OS thread will wait until RunThread is executed
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
void InitializeThread(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Starts the OS thread object to start executing guest code
*
* @param Thread The internal FEX thread state object
*/
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
void RunThread(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Destroys this FEX thread object and stops tracking it internally
*
* @param Thread The internal FEX thread state object
*/
void DestroyThread(FEXCore::Core::InternalThreadState *Thread);
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
void CleanupAfterFork(FEXCore::Core::InternalThreadState *ExceptForThread);
std::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
std::vector<FEXCore::Core::InternalThreadState*> *const GetThreads() { return &Threads; }
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
#if ENABLE_JITSYMBOLS
FEXCore::JITSymbols Symbols;
#endif
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread);
@@ -341,35 +217,24 @@ namespace FEXCore::Context {
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
private:
/**
* @brief Does some final thread initialization
*
* @param Thread The internal FEX thread state object
*
* InitCore and CreateThread both call this to finish up thread object initialization
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
void WaitForIdleWithTimeout();
void NotifyPause();
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length);
FEXCore::CodeLoader *LocalLoader{};
// Entry Cache
std::optional<std::string> GetFilenameHash(std::string const &Filename) const;
void AddThreadRIPsToEntryList(FEXCore::Core::InternalThreadState *Thread);
void SaveEntryList();
std::set<uint64_t> EntryList;
std::vector<uint64_t> InitLocations;
uint64_t StartingRIP;
std::mutex ExitMutex;
std::unique_ptr<GdbServer> DebugServer;
std::shared_mutex AOTIRCacheLock;
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
std::queue<std::function<void()>> AOTIRCaptureCacheWriteoutQueue;
void AOTIRCaptureCacheWriteoutQueue_Flush();
void AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn);
bool StartPaused = false;
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
};
File diff suppressed because it is too large. Load diff
@@ -12,45 +12,6 @@ namespace FEXCore::ArchHelpers::Arm64 {
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
constexpr uint32_t LDAXP_MASK = 0xBF'FF'80'00;
constexpr uint32_t LDAXP_INST = 0x88'7F'80'00;
constexpr uint32_t STLXP_MASK = 0xBF'E0'80'00;
constexpr uint32_t STLXP_INST = 0x88'20'80'00;
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
constexpr uint32_t CBNZ_MASK = 0x7F'00'00'00;
constexpr uint32_t CBNZ_INST = 0x35'00'00'00;
constexpr uint32_t ALU_OP_MASK = 0x7F'00'00'00;
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
constexpr uint32_t AND_INST = 0x0A'00'00'00;
constexpr uint32_t OR_INST = 0x2A'00'00'00;
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
constexpr uint32_t CLREX_MASK = 0xFF'FF'F0'FF;
constexpr uint32_t CLREX_INST = 0xD5'03'30'5F;
enum ExclusiveAtomicPairType {
TYPE_SWAP,
TYPE_ADD,
TYPE_SUB,
TYPE_AND,
TYPE_OR,
TYPE_EOR,
TYPE_NEG, // This is just a sub with zero. Need to know the differences
};
// Load ops are 4 bits
// Acquire and release bits are independent on the instruction
constexpr uint32_t ATOMIC_ADD_OP = 0b0000;
@@ -63,34 +24,7 @@ namespace FEXCore::ArchHelpers::Arm64 {
constexpr uint32_t ATOMIC_UMIN_OP = 0b0111;
constexpr uint32_t ATOMIC_SWAP_OP = 0b1000;
constexpr uint32_t REGISTER_MASK = 0b11111;
constexpr uint32_t RD_OFFSET = 0;
constexpr uint32_t RN_OFFSET = 5;
constexpr uint32_t RM_OFFSET = 16;
constexpr uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
0b1011'0000'0000; // Inner shareable all
inline uint32_t GetRdReg(uint32_t Instr) {
return (Instr >> RD_OFFSET) & REGISTER_MASK;
}
inline uint32_t GetRnReg(uint32_t Instr) {
return (Instr >> RN_OFFSET) & REGISTER_MASK;
}
inline uint32_t GetRmReg(uint32_t Instr) {
return (Instr >> RM_OFFSET) & REGISTER_MASK;
}
bool HandleAtomicLoad(void *_ucontext, void *_info, uint32_t Instr);
bool HandleAtomicStore(void *_ucontext, void *_info, uint32_t Instr);
bool HandleAtomicLoad128(void *_ucontext, void *_info, uint32_t Instr);
uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info);
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr);
uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr);
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr);
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr);
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr);
[[nodiscard]] bool HandleSIGBUS(bool ParanoidTSO, int Signal, void *info, void *ucontext);
}
@@ -1,15 +1,9 @@
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Core/CoreState.h>
#include "aarch64/cpu-aarch64.h"
#include "cpu-features.h"
#include "aarch64/instructions-aarch64.h"
#include "utils-vixl.h"
#include <tuple>
namespace FEXCore::CPU {
#define STATE x28
@@ -33,19 +27,8 @@ Arm64Emitter::Arm64Emitter(size_t size) : vixl::aarch64::Assembler(size, vixl::a
SetCPUFeatures(Features);
if (!SupportsAtomics) {
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
WARN_ONCE("Host CPU doesn't support atomics. Expect bad performance");
}
#ifdef _M_ARM_64
// We need to get the CPU's cache line size
// We expect sane targets that have correct cacheline sizes across clusters
uint64_t CTR;
__asm volatile ("mrs %[ctr], ctr_el0"
: [ctr] "=r"(CTR));
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
ICacheLineSize = 4 << (CTR & 0xF);
#endif
}
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) {
@@ -148,53 +131,23 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
}
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t SpillMask) {
if (StaticRegisterAllocation()) {
for (size_t i = 0; i < SRA64.size(); i+=2) {
auto Reg1 = SRA64[i];
auto Reg2 = SRA64[i+1];
if (((1U << Reg1.GetCode()) & SpillMask) &&
((1U << Reg2.GetCode()) & SpillMask)) {
stp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg1.GetCode()) & SpillMask)) {
str(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg2.GetCode()) & SpillMask)) {
str(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
}
}
void Arm64Emitter::SpillStaticRegs() {
for (size_t i = 0; i < SRA64.size(); i+=2) {
stp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
if (FPRs) {
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
stp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
}
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
stp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
}
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t FillMask) {
if (StaticRegisterAllocation()) {
for (size_t i = 0; i < SRA64.size(); i+=2) {
auto Reg1 = SRA64[i];
auto Reg2 = SRA64[i+1];
if (((1U << Reg1.GetCode()) & FillMask) &&
((1U << Reg2.GetCode()) & FillMask)) {
ldp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg1.GetCode()) & FillMask)) {
ldr(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg2.GetCode()) & FillMask)) {
ldr(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
}
}
void Arm64Emitter::FillStaticRegs() {
for (size_t i = 0; i < SRA64.size(); i+=2) {
ldp(SRA64[i], SRA64[i+1], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
if (FPRs) {
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
ldp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
}
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
ldp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
}
@@ -258,7 +211,7 @@ void Arm64Emitter::ResetStack() {
}
void Arm64Emitter::Align16B() {
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
uint64_t CurrentOffset = GetBuffer()->GetOffsetAddress<uint64_t>(GetCursorOffset());
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
nop();
}
@@ -1,16 +1,7 @@
#pragma once
#include "aarch64/assembler-aarch64.h"
#include "aarch64/constants-aarch64.h"
#include "aarch64/cpu-aarch64.h"
#include "aarch64/operands-aarch64.h"
#include "platform-vixl.h"
#include "FEXCore/Config/Config.h"
#include <array>
#include <stddef.h>
#include <stdint.h>
#include <utility>
namespace FEXCore::CPU {
using namespace vixl;
@@ -65,8 +56,8 @@ protected:
bool SupportsRCPC{};
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant);
void SpillStaticRegs(bool FPRs = true, uint32_t SpillMask = ~0U);
void FillStaticRegs(bool FPRs = true, uint32_t FillMask = ~0U);
void SpillStaticRegs();
void FillStaticRegs();
void PushDynamicRegsAndLR();
void PopDynamicRegsAndLR();
@@ -78,11 +69,6 @@ protected:
void Align16B();
uint32_t SpillSlots{};
uint32_t DCacheLineSize{};
uint32_t ICacheLineSize{};
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
};
}
@@ -1,7 +1,6 @@
#include "Interface/Core/ArchHelpers/Arm64.h"
#include <FEXCore/Utils/LogManager.h>
#include <stdint.h>
namespace FEXCore::ArchHelpers::Arm64 {
@@ -12,16 +11,16 @@ namespace FEXCore::ArchHelpers::Arm64 {
// Obvously such a configuration can't do the actual arm64-specific stuff
bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
ERROR_AND_DIE_FMT("HandleCASPAL Not Implemented");
ERROR_AND_DIE("HandleCASPAL Not Implemented");
}
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
ERROR_AND_DIE_FMT("HandleCASAL Not Implemented");
ERROR_AND_DIE("HandleCASAL Not Implemented");
}
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
ERROR_AND_DIE("HandleAtomicMemOp Not Implemented");
}
#endif
}
}
+22 -75
View File
@@ -13,26 +13,16 @@
namespace FEXCore::ArchHelpers::Context {
enum ContextFlags : uint32_t {
CONTEXT_FLAG_INJIT = (1U << 0),
CONTEXT_FLAG_32BIT = (1U << 1),
};
struct X86ContextBackup {
// Host State
// RIP and RSP is stored in GPRs here
uint64_t GPRs[23];
FEXCore::x86_64::_libc_fpstate FPRState;
uint64_t sa_mask;
// Guest state
int Signal;
uint32_t Flags;
uint64_t OriginalRIP;
uint64_t FPStateLocation;
uint64_t UContextLocation;
uint64_t SigInfoLocation;
FEXCore::Core::CPUState GuestState;
static constexpr int RedZoneSize = 128;
};
@@ -45,26 +35,15 @@ struct ArmContextBackup {
uint32_t FPSR;
uint32_t FPCR;
__uint128_t FPRs[32];
uint64_t sa_mask;
// Guest state
int Signal;
uint32_t Flags;
uint64_t OriginalRIP;
uint64_t FPStateLocation;
uint64_t UContextLocation;
uint64_t SigInfoLocation;
FEXCore::Core::CPUState GuestState;
// Arm64 doesn't have a red zone
static constexpr int RedZoneSize = 0;
};
static inline ucontext_t* GetUContext(void* ucontext) {
ucontext_t* _context = (ucontext_t*)ucontext;
return _context;
}
static inline mcontext_t* GetMContext(void* ucontext) {
ucontext_t* _context = (ucontext_t*)ucontext;
return &_context->uc_mcontext;
@@ -73,20 +52,6 @@ static inline mcontext_t* GetMContext(void* ucontext) {
#ifdef _M_ARM_64
constexpr uint32_t FPR_MAGIC = 0x46508001U;
struct HostCTXHeader {
uint32_t Magic;
uint32_t Size;
};
struct HostFPRState {
HostCTXHeader Head;
uint32_t FPSR;
uint32_t FPCR;
__uint128_t FPRs[32];
};
static inline uint64_t GetSp(void* ucontext) {
return GetMContext(ucontext)->sp;
}
@@ -119,19 +84,24 @@ static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
GetMContext(ucontext)->regs[id] = val;
}
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
auto MContext = GetMContext(ucontext);
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&MContext->__reserved[0]);
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
constexpr uint32_t FPR_MAGIC = 0x46508001U;
return HostState->FPRs[id];
}
struct HostCTXHeader {
uint32_t Magic;
uint32_t Size;
};
struct HostFPRState {
HostCTXHeader Head;
uint32_t FPSR;
uint32_t FPCR;
__uint128_t FPRs[32];
};
using ContextBackup = ArmContextBackup;
template <typename T>
static inline void BackupContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, ArmContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
memcpy(&Backup->GPRs[0], &_mcontext->regs[0], 31 * sizeof(uint64_t));
@@ -141,27 +111,22 @@ static inline void BackupContext(void* ucontext, T *Backup) {
// Host FPR state starts at _mcontext->reserved[0];
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
LogMan::Throw::A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
Backup->FPSR = HostState->FPSR;
Backup->FPCR = HostState->FPCR;
memcpy(&Backup->FPRs[0], &HostState->FPRs[0], 32 * sizeof(__uint128_t));
// Save the signal mask so we can restore it
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
template <typename T>
static inline void RestoreContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, ArmContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
LogMan::Throw::A(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x%08x", HostState->Head.Magic);
memcpy(&HostState->FPRs[0], &Backup->FPRs[0], 32 * sizeof(__uint128_t));
HostState->FPCR = Backup->FPCR;
HostState->FPSR = Backup->FPSR;
@@ -171,12 +136,8 @@ static inline void RestoreContext(void* ucontext, T *Backup) {
ArchHelpers::Context::SetPc(ucontext, Backup->PrevPC);
ArchHelpers::Context::SetSp(ucontext, Backup->PrevSP);
memcpy(&_mcontext->regs[0], &Backup->GPRs[0], 31 * sizeof(uint64_t));
// Restore the signal mask now
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
@@ -209,22 +170,17 @@ static inline void SetState(void* ucontext, uint64_t val) {
}
static inline uint64_t GetArmReg(void* ucontext, uint32_t id) {
ERROR_AND_DIE_FMT("Not impelented for x86 host");
ERROR_AND_DIE("Not impelented for x86 host");
}
static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
ERROR_AND_DIE_FMT("Not impelented for x86 host");
}
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
ERROR_AND_DIE_FMT("Not implemented for x86 host");
ERROR_AND_DIE("Not impelented for x86 host");
}
using ContextBackup = X86ContextBackup;
template <typename T>
static inline void BackupContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, X86ContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
// Copy the GPRs
@@ -232,34 +188,25 @@ static inline void BackupContext(void* ucontext, T *Backup) {
// Copy the FPRState
memcpy(&Backup->FPRState, _mcontext->fpregs, sizeof(X86ContextBackup::FPRState));
// XXX: Save 256bit and 512bit AVX register state
// Save the signal mask so we can restore it
memcpy(&Backup->sa_mask, &_ucontext->uc_sigmask, sizeof(uint64_t));
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
template <typename T>
static inline void RestoreContext(void* ucontext, T *Backup) {
if constexpr (std::is_same<T, X86ContextBackup>::value) {
auto _ucontext = GetUContext(ucontext);
auto _mcontext = GetMContext(ucontext);
// Copy the GPRs
memcpy(&_mcontext->gregs[0], &Backup->GPRs[0], sizeof(X86ContextBackup::GPRs));
// Copy the FPRState
memcpy(_mcontext->fpregs, &Backup->FPRState, sizeof(X86ContextBackup::FPRState));
// Restore the signal mask now
memcpy(&_ucontext->uc_sigmask, &Backup->sa_mask, sizeof(uint64_t));
} else {
// This must be a runtime error
ERROR_AND_DIE_FMT("Wrong context type");
ERROR_AND_DIE("Wrong context type"); // This must be a runtime error
}
}
#endif
} // namespace FEXCore::ArchHelpers::Context
} // namespace FEXCore::ArchHelpers::Context
@@ -2,7 +2,6 @@
#include <FEXCore/Utils/LogManager.h>
#include <cstring>
#include <fstream>
#include <utility>
namespace FEXCore {
void BlockSamplingData::DumpBlockData() {
@@ -27,7 +26,7 @@ namespace FEXCore {
<< std::endl;
}
Output.close();
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
LogMan::Msg::D("Dumped %d blocks of sampling data", SamplingMap.size());
}
BlockSamplingData::BlockData *BlockSamplingData::GetBlockData(uint64_t RIP) {
+1 -2
View File
@@ -1,7 +1,6 @@
#pragma once
#include <cstdint>
#include <unordered_map>
#include <stdint.h>
namespace FEXCore {
class BlockSamplingData {
+162 -565
View File
@@ -5,14 +5,8 @@ desc: Handles presented capability bits for guest cpu
$end_info$
*/
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CPUID.h>
#include "Common/StringConv.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/HostFeatures.h"
#include "Utils/FileLoading.h"
#include "git_version.h"
#include <cstring>
@@ -21,26 +15,7 @@ $end_info$
#endif
namespace FEXCore {
constexpr uint32_t SUPPORTS_AVX = 0;
// #define CPUID_AMD
#ifdef CPUID_AMD
constexpr uint32_t FAMILY_IDENTIFIER =
0 | // Stepping
(0xA << 4) | // Model
(0xF << 8) | // Family ID
(0 << 12) | // Processor type
(0 << 16) | // Extended model ID
(1 << 20); // Extended family ID
#else
constexpr uint32_t FAMILY_IDENTIFIER =
0 | // Stepping
(0x7 << 4) | // Model
(0x6 << 8) | // Family ID
(0 << 12) | // Processor type
(1 << 16) | // Extended model ID
(0x0 << 20); // Extended family ID
#endif
//#define CPUID_AMD
#ifdef _M_ARM_64
static uint32_t GetCycleCounterFrequency() {
uint64_t Result{};
@@ -48,63 +23,6 @@ static uint32_t GetCycleCounterFrequency() {
: [Res] "=r" (Result));
return Result;
}
static bool GetHostHybridFlag() {
int MaxCPUs = 64;
size_t AllocSize = CPU_ALLOC_SIZE(MaxCPUs);
cpu_set_t *Set = CPU_ALLOC(MaxCPUs);
CPU_ZERO_S(AllocSize, Set);
int Result{};
for (;;) {
Result = sched_getaffinity(0, AllocSize, Set);
if (Result == 0 ||
(Result == -1 && errno != EINVAL)) {
break;
}
MaxCPUs <<= 1;
CPU_FREE(Set);
Set = CPU_ALLOC(MaxCPUs);
AllocSize = CPU_ALLOC_SIZE(MaxCPUs);
CPU_ZERO_S(AllocSize, Set);
}
if (Result != 0) {
return false;
}
int CPUs = CPU_COUNT_S(AllocSize, Set);
bool Hybrid = false;
uint64_t MIDR{};
for (int i = 0; i < CPUs; ++i) {
if (CPU_ISSET_S(i, AllocSize, Set)) {
std::error_code ec{};
std::string MIDRPath = "/sys/devices/system/cpu/cpu" + std::to_string(i) + "/regs/identification/midr_el1";
if (std::filesystem::exists(MIDRPath, ec)) {
std::vector<char> Data{};
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
// Only read 18 bytes for a 64bit value prefixed with 0x
if (FEXCore::FileLoading::LoadFile(Data, MIDRPath, 18)) {
uint64_t NewMIDR{};
if (FEXCore::StrConv::Conv(&Data.at(0), &NewMIDR)) {
if (MIDR != 0 && MIDR != NewMIDR) {
// CPU mismatch, claim hybrid
Hybrid = true;
break;
}
MIDR = NewMIDR;
}
}
}
}
}
CPU_FREE(Set);
return Hybrid;
}
#else
static uint32_t GetCycleCounterFrequency() {
uint32_t eax, ebx, ecx, edx;
@@ -118,22 +36,9 @@ static uint32_t GetCycleCounterFrequency() {
}
return 0;
}
static bool GetHostHybridFlag() {
uint32_t eax, ebx, ecx, edx;
__cpuid(0, eax, ebx, ecx, edx);
if (eax >= 0x7) {
__cpuid(0x7, eax, ebx, ecx, edx);
// Bit 15 of edx claims hybrid CPU
return (edx & (1U << 15)) != 0;
}
return false;
}
#endif
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h() {
FEXCore::CPUID::FunctionResults Res{};
// EBX, EDX, ECX become the manufacturer id string
@@ -152,27 +57,29 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
}
// Processor Info and Features bits
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h() {
FEXCore::CPUID::FunctionResults Res{};
uint32_t CoreCount = Cores();
Res.eax = FAMILY_IDENTIFIER;
Res.eax = 0 | // Stepping
(0 << 4) | // Model
(0xF << 8) | // Family ID
(0 << 12) | // Processor type
(0 << 16) | // Extended model ID
(0 << 20); // Extended family ID
Res.ebx = 0 | // Brand index
(8 << 8) | // Cache line size in bytes
(CoreCount << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(8 << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(0 << 24); // Local APIC ID
Res.ecx =
(1 << 0) | // SSE3
(0 << 1) | // PCLMULQDQ
(1 << 2) | // DS area supports 64bit layout
(1 << 3) | // MWait
(0 << 4) | // DS-CPL
(1 << 4) | // DS-CPL
(0 << 5) | // VMX
(0 << 6) | // SMX
(0 << 7) | // Intel SpeedStep
(1 << 8) | // Thermal Monitor 2
(0 << 8) | // Thermal Monitor 2
(1 << 9) | // SSSE3
(0 << 10) | // L1 context ID
(0 << 11) | // Silicon debug
@@ -182,8 +89,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
(0 << 15) | // Perfmon and debug capability
(0 << 16) | // Reserved
(0 << 17) | // Process-context identifiers
(0 << 18) | // Prefetching from memory mapped device
(1 << 19) | // SSE4.1
(1 << 18) | // Prefetching from memory mapped device
(0 << 19) | // SSE4.1
(0 << 20) | // SSE4.2
(0 << 21) | // X2APIC
(1 << 22) | // MOVBE
@@ -192,7 +99,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
(CTX->HostFeatures.SupportsAES << 25) | // AES
(0 << 26) | // XSAVE
(0 << 27) | // OSXSAVE
(SUPPORTS_AVX << 28) | // AVX
(0 << 28) | // AVX
(0 << 29) | // F16C
(0 << 30) | // RDRAND
(0 << 31); // Hypervisor always returns zero
@@ -217,7 +124,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
(1 << 16) | // Page Attribute Table
(1 << 17) | // 36bit page size extension
(0 << 18) | // Processor serial number
(1 << 19) | // CLFLUSH
(0 << 19) | // CLFLUSH
(0 << 20) | // Reserved
(0 << 21) | // Debug store
(0 << 22) | // Thermal monitor and software controled clock
@@ -225,16 +132,16 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
(1 << 24) | // FXSAVE/FXRSTOR
(1 << 25) | // SSE
(1 << 26) | // SSE2
(0 << 27) | // Self Snoop
(1 << 27) | // Self Snoop
(1 << 28) | // Max APIC IDs reserved field is valid
(1 << 29) | // Thermal monitor
(0 << 30) | // Reserved
(0 << 31); // Pending break enable
(1 << 31); // Pending break enable
return Res;
}
// 2: Cache and TLB information
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h() {
FEXCore::CPUID::FunctionResults Res{};
// returns default values from i7 model 1Ah
@@ -258,286 +165,124 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h(uint32_t Leaf) {
return Res;
}
// 4: Deterministic cache parameters for each level
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults Res{};
constexpr uint32_t CacheType_Data = 1;
constexpr uint32_t CacheType_Instruction = 2;
constexpr uint32_t CacheType_Unified = 3;
if (Leaf == 0) {
// Report L1D
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Data | // Cache type
(0b001 << 5) | // Cache level
(1 << 8) | // Self initializing cache level
(0 << 9) | // Fully associative
(0 << 14) | // Maximum number of addressable IDs for logical processors sharing this cache (With SMT this would be 1)
(CoreCount << 26); // Maximum number of addressable IDs for processor cores in the physical package
Res.ebx =
(63 << 0) | // Line Size - 1 : Claiming 64 byte
(0 << 12) | // Physical Line partitions
(7 << 22); // Associativity - 1 : Claiming 8 way
// 32KB
Res.ecx = 63; // Number of sets - 1 : Claiming 64 sets
Res.edx =
(0 << 0) | // Write-back invalidate
(0 << 1) | // Cache inclusiveness - Includes lower caches
(0 << 2); // Complex cache indexing - 0: Direct, 1: Complex
}
else if (Leaf == 1) {
// Report L1I
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Instruction | // Cache type
(0b001 << 5) | // Cache level
(1 << 8) | // Self initializing cache level
(0 << 9) | // Fully associative
(0 << 14) | // Maximum number of addressable IDs for logical processors sharing this cache (With SMT this would be 1)
(CoreCount << 26); // Maximum number of addressable IDs for processor cores in the physical package
Res.ebx =
(63 << 0) | // Line Size - 1 : Claiming 64 byte
(0 << 12) | // Physical Line partitions
(7 << 22); // Associativity - 1 : Claiming 8 way
// 32KB
Res.ecx = 63; // Number of sets - 1 : Claiming 64 sets
Res.edx =
(0 << 0) | // Write-back invalidate
(0 << 1) | // Cache inclusiveness - Includes lower caches
(0 << 2); // Complex cache indexing - 0: Direct, 1: Complex
}
else if (Leaf == 2) {
// Report L2
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Unified | // Cache type
(0b010 << 5) | // Cache level
(1 << 8) | // Self initializing cache level
(0 << 9) | // Fully associative
(0 << 14) | // Maximum number of addressable IDs for logical processors sharing this cache
(CoreCount << 26); // Maximum number of addressable IDs for processor cores in the physical package
Res.ebx =
(63 << 0) | // Line Size - 1 : Claiming 64 byte
(0 << 12) | // Physical Line partitions
(7 << 22); // Associativity - 1 : Claiming 8 way
// 512KB
Res.ecx = 0x3FF; // Number of sets - 1 : Claiming 1024 sets
Res.edx =
(0 << 0) | // Write-back invalidate
(0 << 1) | // Cache inclusiveness - Includes lower caches
(0 << 2); // Complex cache indexing - 0: Direct, 1: Complex
}
else if (Leaf == 3) {
// Report L3
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Unified | // Cache type
(0b011 << 5) | // Cache level
(1 << 8) | // Self initializing cache level
(0 << 9) | // Fully associative
(CoreCount << 14) | // Maximum number of addressable IDs for logical processors sharing this cache
(CoreCount << 26); // Maximum number of addressable IDs for processor cores in the physical package
Res.ebx =
(63 << 0) | // Line Size - 1 : Claiming 64 byte
(0 << 12) | // Physical Line partitions
(7 << 22); // Associativity - 1 : Claiming 8 way
// 8MB
Res.ecx = 0x4000; // Number of sets - 1 : Claiming 16384 sets
Res.edx =
(0 << 0) | // Write-back invalidate
(0 << 1) | // Cache inclusiveness - Includes lower caches
(1 << 2); // Complex cache indexing - 0: Direct, 1: Complex
}
return Res;
}
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h() {
FEXCore::CPUID::FunctionResults Res{};
Res.eax = (1 << 2); // Always running APIC
Res.ecx = (0 << 3); // Intel performance energy bias preference (EPB)
return Res;
}
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h() {
FEXCore::CPUID::FunctionResults Res{};
if (Leaf == 0) {
// Number of subfunctions
Res.eax = 0x0;
Res.ebx =
(1 << 0) | // FS/GS support
(0 << 1) | // TSC adjust MSR
(0 << 2) | // SGX
(1 << 3) | // BMI1
(0 << 4) | // Intel Hardware Lock Elison
(0 << 5) | // AVX2 support
(1 << 6) | // FPU data pointer updated only on exception
(1 << 7) | // SMEP support
(0 << 8) | // BMI2
(0 << 9) | // Enhanced REP MOVSB/STOSB
(1 << 10) | // INVPCID for system software control of process-context
(0 << 11) | // Restricted transactional memory
(0 << 12) | // Intel resource directory technology Monitoring
(1 << 13) | // Deprecates FPU CS and DS
(0 << 14) | // Intel MPX
(0 << 15) | // Intel Resource Directory Technology Allocation
(0 << 16) | // Reserved
(0 << 17) | // Reserved
(0 << 18) | // RDSEED
(1 << 19) | // ADCX and ADOX instructions
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
(0 << 21) | // Reserved
(0 << 22) | // Reserved
(0 << 23) | // CLFLUSHOPT instruction
(0 << 24) | // CLWB instruction
(0 << 25) | // Intel processor trace
(0 << 26) | // Reserved
(0 << 27) | // Reserved
(0 << 28) | // Reserved
(0 << 29) | // SHA instructions
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.ecx =
(1 << 0) | // PREFETCHWT1
(0 << 1) | // AVX512VBMI
(0 << 2) | // Usermode instruction prevention
(0 << 3) | // Protection keys for user mode pages
(0 << 4) | // OS protection keys
(0 << 5) | // waitpkg
(0 << 6) | // AVX512_VBMI2
(0 << 7) | // CET shadow stack
(0 << 8) | // GFNI
(0 << 9) | // VAES
(0 << 10) | // VPCLMULQDQ
(0 << 11) | // AVX512_VNNI
(0 << 12) | // AVX512_BITALG
(0 << 13) | // Intel Total Memory Encryption
(0 << 14) | // AVX512_VPOPCNTDQ
(0 << 15) | // Reserved
(0 << 16) | // 5 Level page tables
(0 << 17) | // MPX MAWAU
(0 << 18) | // MPX MAWAU
(0 << 19) | // MPX MAWAU
(0 << 20) | // MPX MAWAU
(0 << 21) | // MPX MAWAU
(0 << 22) | // RDPID Read Processor ID
(0 << 23) | // Reserved
(0 << 24) | // Reserved
(0 << 25) | // CLDEMOTE
(0 << 26) | // Reserved
(0 << 27) | // MOVDIRI
(0 << 28) | // MOVDIR64B
(0 << 29) | // Reserved
(0 << 30) | // SGX Launch configuration
(0 << 31); // Reserved
// Number of subfunctions
Res.eax = 0x0;
Res.ebx =
(1 << 0) | // FS/GS support
(0 << 1) | // TSC adjust MSR
(0 << 2) | // SGX
(0 << 3) | // BMI1
(0 << 4) | // Intel Hardware Lock Elison
(0 << 5) | // AVX2 support
(1 << 6) | // FPU data pointer updated only on exception
(1 << 7) | // SMEP support
(0 << 8) | // BMI2
(0 << 9) | // Enhanced REP MOVSB/STOSB
(1 << 10) | // INVPCID for system software control of process-context
(0 << 11) | // Restricted transactional memory
(0 << 12) | // Intel resource directory technology Monitoring
(1 << 13) | // Deprecates FPU CS and DS
(0 << 14) | // Intel MPX
(0 << 15) | // Intel Resource Directory Technology Allocation
(0 << 16) | // Reserved
(0 << 17) | // Reserved
(0 << 18) | // RDSEED
(0 << 19) | // ADCX and ADOX instructions
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
(0 << 21) | // Reserved
(0 << 22) | // Reserved
(0 << 23) | // CLFLUSHOPT instruction
(0 << 24) | // CLWB instruction
(0 << 25) | // Intel processor trace
(0 << 26) | // Reserved
(0 << 27) | // Reserved
(0 << 28) | // Reserved
(0 << 29) | // SHA instructions
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.edx =
(0 << 0) | // Reserved
(0 << 1) | // Reserved
(0 << 2) | // AVX512_4VNNIW
(0 << 3) | // AVX512_4FMAPS
(0 << 4) | // Fast Short Rep Mov
(0 << 5) | // Reserved
(0 << 6) | // Reserved
(0 << 7) | // Reserved
(0 << 8) | // AVX512_VP2INTERSECT
(0 << 9) | // SRBDS_CTRL (Special Register Buffer Data Sampling Mitigations)
(0 << 10) | // VERW clears CPU buffers
(0 << 11) | // Reserved
(0 << 12) | // Reserved
(0 << 13) | // TSX Force Abort (TSX will force abort if attempted)
(0 << 14) | // SERIALIZE instruction
((Hybrid ? 1U : 0U) << 15) | // Hybrid
(0 << 16) | // TSXLDTRK (TSX Suspend load address tracking) - Allows untracked memory loads inside TSX region
(0 << 17) | // Reserved
(0 << 18) | // Intel PCONFIG
(0 << 19) | // Intel Architectural LBR
(0 << 20) | // Intel CET
(0 << 21) | // Reserved
(0 << 22) | // AMX-BF16 - Tile computation on bfloat16
(0 << 23) | // AVX512_FP16 - FP16 AVX512 instructions
(0 << 24) | // AMX-tile - If AMX is implemented
(0 << 25) | // AMX-int8 - AMX on 8-bit integers
(0 << 26) | // IBRS_IBPB - Speculation control
(0 << 27) | // STIBP - Single Thread Indirect Branch Predictor, Part of IBC
(0 << 28) | // L1D Flush
(0 << 29) | // Arch capabilities - Speculative side channel mitigations
(0 << 30) | // Arch capabilities - MSR module specific
(0 << 31); // SSBD - Speculative Store Bypass Disable
}
Res.ecx =
(1 << 0) | // PREFETCHWT1
(0 << 1) | // AVX512VBMI
(0 << 2) | // Usermode instruction prevention
(0 << 3) | // Protection keys for user mode pages
(1 << 4) | // OS protection keys
(0 << 5) | // waitpkg
(0 << 6) | // AVX512_VBMI2
(0 << 7) | // CET shadow stack
(0 << 8) | // GFNI
(0 << 9) | // VAES
(0 << 10) | // VPCLMULQDQ
(0 << 11) | // AVX512_VNNI
(0 << 12) | // AVX512_BITALG
(0 << 13) | // Intel Total Memory Encryption
(0 << 14) | // AVX512_VPOPCNTDQ
(0 << 15) | // Reserved
(0 << 16) | // 5 Level page tables
(0 << 17) | // MPX MAWAU
(0 << 18) | // MPX MAWAU
(0 << 19) | // MPX MAWAU
(0 << 20) | // MPX MAWAU
(0 << 21) | // MPX MAWAU
(0 << 22) | // RDPID Read Processor ID
(0 << 23) | // Reserved
(0 << 24) | // Reserved
(0 << 25) | // CLDEMOTE
(0 << 26) | // Reserved
(0 << 27) | // MOVDIRI
(0 << 28) | // MOVDIR64B
(0 << 29) | // Reserved
(0 << 30) | // SGX Launch configuration
(0 << 31); // Reserved
Res.edx =
(0 << 0) | // Reserved
(0 << 1) | // Reserved
(0 << 2) | // AVX512_4VNNIW
(0 << 3) | // AVX512_4FMAPS
(0 << 4) | // Fast Short Rep Mov
(0 << 5) | // Reserved
(0 << 6) | // Reserved
(0 << 7) | // Reserved
(0 << 8) | // AVX512_VP2INTERSECT
(0 << 9) | // Reserved
(0 << 10) | // VERW clears CPU buffers
(0 << 11) | // Reserved
(0 << 12) | // Reserved
(0 << 13) | // Reserved
(0 << 14) | // SERIALIZE instruction
(0 << 15) | // Reserved
(0 << 16) | // Reserved
(0 << 17) | // Reserved
(0 << 18) | // Intel PCONFIG
(0 << 19) | // Intel Architectural LBR
(0 << 20) | // Intel CET
(0 << 21) | // Reserved
(0 << 22) | // Reserved
(0 << 23) | // Reserved
(0 << 24) | // Reserved
(0 << 25) | // Reserved
(0 << 26) | // Reserved
(0 << 27) | // Reserved
(0 << 28) | // L1D Flush
(0 << 29) | // Arch capabilities
(0 << 30) | // Reserved
(0 << 31); // Reserved
return Res;
}
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
// Leaf 0
FEXCore::CPUID::FunctionResults Res{};
uint32_t XFeatureSupportedSizeMax = SUPPORTS_AVX ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
if (Leaf == 0) {
// XFeatureSupportedMask[31:0]
Res.eax =
(1 << 0) | // X87 support
(1 << 1) | // 128-bit SSE support
(SUPPORTS_AVX << 2) | // 256-bit AVX support
(0b00 << 3) | // MPX State
(0b000 << 5) | // AVX-512 state
(0 << 8) | // "Used for IA32_XSS" ... Used for what?
(0 << 9); // PKRU state
// EBX and ECX doesn't need to match if a feature is supported but not enabled
Res.ebx = XFeatureSupportedSizeMax;
Res.ecx = XFeatureSupportedSizeMax; // XFeatureSupportedSizeMax: Size in bytes of XSAVE/XRSTOR area
// XFeatureSupportedMask[63:32]
Res.edx = 0; // Upper 32-bits of XFeatureSupportedMask
}
else if (Leaf == 1) {
Res.eax =
(0 << 0) | // XSAVEOPT
(0 << 1) | // XSAVEC (and XRSTOR)
(0 << 2) | // XGETBV - XGETBV with ECX=1 supported
(0 << 3); // XSAVES - XSAVES, XRSTORS, and IA32_XSS supported
// Same information as Leaf 0 for ebx
Res.ebx = XFeatureSupportedSizeMax;
// Lower supported 32bits of IA32_XSS MSR. IA32_XSS[n] can only be set to 1 if ECX[n] is 1
Res.ecx =
(0b0000'0000 << 0) | // Used for XCR0
(0 << 8) | // PT state
(0 << 9); // Used for XCR0
// Upper supported 32bits of IA32_XSS MSR. IA32_XSS[n+32] can only be set to 1 if EDX[n] is 1
// Entirely reserved atm
Res.edx = 0;
}
else if (Leaf == 2) {
Res.eax = SUPPORTS_AVX ? 0x0000'0100 : 0; // YmmSaveStateSize
Res.ebx = SUPPORTS_AVX ? 0x0000'0240 : 0; // YmmSaveStateOffset
// Reserved
Res.ecx = 0;
Res.edx = 0;
}
return Res;
}
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h() {
FEXCore::CPUID::FunctionResults Res{};
// TSC frequency = ECX * EBX / EAX
uint32_t FrequencyHz = GetCycleCounterFrequency();
@@ -550,7 +295,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) {
}
// Highest extended function implemented
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h() {
FEXCore::CPUID::FunctionResults Res{};
Res.eax = 0x8000001F;
@@ -569,10 +314,15 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
}
// Extended processor and feature bits
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h() {
FEXCore::CPUID::FunctionResults Res{};
Res.eax = FAMILY_IDENTIFIER;
Res.eax = 0 | // Stepping
(0 << 4) | // Model
(0 << 8) | // Family ID
(0 << 12) | // Processor type
(0 << 16) | // Extended model ID
(0 << 20); // Extended family ID
Res.ecx =
(1 << 0) | // LAHF/SAHF
@@ -597,13 +347,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
(0 << 19) | // Reserved
(0 << 20) | // Reserved
(0 << 21) | // Reserved
(0 << 22) | // Topology extensions support
(0 << 23) | // Core performance counter extensions
(0 << 24) | // NB performance counter extensions
(1 << 22) | // Topology extensions support
(1 << 23) | // Core performance counter extensions
(1 << 24) | // NB performance counter extensions
(0 << 25) | // Reserved
(0 << 26) | // Data breakpoints extensions
(0 << 27) | // Performance TSC
(0 << 28) | // L2 perf counter extensions
(1 << 27) | // Performance TSC
(1 << 28) | // L2 perf counter extensions
(0 << 29) | // Reserved
(0 << 30) | // Reserved
(0 << 31); // Reserved
@@ -635,7 +385,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
(1 << 23) | // MMX
(1 << 24) | // FXSAVE/FXRSTOR
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
(0 << 26) | // 1 gigabit pages
(1 << 26) | // 1 gigabit pages
(0 << 27) | // RDTSCP
(0 << 28) | // Reserved
(1 << 29) | // Long Mode
@@ -650,26 +400,26 @@ constexpr char ProcessorBrand[48] = {
};
//Processor brand string
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h() {
FEXCore::CPUID::FunctionResults Res{};
memcpy(&Res, &ProcessorBrand[0], sizeof(FEXCore::CPUID::FunctionResults));
return Res;
}
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h() {
FEXCore::CPUID::FunctionResults Res{};
memcpy(&Res, &ProcessorBrand[16], sizeof(FEXCore::CPUID::FunctionResults));
return Res;
}
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h() {
FEXCore::CPUID::FunctionResults Res{};
memcpy(&Res, &ProcessorBrand[32], sizeof(FEXCore::CPUID::FunctionResults));
return Res;
}
// L1 Cache and TLB identifiers
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0005h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0005h() {
FEXCore::CPUID::FunctionResults Res{};
// L1 TLB Information for 2MB and 4MB pages
@@ -704,7 +454,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0005h(uint32_t Leaf) {
}
// L2 Cache identifiers
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0006h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0006h() {
FEXCore::CPUID::FunctionResults Res{};
// L2 TLB Information for 2MB and 4MB pages
@@ -738,7 +488,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0006h(uint32_t Leaf) {
}
// Advanced power management
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0007h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0007h() {
FEXCore::CPUID::FunctionResults Res{};
Res.eax = (1 << 2); // APIC timer not affected by p-state
Res.edx =
@@ -746,167 +496,27 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0007h(uint32_t Leaf) {
return Res;
}
// Virtual and physical address sizes
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults Res{};
Res.eax =
(48 << 0) | // PhysAddrSize = 48-bit
(48 << 8) | // LinAddrSize = 48-bit
(0 << 16); // GuestPhysAddrSize == PhysAddrSize
Res.ebx =
(0 << 2) | // XSaveErPtr: Saving and restoring error pointers
(0 << 1) | // IRPerf: Instructions retired count support
(0 << 0); // CLZERO support
uint32_t CoreCount = Cores() - 1;
Res.ecx =
(0 << 16) | // PerfTscSize: Performance timestamp count size
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
(CoreCount << 0); // Count count subtract one
return Res;
}
// TLB 1GB page identifiers
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0019h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults Res{};
Res.eax =
(0xF << 28) | // L1 DTLB associativity for 1GB pages
(64 << 16) | // L1 DTLB entry count for 1GB pages
(0xF << 12) | // L1 ITLB associativity for 1GB pages
(64 << 0); // L1 ITLB entry count for 1GB pages
Res.ebx =
(0 << 28) | // L2 DTLB associativity for 1GB pages
(0 << 16) | // L2 DTLB entry count for 1GB pages
(0 << 12) | // L2 ITLB associativity for 1GB pages
(0 << 0); // L2 ITLB entry count for 1GB pages
return Res;
}
// Deterministic cache parameters for each level
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_001Dh(uint32_t Leaf) {
// This is nearly a copy of CPUID function 4h
// There are some minor changes though
FEXCore::CPUID::FunctionResults Res{};
constexpr uint32_t CacheType_Data = 1;
constexpr uint32_t CacheType_Instruction = 2;
constexpr uint32_t CacheType_Unified = 3;
if (Leaf == 0) {
// Report L1D
Res.eax = CacheType_Data | // Cache type
(0b001 << 5) | // Cache level
(1 << 8) | // Self initializing cache level
(0 << 9) | // Fully associative
(0 << 14); // Maximum number of addressable IDs for logical processors sharing this cache (With SMT this would be 1)
Res.ebx =
(63 << 0) | // Line Size - 1 : Claiming 64 byte
(0 << 12) | // Physical Line partitions
(7 << 22); // Associativity - 1 : Claiming 8 way
// 32KB
Res.ecx = 63; // Number of sets - 1 : Claiming 64 sets
Res.edx =
(0 << 0) | // Write-back invalidate
(0 << 1); // Cache inclusiveness - Includes lower caches
}
else if (Leaf == 1) {
// Report L1I
Res.eax = CacheType_Instruction | // Cache type
(0b001 << 5) | // Cache level
(1 << 8) | // Self initializing cache level
(0 << 9) | // Fully associative
(0 << 14); // Maximum number of addressable IDs for logical processors sharing this cache (With SMT this would be 1)
Res.ebx =
(63 << 0) | // Line Size - 1 : Claiming 64 byte
(0 << 12) | // Physical Line partitions
(7 << 22); // Associativity - 1 : Claiming 8 way
// 32KB
Res.ecx = 63; // Number of sets - 1 : Claiming 64 sets
Res.edx =
(0 << 0) | // Write-back invalidate
(0 << 1); // Cache inclusiveness - Includes lower caches
}
else if (Leaf == 2) {
// Report L2
Res.eax = CacheType_Unified | // Cache type
(0b010 << 5) | // Cache level
(1 << 8) | // Self initializing cache level
(0 << 9) | // Fully associative
(0 << 14); // Maximum number of addressable IDs for logical processors sharing this cache
Res.ebx =
(63 << 0) | // Line Size - 1 : Claiming 64 byte
(0 << 12) | // Physical Line partitions
(7 << 22); // Associativity - 1 : Claiming 8 way
// 512KB
Res.ecx = 0x3FF; // Number of sets - 1 : Claiming 1024 sets
Res.edx =
(0 << 0) | // Write-back invalidate
(0 << 1); // Cache inclusiveness - Includes lower caches
}
else if (Leaf == 3) {
// Report L3
uint32_t CoreCount = Cores() - 1;
Res.eax = CacheType_Unified | // Cache type
(0b011 << 5) | // Cache level
(1 << 8) | // Self initializing cache level
(0 << 9) | // Fully associative
(CoreCount << 14); // Maximum number of addressable IDs for logical processors sharing this cache
Res.ebx =
(63 << 0) | // Line Size - 1 : Claiming 64 byte
(0 << 12) | // Physical Line partitions
(7 << 22); // Associativity - 1 : Claiming 8 way
// 8MB
Res.ecx = 0x4000; // Number of sets - 1 : Claiming 16384 sets
Res.edx =
(0 << 0) | // Write-back invalidate
(0 << 1); // Cache inclusiveness - Includes lower caches
}
return Res;
}
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved() {
FEXCore::CPUID::FunctionResults Res{};
return Res;
}
void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
CTX = ctx;
RegisterFunction(0, &CPUIDEmu::Function_0h);
RegisterFunction(1, &CPUIDEmu::Function_01h);
RegisterFunction(2, &CPUIDEmu::Function_02h);
RegisterFunction(0, std::bind(&CPUIDEmu::Function_0h, this));
RegisterFunction(1, std::bind(&CPUIDEmu::Function_01h, this));
RegisterFunction(2, std::bind(&CPUIDEmu::Function_02h, this));
// 3: Serial Number(previously), now reserved
#ifndef CPUID_AMD
// Deterministic cache parameters for each level
RegisterFunction(0x4, &CPUIDEmu::Function_04h);
#endif
// 4: Deterministic cache parameters for each level
// 5: Monitor/mwait
// Thermal and power management
RegisterFunction(6, &CPUIDEmu::Function_06h);
RegisterFunction(6, std::bind(&CPUIDEmu::Function_06h, this));
// Extended feature flags
RegisterFunction(7, &CPUIDEmu::Function_07h);
RegisterFunction(7, std::bind(&CPUIDEmu::Function_07h, this));
// 9: Direct Cache Access information
// 0x0A: Architectural performance monitoring
// 0x0B: Extended topology enumeration
// 0x0D: Processor extended state enumeration
RegisterFunction(0x0D, &CPUIDEmu::Function_0Dh);
// 0x0F: Intel RDT monitoring
// 0x10: Intel RDT allocation enumeration
// 0x12: Intel SGX capability enumeration
@@ -915,52 +525,39 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
#ifndef CPUID_AMD
// Timestamp counter information
// Doesn't exist on AMD hardware
RegisterFunction(0x15, &CPUIDEmu::Function_15h);
RegisterFunction(0x15, std::bind(&CPUIDEmu::Function_15h, this));
#endif
// 0x16: Processor frequency information
// 0x17: SoC vendor attribute enumeration
// Largest extended function number
RegisterFunction(0x8000'0000, &CPUIDEmu::Function_8000_0000h);
RegisterFunction(0x8000'0000, std::bind(&CPUIDEmu::Function_8000_0000h, this));
// Processor vendor
RegisterFunction(0x8000'0001, &CPUIDEmu::Function_8000_0001h);
RegisterFunction(0x8000'0001, std::bind(&CPUIDEmu::Function_8000_0001h, this));
// Processor brand string
RegisterFunction(0x8000'0002, &CPUIDEmu::Function_8000_0002h);
RegisterFunction(0x8000'0002, std::bind(&CPUIDEmu::Function_8000_0002h, this));
// Processor brand string continued
RegisterFunction(0x8000'0003, &CPUIDEmu::Function_8000_0003h);
RegisterFunction(0x8000'0003, std::bind(&CPUIDEmu::Function_8000_0003h, this));
// Processor brand string continued
RegisterFunction(0x8000'0004, &CPUIDEmu::Function_8000_0004h);
RegisterFunction(0x8000'0004, std::bind(&CPUIDEmu::Function_8000_0004h, this));
// 0x8000'0005: L1 Cache and TLB identifiers
#ifdef CPUID_AMD
RegisterFunction(0x8000'0005, &CPUIDEmu::Function_8000_0005h);
#else
// This is full reserved on Intel platforms
RegisterFunction(0x8000'0005, &CPUIDEmu::Function_Reserved);
RegisterFunction(0x8000'0005, std::bind(&CPUIDEmu::Function_8000_0005h, this));
#endif
// 0x8000'0006: L2 Cache identifiers
RegisterFunction(0x8000'0006, &CPUIDEmu::Function_8000_0006h);
RegisterFunction(0x8000'0006, std::bind(&CPUIDEmu::Function_8000_0006h, this));
// Advanced power management information
RegisterFunction(0x8000'0007, &CPUIDEmu::Function_8000_0007h);
// Virtual and physical address sizes
RegisterFunction(0x8000'0008, &CPUIDEmu::Function_8000_0008h);
RegisterFunction(0x8000'0007, std::bind(&CPUIDEmu::Function_8000_0007h, this));
// 0x8000'0008: Virtual and physical address sizes
// 0x8000'000A: SVM Revision
// TLB 1GB page identifiers
RegisterFunction(0x8000'0019, &CPUIDEmu::Function_8000_0019h);
// 0x8000'0019: TLB 1GB page identifiers
// 0x8000'001A: Performance optimization identifiers
// 0x8000'001B: Instruction based sampling identifiers
// 0x8000'001C: Lightweight profiling capabilities
// 0x8000'001D: Cache properties
#ifdef CPUID_AMD
// Deterministic cache parameters for each level
RegisterFunction(0x8000'001D, &CPUIDEmu::Function_8000_001Dh);
#endif
// 0x8000'001E: Extended APIC ID
// 0x8000'001F: AMD Secure Encryption
// Setup some state tracking
Hybrid = GetHostHybridFlag();
}
}
+27 -33
View File
@@ -1,11 +1,9 @@
#pragma once
#include <cstdint>
#include <functional>
#include <unordered_map>
#include <utility>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
namespace FEXCore {
namespace Context {
@@ -25,48 +23,44 @@ private:
public:
void Init(FEXCore::Context::Context *ctx);
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) {
const auto Handler = FunctionHandlers.find(Function);
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, [[maybe_unused]] uint32_t Leaf) {
auto Handler = FunctionHandlers.find(Function);
if (Handler == FunctionHandlers.end()) {
return Function_Reserved(Leaf);
#ifndef NDEBUG
LogMan::Msg::E("Unhandled CPU ID function, 0x%x", Function);
#endif
return Function_Reserved();
}
return (this->*Handler->second)(Leaf);
return Handler->second();
}
private:
FEXCore::Context::Context *CTX;
bool Hybrid{};
FEX_CONFIG_OPT(Cores, THREADS);
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
using FunctionHandler = std::function<FEXCore::CPUID::FunctionResults()>;
void RegisterFunction(uint32_t Function, FunctionHandler Handler) {
FunctionHandlers.insert_or_assign(Function, Handler);
FunctionHandlers[Function] = Handler;
}
std::unordered_map<uint32_t, FunctionHandler> FunctionHandlers;
// Functions
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_01h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_02h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_04h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_06h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_07h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0005h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0008h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0009h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_0h();
FEXCore::CPUID::FunctionResults Function_01h();
FEXCore::CPUID::FunctionResults Function_02h();
FEXCore::CPUID::FunctionResults Function_06h();
FEXCore::CPUID::FunctionResults Function_07h();
FEXCore::CPUID::FunctionResults Function_15h();
FEXCore::CPUID::FunctionResults Function_8000_0000h();
FEXCore::CPUID::FunctionResults Function_8000_0001h();
FEXCore::CPUID::FunctionResults Function_8000_0002h();
FEXCore::CPUID::FunctionResults Function_8000_0003h();
FEXCore::CPUID::FunctionResults Function_8000_0004h();
FEXCore::CPUID::FunctionResults Function_8000_0005h();
FEXCore::CPUID::FunctionResults Function_8000_0006h();
FEXCore::CPUID::FunctionResults Function_8000_0007h();
FEXCore::CPUID::FunctionResults Function_Reserved();
};
}
+44 -41
View File
@@ -1,21 +1,8 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/CompileService.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "FEXCore/Debug/InternalThreadState.h"
#include "FEXCore/HLE/Linux/ThreadManagement.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Threads.h>
#include <memory>
#include <pthread.h>
#include <stdio.h>
namespace FEXCore {
static void* ThreadHandler(void *Arg) {
@@ -35,9 +22,7 @@ namespace FEXCore {
CTX->InitializeCompiler(CompileThreadData.get(), true);
CompileThreadData->CPUBackend->CopyNecessaryDataForCompileThread(ParentThread->CPUBackend.get());
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
FEXCore::Threads::SetSignalMask(OldMask);
}
void CompileService::Initialize() {
@@ -60,15 +45,23 @@ namespace FEXCore {
// Grab the work queue and clear it
// We don't need to grab the queue mutex since this thread will no longer receive any work events
// Threads are bounded 1:1
while (!WorkQueue.empty()) {
while (WorkQueue.size()) {
WorkItem *Item = WorkQueue.front();
WorkQueue.pop();
delete Item;
}
// Go through the garbage collection array and clear it
// It's safe to clear things that aren't marked safe since we are clearing cache
GCArray.clear();
if (GCArray.size()) {
// Clean up our GC array
for (auto it = GCArray.begin(); it != GCArray.end();) {
delete *it;
it = GCArray.erase(it);
}
}
LOGMAN_THROW_A_FMT(CompileThreadData->LocalIRCache.empty(), "Compile service must never have LocalIRCache");
LogMan::Throw::A(CompileThreadData->LocalIRCache.size() == 0, "Compile service must never have LocalIRCache");
CompileMutex.unlock();
}
@@ -79,26 +72,28 @@ namespace FEXCore {
SelectedThread->CPUBackend->ClearCache();
}
CompileService::WorkItem *CompileService::CompileCode(uint64_t RIP) {
WorkItem* ResultItem = nullptr;
// Tell the worker thread to compile code for us
WorkItem *Item = new WorkItem{};
Item->RIP = RIP;
{
// Tell the worker thread to compile code for us
auto Item = std::make_unique<WorkItem>();
Item->RIP = RIP;
// Fill the threads work queue
std::scoped_lock lk(QueueMutex);
ResultItem = WorkQueue.emplace(std::move(Item)).get();
std::scoped_lock<std::mutex> lk(QueueMutex);
WorkQueue.emplace(Item);
}
// Notify the thread that it has more work
StartWork.NotifyAll();
return ResultItem;
return Item;
}
void CompileService::ExecutionThread() {
// Ignore signals coming from the guest
CTX->SignalDelegation->MaskThreadSignals();
// Set our thread name so we can see its relation
char ThreadName[16]{};
snprintf(ThreadName, 16, "%ld-CS", ParentThread->ThreadManager.TID.load());
@@ -110,18 +105,18 @@ namespace FEXCore {
if (ShuttingDown.load()) {
break;
}
std::scoped_lock<std::mutex> lk(CompileMutex);
std::scoped_lock lk(CompileMutex);
size_t WorkItems{};
do {
// Grab a work item
std::unique_ptr<WorkItem> Item{};
WorkItem *Item{};
{
std::scoped_lock lk(QueueMutex);
std::scoped_lock<std::mutex> lk(QueueMutex);
WorkItems = WorkQueue.size();
if (WorkItems != 0) {
Item = std::move(WorkQueue.front());
if (WorkItems) {
Item = WorkQueue.front();
WorkQueue.pop();
}
}
@@ -129,7 +124,7 @@ namespace FEXCore {
// If we had a work item then work on it
if (Item) {
// Make sure it's not in lookup cache by accident
LOGMAN_THROW_A_FMT(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
LogMan::Throw::A(CompileThreadData->LookupCache->FindBlock(Item->RIP) == 0, "Compile Service must never have entries in the LookupCache");
// Code isn't in cache, compile now
// Set our thread state's RIP
@@ -137,11 +132,11 @@ namespace FEXCore {
auto [CodePtr, IRList, DebugData, RAData, Generated, StartAddr, Length] = CTX->CompileCode(CompileThreadData.get(), Item->RIP);
LOGMAN_THROW_A_FMT(Generated == true, "Compile Service doesn't have IR Cache");
LogMan::Throw::A(Generated == true, "Compile Service doesn't have IR Cache");
if (!CodePtr) {
// XXX: We currently have the expectation that compile service code will be significantly smaller than regular thread's code
ERROR_AND_DIE_FMT("Couldn't compile code for thread at RIP: 0x{:x}", Item->RIP);
ERROR_AND_DIE("Couldn't compile code for thread at RIP: 0x%lx", Item->RIP);
}
Item->CodePtr = CodePtr;
@@ -151,15 +146,23 @@ namespace FEXCore {
Item->StartAddr = StartAddr;
Item->Length = Length;
auto& GCItem = GCArray.emplace_back(std::move(Item));
GCItem->ServiceWorkDone.NotifyAll();
GCArray.emplace_back(Item);
Item->ServiceWorkDone.NotifyAll();
}
} while (WorkItems != 0);
// Clean up any safe entries in our GC array if we have any.
std::erase_if(GCArray, [](const auto& Entry) {
return Entry->SafeToClear.load(std::memory_order_relaxed);
});
if (GCArray.size()) {
// Clean up our GC array
for (auto it = GCArray.begin(); it != GCArray.end();) {
if ((*it)->SafeToClear) {
delete *it;
it = GCArray.erase(it);
}
else {
++it;
}
}
}
}
}
}
+9 -11
View File
@@ -1,22 +1,23 @@
#pragma once
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/Threads.h>
#include <atomic>
#include <memory>
#include <mutex>
#include <thread>
#include <unordered_map>
#include <queue>
#include <stdint.h>
#include <vector>
namespace FEXCore {
namespace Context {
struct Context;
}
namespace Core {
struct InternalThreadState;
}
namespace IR {
class IRListView;
class RegisterAllocationData;
};
class CompileService final {
@@ -48,9 +49,6 @@ class CompileService final {
// Public for threading
void ExecutionThread();
bool IsAddressInJITCode(uint64_t Address) const {
return CompileThreadData->CPUBackend->IsAddressInJITCode(Address, false, false);
}
private:
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *ParentThread;
@@ -60,8 +58,8 @@ class CompileService final {
std::mutex QueueMutex{};
std::mutex CompileMutex{};
std::queue<std::unique_ptr<WorkItem>> WorkQueue{};
std::vector<std::unique_ptr<WorkItem>> GCArray{};
std::queue<WorkItem*> WorkQueue{};
std::vector<WorkItem*> GCArray{};
Event StartWork{};
std::atomic_bool ShuttingDown{false};
};
File diff suppressed because it is too large. Load diff
@@ -1,32 +1,15 @@
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/X86HelperGen.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <array>
#include <bit>
#include <cmath>
#include <cstdint>
#include <memory>
#include <stddef.h>
#include "aarch64/assembler-aarch64.h"
#include "aarch64/constants-aarch64.h"
#include "aarch64/operands-aarch64.h"
#include "aarch64/cpu-aarch64.h"
#include "code-buffer-vixl.h"
#include "platform-vixl.h"
#include <sys/syscall.h>
#include <unistd.h>
#include "aarch64/disasm-aarch64.h"
namespace FEXCore::CPU {
@@ -41,7 +24,8 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
SRAEnabled = config.StaticRegisterAssignment;
SetAllowAssembler(true);
DispatchPtr = GetCursorAddress<CPUBackend::AsmDispatch>();
auto Buffer = GetBuffer();
DispatchPtr = Buffer->GetOffsetAddress<CPUBackend::AsmDispatch>(GetCursorOffset());
// while (true) {
// Ptr = FindBlock(RIP)
@@ -74,7 +58,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
add(x0, sp, 0);
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
AbsoluteLoopTopAddressFillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled) {
FillStaticRegs();
@@ -83,7 +67,6 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
aarch64::Label FullLookup{};
aarch64::Label CallBlock{};
aarch64::Label LoopTop{};
aarch64::Label ExitSpillSRA{};
aarch64::Label ThreadPauseHandler{};
@@ -96,19 +79,16 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
auto RipReg = x2;
// L1 Cache
ldr(x0, &l_L1Ptr);
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x3, Shift::LSL, 4));
ldp(x3, x0, MemOperand(x0));
cmp(x0, RipReg);
b(&FullLookup, Condition::ne);
if (!config.ExecuteBlocksWithCall) {
br(x3);
} else {
b(&CallBlock);
// L1 Cache
ldr(x0, &l_L1Ptr);
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x3, Shift::LSL, 4));
ldp(x1, x0, MemOperand(x0));
cmp(x0, RipReg);
b(&FullLookup, Condition::ne);
br(x1);
}
// L1C check failed, do a full lookup
@@ -119,7 +99,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
ldr(x0, &l_PagePtr);
// Mask the address by the virtual address size so we can check for aliases
if (std::popcount(VirtualMemorySize) == 1) {
if (__builtin_popcountl(VirtualMemorySize) == 1) {
and_(x3, RipReg, Thread->LookupCache->GetVirtualMemorySize() - 1);
}
else {
@@ -156,48 +136,51 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldr(x0, &l_L1Ptr);
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x1, Shift::LSL, 4));
stp(x3, x2, MemOperand(x0));
// Jump to the block
if (!config.ExecuteBlocksWithCall) {
// update L1 cache
ldr(x0, &l_L1Ptr);
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x1, Shift::LSL, 4));
stp(x3, x2, MemOperand(x0));
br(x3);
} else {
bind(&CallBlock);
mov(x0, STATE);
blr(x3);
}
}
if (CTX->GetGdbServerStatus()) {
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
if (config.ExecuteBlocksWithCall) {
// Interpreter continues execution here
if (CTX->GetGdbServerStatus()) {
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
ldr(x0, &l_CTX);
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
// If the value == 0 then branch to the top
cbz(x0, &LoopTop);
// Else we need to pause now
b(&ThreadPauseHandler);
} else {
// Unconditionally loop to the top
// We will only stop on error when compiling a block or signal
b(&LoopTop);
}
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
ldr(x0, &l_CTX);
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
// If the value == 0 then branch to the top
cbz(x0, &LoopTop);
// Else we need to pause now
b(&ThreadPauseHandler);
}
else {
// Unconditionally loop to the top
// We will only stop on error when compiling a block or signal
b(&LoopTop);
}
}
}
{
bind(&ExitSpillSRA);
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
ThreadStopHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled)
SpillStaticRegs();
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
ThreadStopHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
PopCalleeSavedRegisters();
@@ -206,31 +189,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
ret();
}
constexpr bool SignalSafeCompile = true;
{
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
ExitFunctionLinkerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled)
SpillStaticRegs();
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// Args:
// X0: SETMASK
// X1: Pointer to mask value (uint64_t)
// X2: Pointer to old mask value (uint64_t)
// X3: Size of mask, sizeof(uint64_t)
// X8: Syscall
LoadConstant(x0, ~0ULL);
stp(x0, x0, MemOperand(sp, -16, PreIndex));
LoadConstant(x0, SIG_SETMASK);
add(x1, sp, 0);
add(x2, sp, 0);
LoadConstant(x3, 8);
LoadConstant(x8, SYS_rt_sigprocmask);
svc(0);
}
ldr(x0, &l_ExitFunctionLinkThis);
mov(x1, STATE);
mov(x2, lr);
@@ -238,24 +201,6 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
ldr(x3, &l_ExitFunctionLink);
blr(x3);
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
mov(x4, x0);
LoadConstant(x0, SIG_SETMASK);
add(x1, sp, 0);
LoadConstant(x2, 0);
LoadConstant(x3, 8);
LoadConstant(x8, SYS_rt_sigprocmask);
svc(0);
// Bring stack back
add(sp, sp, 16);
mov(x0, x4);
}
if (SRAEnabled)
FillStaticRegs();
br(x0);
@@ -265,53 +210,16 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
{
bind(&NoBlock);
if (SRAEnabled)
SpillStaticRegs();
if (SignalSafeCompile) {
// When compiling code, mask all signals to reduce the chance of reentrant allocations
// Args:
// X0: SETMASK
// X1: Pointer to mask value (uint64_t)
// X2: Pointer to old mask value (uint64_t)
// X3: Size of mask, sizeof(uint64_t)
// X8: Syscall
LoadConstant(x0, ~0ULL);
stp(x0, x2, MemOperand(sp, -16, PreIndex));
LoadConstant(x0, SIG_SETMASK);
add(x1, sp, 0);
add(x2, sp, 0);
LoadConstant(x3, 8);
LoadConstant(x8, SYS_rt_sigprocmask);
svc(0);
// Reload x2 to bring back RIP
ldr(x2, MemOperand(sp, 8, Offset));
}
ldr(x0, &l_CTX);
mov(x1, STATE);
ldr(x3, &l_CompileBlock);
if (SRAEnabled)
SpillStaticRegs();
// X2 contains our guest RIP
blr(x3); // { CTX, Frame, RIP}
if (SignalSafeCompile) {
// Now restore the signal mask
// Living in the same location
LoadConstant(x0, SIG_SETMASK);
add(x1, sp, 0);
LoadConstant(x2, 0);
LoadConstant(x3, 8);
LoadConstant(x8, SYS_rt_sigprocmask);
svc(0);
// Bring stack back
add(sp, sp, 16);
}
if (SRAEnabled)
FillStaticRegs();
@@ -319,7 +227,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
}
{
SignalHandlerReturnAddress = GetCursorAddress<uint64_t>();
SignalHandlerReturnAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
// Now to get back to our old location we need to do a fault dance
// We can't use SIGTRAP here since gdb catches it and never gives it to the application!
@@ -327,23 +235,12 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
}
{
// Guest SIGILL handler
// Needs to be distinct from the SignalHandlerReturnAddress
UnimplementedInstructionAddress = GetCursorAddress<uint64_t>();
if (SRAEnabled)
SpillStaticRegs();
hlt(0);
}
{
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
ThreadPauseHandlerAddressSpillSRA = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (SRAEnabled)
SpillStaticRegs();
bind(&ThreadPauseHandler);
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
ThreadPauseHandlerAddress = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
// We are pausing, this means the frontend should be waiting for this thread to idle
// We will have faulted and jumped to this location at this point
@@ -353,7 +250,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
ldr(x2, &l_Sleep);
blr(x2);
PauseReturnInstruction = GetCursorAddress<uint64_t>();
PauseReturnInstruction = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
// Fault to start running again
hlt(0);
}
@@ -374,7 +271,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
// On return to the thunk, the thunk can get whatever its return value is from the thread context depending on ABI handling on its end
// When the thunk itself returns, it'll do its regular return logic there
// void ReentrantCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
CallbackPtr = GetCursorAddress<CPUBackend::JITCallback>();
CallbackPtr = Buffer->GetOffsetAddress<CPUBackend::JITCallback>(GetCursorOffset());
// We expect the thunk to have previously pushed the registers it was using
PushCalleeSavedRegisters();
@@ -423,32 +320,28 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
FinalizeCode();
Start = reinterpret_cast<uint64_t>(DispatchPtr);
End = GetCursorAddress<uint64_t>();
End = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
GetBuffer()->SetExecutable();
if (CTX->Config.BlockJITNaming()) {
std::string Name = "Dispatch_" + std::to_string(::gettid());
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
}
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
}
#if ENABLE_JITSYMBOLS
std::string Name = "Dispatch_" + std::to_string(::gettid());
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
#endif
}
void Arm64Dispatcher::SpillSRA(void *ucontext, uint32_t IgnoreMask) {
void Arm64Dispatcher::SpillSRA(void *ucontext) {
for(int i = 0; i < SRA64.size(); i++) {
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
// Skip this one, it's already spilled
continue;
}
ThreadState->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
}
// TODO: Also recover FPRs, not sure where the neon context is
// This is usually not needed
/*
for(int i = 0; i < SRAFPR.size(); i++) {
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
memcpy(&ThreadState->CurrentFrame->State.xmm[i][0], &FPR, sizeof(__uint128_t));
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
State->State.State.xmm[i][0] = _mcontext.neon[SRAFPR[i].GetCode()];
}
*/
}
#ifdef _M_ARM_64
@@ -457,7 +350,7 @@ void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore:
DispatcherConfig config;
config.ExecuteBlocksWithCall = true;
Dispatcher = std::make_unique<Arm64Dispatcher>(ctx, Thread, config);
Dispatcher = new Arm64Dispatcher(ctx, Thread, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
@@ -3,13 +3,7 @@
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
namespace FEXCore::Context {
struct Context;
}
namespace FEXCore::Core {
struct InternalThreadState;
}
#include "aarch64/assembler-aarch64.h"
namespace FEXCore::CPU {
@@ -18,7 +12,7 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
protected:
void SpillSRA(void *ucontext, uint32_t IgnoreMask) override;
void SpillSRA(void *ucontext) override;
};
}
}
@@ -1,24 +1,8 @@
#include "Interface/Core/ArchHelpers/MContext.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Core/CompileService.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/UContext.h>
#include "Common/MathUtils.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <atomic>
#include <condition_variable>
#include <bits/types/siginfo_t.h>
#include <csignal>
#include <cstring>
namespace FEXCore::CPU {
@@ -28,19 +12,15 @@ void Dispatcher::SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuS
--ctx->IdleWaitRefCount;
ctx->IdleWaitCV.notify_all();
Thread->RunningEvents.ThreadSleeping = true;
// Go to sleep
Thread->StartRunning.Wait();
Thread->RunningEvents.Running = true;
++ctx->IdleWaitRefCount;
Thread->RunningEvents.ThreadSleeping = false;
ctx->IdleWaitCV.notify_all();
}
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, void *ucontext) {
void Dispatcher::StoreThreadState(int Signal, void *ucontext) {
// We can end up getting a signal at any point in our host state
// Jump to a handler that saves all state so we can safely return
uint64_t OldSP = ArchHelpers::Context::GetSp(ucontext);
@@ -70,31 +50,12 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
// Set the new SP
ArchHelpers::Context::SetSp(ucontext, NewSP);
// Signal frames are only used on the interpreter
// The JITS require the stack to be setup correctly on rt_sigreturn
if (CTX->Config.Core() == FEXCore::Config::CONFIG_INTERPRETER) {
SignalFrames.push(NewSP);
}
Context->Flags = 0;
Context->FPStateLocation = 0;
Context->UContextLocation = 0;
Context->SigInfoLocation = 0;
return Context;
SignalFrames.push(NewSP);
}
void Dispatcher::RestoreThreadState(void *ucontext) {
uint64_t OldSP{};
if (CTX->Config.Core() == FEXCore::Config::CONFIG_IRJIT) {
OldSP = ArchHelpers::Context::GetSp(ucontext);
}
else {
LOGMAN_THROW_A_FMT(!SignalFrames.empty(), "Trying to restore a signal frame when we don't have any");
OldSP = SignalFrames.top();
SignalFrames.pop();
}
uint64_t OldSP = SignalFrames.top();
SignalFrames.pop();
uintptr_t NewSP = OldSP;
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
@@ -104,240 +65,69 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
// Now restore host state
ArchHelpers::Context::RestoreContext(ucontext, Context);
if (Context->UContextLocation) {
auto Frame = ThreadState->CurrentFrame;
if (Context->Flags &ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT) {
// XXX: Unsupported since it needs state reconstruction
// If we are in the JIT then SRA might need to be restored to values from the context
// We can't currently support this since it might result in tearing without real state reconstruction
}
if (!(Context->Flags & ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT)) {
auto *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(Context->UContextLocation);
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<siginfo_t*>(Context->SigInfoLocation);
// If the guest modified the RIP then we need to take special precautions here
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP]) {
// Hack! Go back to the top of the dispatcher top
// This is only safe inside the JIT rather than anything outside of it
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
// Set our state register to point to our guest thread data
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
// XXX: Full context setting
#define COPY_REG(x) \
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x];
COPY_REG(R8);
COPY_REG(R9);
COPY_REG(R10);
COPY_REG(R11);
COPY_REG(R12);
COPY_REG(R13);
COPY_REG(R14);
COPY_REG(R15);
COPY_REG(RDI);
COPY_REG(RSI);
COPY_REG(RBP);
COPY_REG(RBX);
COPY_REG(RDX);
COPY_REG(RAX);
COPY_REG(RCX);
COPY_REG(RSP);
#undef COPY_REG
}
}
else {
auto *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(Context->UContextLocation);
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(Context->SigInfoLocation);
// If the guest modified the RIP then we need to take special precautions here
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP]) {
// Hack! Go back to the top of the dispatcher top
// This is only safe inside the JIT rather than anything outside of it
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
// Set our state register to point to our guest thread data
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
// XXX: Full context setting
#define COPY_REG(x) \
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
COPY_REG(RDI);
COPY_REG(RSI);
COPY_REG(RBP);
COPY_REG(RBX);
COPY_REG(RDX);
COPY_REG(RAX);
COPY_REG(RCX);
COPY_REG(RSP);
#undef COPY_REG
}
}
}
}
static uint32_t ConvertSignalToTrapNo(int Signal, siginfo_t *HostSigInfo) {
switch (Signal) {
case SIGSEGV:
if (HostSigInfo->si_code == SEGV_MAPERR ||
HostSigInfo->si_code == SEGV_ACCERR) {
// Protection fault
return X86State::X86_TRAPNO_PF;
}
break;
}
// Unknown mapping, fall back to old behaviour and just pass signal
return Signal;
}
static uint32_t ConvertSignalToError(int Signal, siginfo_t *HostSigInfo) {
switch (Signal) {
case SIGSEGV:
if (HostSigInfo->si_code == SEGV_MAPERR ||
HostSigInfo->si_code == SEGV_ACCERR) {
// Protection fault
// Always a user fault for us
// XXX: PF_PROT and PF_WRITE
return X86State::X86_PF_USER;
}
break;
}
// Not a page fault issue
return 0;
// Restore the previous signal state
// This allows recursive signals to properly handle signal masking as we are walking back up the list of signals
CTX->SignalDelegation->SetCurrentSignal(Context->Signal);
}
bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
auto ContextBackup = StoreThreadState(Signal, ucontext);
StoreThreadState(Signal, ucontext);
auto Frame = ThreadState->CurrentFrame;
// Ref count our faults
// We use this to track if it is safe to clear cache
++SignalHandlerRefCounter;
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
// Set the new PC
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
// Set our state register to point to our guest thread data
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
uint64_t OldGuestSP = Frame->State.gregs[X86State::REG_RSP];
uint64_t NewGuestSP = OldGuestSP;
// Pulling from context here
bool Is64BitMode = CTX->Config.Is64BitMode;
uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
// Spill the SRA regardless of signal handler type
// We are going to be returning to the top of the dispatcher which will fill again
// Otherwise we might load garbage
if (SRAEnabled) {
if (IsAddressInJITCode(OldPC, false)) {
uint32_t IgnoreMask{};
#ifdef _M_ARM_64
if (Frame->InSyscallInfo != 0) {
// We are in a syscall, this means we are in a weird register state
// We need to spill SRA but only some of it, since some values have already been spilled
// Lower 16 bits tells us which registers are already spilled to the context
// So we ignore spilling those ones
uint16_t NumRegisters = std::popcount(Frame->InSyscallInfo & 0xFFFF);
if (NumRegisters >= 4) {
// Unhandled case
IgnoreMask = 0;
}
else {
IgnoreMask = Frame->InSyscallInfo & 0xFFFF;
}
}
else {
// We must spill everything
IgnoreMask = 0;
}
#endif
// We are in jit, SRA must be spilled
SpillSRA(ucontext, IgnoreMask);
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT;
} else {
if (!IsAddressInJITCode(OldPC, true)) {
// This is likely to cause issues but in some cases it isn't fatal
// This can also happen if we have put a signal on hold, then we just reenabled the signal
// So we are in the syscall handler
// Only throw a log message in this case
LogMan::Msg::EFmt("Signals in dispatcher have unsynchronized context");
}
if (!(GuestStack->ss_flags & SS_DISABLE)) {
// If our guest is already inside of the alternative stack
// Then that means we are hitting recursive signals and we need to walk back the stack correctly
uint64_t AltStackBase = reinterpret_cast<uint64_t>(GuestStack->ss_sp);
uint64_t AltStackEnd = AltStackBase + GuestStack->ss_size;
if (OldGuestSP >= AltStackBase &&
OldGuestSP <= AltStackEnd) {
// We are already in the alt stack, the rest of the code will handle adjusting this
}
else {
NewGuestSP = AltStackEnd;
}
}
// altstack is only used if the signal handler was setup with SA_ONSTACK
if (GuestAction->sa_flags & SA_ONSTACK) {
// Additionally the altstack is only used if the enabled (SS_DISABLE flag is not set)
if (!(GuestStack->ss_flags & SS_DISABLE)) {
// If our guest is already inside of the alternative stack
// Then that means we are hitting recursive signals and we need to walk back the stack correctly
uint64_t AltStackBase = reinterpret_cast<uint64_t>(GuestStack->ss_sp);
uint64_t AltStackEnd = AltStackBase + GuestStack->ss_size;
if (OldGuestSP >= AltStackBase &&
OldGuestSP <= AltStackEnd) {
// We are already in the alt stack, the rest of the code will handle adjusting this
}
else {
NewGuestSP = AltStackEnd;
}
}
}
if (Is64BitMode) {
// Back up past the redzone, which is 128bytes
// 32-bit doesn't have a redzone
NewGuestSP -= 128;
}
// siginfo_t
siginfo_t *HostSigInfo = reinterpret_cast<siginfo_t*>(info);
// Backup where we think the RIP currently is
ContextBackup->OriginalRIP = Frame->State.rip;
// Back up past the redzone, which is 128bytes
// Don't need this offset if we aren't going to be putting siginfo in to it
NewGuestSP -= 128;
if (GuestAction->sa_flags & SA_SIGINFO) {
if (SRAEnabled) {
if (!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
LogMan::Throw::A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
} else {
// We are in jit, SRA must be spilled
SpillSRA(ucontext);
}
}
// Setup ucontext a bit
if (Is64BitMode) {
NewGuestSP -= sizeof(FEXCore::x86_64::_libc_fpstate);
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::_libc_fpstate));
uint64_t FPStateLocation = NewGuestSP;
if (CTX->Config.Is64BitMode) {
NewGuestSP -= sizeof(FEXCore::x86_64::ucontext_t);
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::ucontext_t));
uint64_t UContextLocation = NewGuestSP;
NewGuestSP -= sizeof(siginfo_t);
NewGuestSP = AlignDown(NewGuestSP, alignof(siginfo_t));
uint64_t SigInfoLocation = NewGuestSP;
ContextBackup->FPStateLocation = FPStateLocation;
ContextBackup->UContextLocation = UContextLocation;
ContextBackup->SigInfoLocation = SigInfoLocation;
FEXCore::x86_64::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(UContextLocation);
siginfo_t *guest_siginfo = reinterpret_cast<siginfo_t*>(SigInfoLocation);
// We have extended float information
guest_uctx->uc_flags = FEXCore::x86_64::UC_FP_XSTATE;
guest_uctx->uc_flags |= FEXCore::x86_64::UC_FP_XSTATE;
// Pointer to where the fpreg memory is
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = Frame->State.rip;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CSGSFS] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_OLDMASK] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CR2] = 0;
guest_uctx->uc_mcontext.fpregs = &guest_uctx->__fpregs_mem;
#define COPY_REG(x) \
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
@@ -360,15 +150,14 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
#undef COPY_REG
// Copy float registers
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
memcpy(guest_uctx->__fpregs_mem._st, Frame->State.mm, sizeof(Frame->State.mm));
memcpy(guest_uctx->__fpregs_mem._xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
// FCW store default
fpstate->fcw = Frame->State.FCW;
fpstate->ftw = Frame->State.FTW;
guest_uctx->__fpregs_mem.fcw = Frame->State.FCW;
// Reconstruct FSW
fpstate->fsw =
guest_uctx->__fpregs_mem.fsw =
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
@@ -380,171 +169,44 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
guest_uctx->uc_stack.ss_sp = GuestStack->ss_sp;
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
// SI_USER could also potentially have random data in it, needs to be bit perfect
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
*guest_siginfo = *HostSigInfo;
Frame->State.gregs[X86State::REG_RSI] = SigInfoLocation;
// XXX: siginfo_t(RSI)
Frame->State.gregs[X86State::REG_RSI] = 0x4142434445460000;
Frame->State.gregs[X86State::REG_RDX] = UContextLocation;
}
else {
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT;
NewGuestSP -= sizeof(FEXCore::x86::_libc_fpstate);
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::_libc_fpstate));
uint64_t FPStateLocation = NewGuestSP;
// XXX: 32bit Support
NewGuestSP -= sizeof(FEXCore::x86::ucontext_t);
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::ucontext_t));
uint64_t UContextLocation = NewGuestSP;
uint64_t UContextLocation = 0; // NewGuestSP;
NewGuestSP -= sizeof(FEXCore::x86::siginfo_t);
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::siginfo_t));
uint64_t SigInfoLocation = NewGuestSP;
ContextBackup->FPStateLocation = FPStateLocation;
ContextBackup->UContextLocation = UContextLocation;
ContextBackup->SigInfoLocation = SigInfoLocation;
FEXCore::x86::ucontext_t *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(UContextLocation);
FEXCore::x86::siginfo_t *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(SigInfoLocation);
// We have extended float information
guest_uctx->uc_flags = FEXCore::x86::UC_FP_XSTATE;
// Pointer to where the fpreg memory is
guest_uctx->uc_mcontext.fpregs = static_cast<uint32_t>(FPStateLocation);
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(FPStateLocation);
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss;
#define COPY_REG(x) \
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
COPY_REG(RDI);
COPY_REG(RSI);
COPY_REG(RBP);
COPY_REG(RBX);
COPY_REG(RDX);
COPY_REG(RAX);
COPY_REG(RCX);
COPY_REG(RSP);
#undef COPY_REG
// Copy float registers
for (size_t i = 0; i < 8; ++i) {
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
}
// Extended XMM state
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
// FCW store default
fpstate->fcw = Frame->State.FCW;
fpstate->ftw = Frame->State.FTW;
// Reconstruct FSW
fpstate->fsw =
(Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] << 10) |
(Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] << 14);
// Copy over signal stack information
guest_uctx->uc_stack.ss_flags = GuestStack->ss_flags;
guest_uctx->uc_stack.ss_sp = static_cast<uint32_t>(reinterpret_cast<uint64_t>(GuestStack->ss_sp));
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
// These three elements are in every siginfo
guest_siginfo->si_signo = HostSigInfo->si_signo;
guest_siginfo->si_errno = HostSigInfo->si_errno;
guest_siginfo->si_code = HostSigInfo->si_code;
switch (Signal) {
case SIGSEGV:
case SIGBUS:
// Macro expansion to get the si_addr
// This is the address trying to be accessed, not the RIP
guest_siginfo->_sifields._sigfault.addr = static_cast<uint32_t>(reinterpret_cast<uintptr_t>(HostSigInfo->si_addr));
break;
case SIGFPE:
case SIGILL:
// Macro expansion to get the si_addr
// Can't really give a real result here. Pull from the context for now
guest_siginfo->_sifields._sigfault.addr = Frame->State.rip;
break;
case SIGCHLD:
guest_siginfo->_sifields._sigchld.pid = HostSigInfo->si_pid;
guest_siginfo->_sifields._sigchld.uid = HostSigInfo->si_uid;
guest_siginfo->_sifields._sigchld.status = HostSigInfo->si_status;
guest_siginfo->_sifields._sigchld.utime = HostSigInfo->si_utime;
guest_siginfo->_sifields._sigchld.stime = HostSigInfo->si_stime;
break;
case SIGALRM:
case SIGVTALRM:
guest_siginfo->_sifields._timer.tid = HostSigInfo->si_timerid;
guest_siginfo->_sifields._timer.overrun = HostSigInfo->si_overrun;
guest_siginfo->_sifields._timer.sigval.sival_int = HostSigInfo->si_int;
break;
default:
LogMan::Msg::EFmt("Unhandled siginfo_t for signal: {}\n", Signal);
break;
}
uint64_t SigInfoLocation = 0; // NewGuestSP;
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = UContextLocation;
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = SigInfoLocation;
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = Signal;
}
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.sigaction);
}
else {
if (!Is64BitMode) {
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = Signal;
}
Frame->State.rip = reinterpret_cast<uint64_t>(GuestAction->sigaction_handler.handler);
}
if (Is64BitMode) {
Frame->State.gregs[FEXCore::X86State::REG_RDI] = Signal;
if (CTX->Config.Is64BitMode) {
Frame->State.gregs[X86State::REG_RDI] = Signal;
// Set up the new SP for stack handling
NewGuestSP -= 8;
*(uint64_t*)NewGuestSP = SignalReturn;
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
*(uint64_t*)NewGuestSP = CTX->X86CodeGen.SignalReturn;
Frame->State.gregs[X86State::REG_RSP] = NewGuestSP;
}
else {
NewGuestSP -= 4;
*(uint32_t*)NewGuestSP = SignalReturn;
LOGMAN_THROW_A_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
*(uint32_t*)NewGuestSP = CTX->X86CodeGen.SignalReturn;
LogMan::Throw::A(CTX->X86CodeGen.SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
Frame->State.gregs[X86State::REG_RSP] = NewGuestSP;
}
// The guest starts its signal frame with a zero initialized FPU
// Set that up now. Little bit costly but it's a requirement
// This state will be restored on rt_sigreturn
memset(Frame->State.xmm, 0, sizeof(Frame->State.xmm));
memset(Frame->State.mm, 0, sizeof(Frame->State.mm));
Frame->State.FCW = 0x37F;
Frame->State.FTW = 0xFFFF;
return true;
}
@@ -575,7 +237,7 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
FEXCore::Core::SignalEvent SignalReason = ThreadState->SignalReason.load();
auto Frame = ThreadState->CurrentFrame;
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_PAUSE) {
// Store our thread state so we can come back to this
StoreThreadState(Signal, ucontext);
@@ -585,12 +247,14 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
} else {
if (SRAEnabled) {
// We are in non-jit, SRA is already spilled
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
"Signals in dispatcher have unsynchronized context");
LogMan::Throw::A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
}
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
}
// Set the new PC
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
// Set our state register to point to our guest thread data
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
@@ -598,11 +262,11 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
// We use this to track if it is safe to clear cache
++SignalHandlerRefCounter;
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
return true;
}
if (SignalReason == FEXCore::Core::SignalEvent::Stop) {
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_STOP) {
// Our thread is stopping
// We don't care about anything at this point
// Set the stack to our starting location when we entered the core and get out safely
@@ -618,33 +282,23 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
} else {
if (SRAEnabled) {
// We are in non-jit, SRA is already spilled
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
"Signals in dispatcher have unsynchronized context");
LogMan::Throw::A(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true), "Signals in dispatcher have unsynchronized context");
}
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
}
// We need to be a little bit careful here
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
// Then we need to ensure we don't double decrement our idle thread counter
if (ThreadState->RunningEvents.ThreadSleeping) {
// If the thread was sleeping then its idle counter was decremented
// Reincrement it here to not break logic
++ThreadState->CTX->IdleWaitRefCount;
}
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
return true;
}
if (SignalReason == FEXCore::Core::SignalEvent::Return) {
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_RETURN) {
RestoreThreadState(ucontext);
// Ref count our faults
// We use this to track if it is safe to clear cache
--SignalHandlerRefCounter;
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
return true;
}
@@ -652,14 +306,14 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
}
uint64_t Dispatcher::GetCompileBlockPtr() {
using ClassPtrType = void (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
PtrCast CompileBlockPtr;
CompileBlockPtr.ClassPtr = &FEXCore::Context::Context::CompileBlockJit;
CompileBlockPtr.ClassPtr = &FEXCore::Context::Context::CompileBlock;
return CompileBlockPtr.Data;
}
@@ -673,19 +327,15 @@ void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
}
}
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher, bool IncludeCompileService) const {
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher) {
for (auto [start, end] : CodeBuffers) {
if (Address >= start && Address < end) {
return true;
}
}
if (IncludeDispatcher && IsAddressInDispatcher(Address)) {
return true;
}
if (IncludeCompileService && ThreadState->CompileService && ThreadState->CompileService->IsAddressInJITCode(Address)) {
return true;
if (IncludeDispatcher) {
return IsAddressInDispatcher(Address);
}
return false;
}
@@ -1,25 +1,11 @@
#pragma once
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/SignalDelegator.h>
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include <bits/types/stack_t.h>
#include <cstdint>
#include <stddef.h>
#include <stack>
#include <tuple>
#include <vector>
namespace FEXCore {
struct GuestSigAction;
}
namespace FEXCore::Core {
struct CpuStateFrame;
struct InternalThreadState;
}
namespace FEXCore::CPU {
@@ -32,7 +18,6 @@ struct DispatcherConfig {
class Dispatcher {
public:
virtual ~Dispatcher() = default;
CPUBackend::AsmDispatch DispatchPtr;
CPUBackend::JITCallback CallbackPtr;
FEXCore::Context::Context::IntCallbackReturn ReturnPtr;
@@ -48,8 +33,6 @@ public:
uint64_t ThreadPauseHandlerAddressSpillSRA{};
uint64_t ExitFunctionLinkerAddress{};
uint64_t SignalHandlerReturnAddress{};
uint64_t UnimplementedInstructionAddress{};
uint64_t PauseReturnInstruction{};
/** @} */
@@ -70,8 +53,8 @@ public:
void RemoveCodeBuffer(uint8_t* start);
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const;
bool IsAddressInDispatcher(uint64_t Address) const {
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true);
bool IsAddressInDispatcher(uint64_t Address) {
return Address >= Start && Address < End;
}
@@ -80,12 +63,12 @@ protected:
: CTX {ctx}
, ThreadState {Thread} {}
ArchHelpers::Context::ContextBackup* StoreThreadState(int Signal, void *ucontext);
void StoreThreadState(int Signal, void *ucontext);
void RestoreThreadState(void *ucontext);
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
std::stack<uint64_t> SignalFrames;
bool SRAEnabled = false;
virtual void SpillSRA(void *ucontext, uint32_t IgnoreMask) {}
virtual void SpillSRA(void *ucontext) {}
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *ThreadState;
@@ -98,4 +81,4 @@ private:
std::vector<std::tuple<uint64_t, uint64_t>> CodeBuffers; // Start, End
};
}
}
@@ -1,23 +1,10 @@
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Context/Context.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/Allocator.h>
#include <cmath>
#include <memory>
#include <stddef.h>
#include <stdint.h>
#include <sys/mman.h>
#include "xbyak/xbyak.h"
namespace FEXCore::CPU {
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
@@ -25,9 +12,7 @@ static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
: Dispatcher(ctx, Thread)
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
nullptr) {
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE) {
using namespace Xbyak;
using namespace Xbyak::util;
@@ -81,7 +66,6 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
Label LoopTop;
Label FullLookup;
Label CallBlock;
Label NoBlock;
Label ExitBlock;
Label ThreadPauseHandler;
@@ -93,20 +77,17 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
// Load our RIP
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
// L1 Cache
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(rax, rdx);
if (!config.ExecuteBlocksWithCall)
{
// L1 Cache
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(rax, rdx);
and_(rax, LookupCache::L1_ENTRIES_MASK);
shl(rax, 4);
cmp(qword[r13 + rax + 8], rdx);
jne(FullLookup);
if (!config.ExecuteBlocksWithCall) {
and_(rax, LookupCache::L1_ENTRIES_MASK);
shl(rax, 4);
cmp(qword[r13 + rax + 8], rdx);
jne(FullLookup);
jmp(qword[r13 + rax + 0]);
} else {
mov(rax, qword[r13 + rax + 0]);
jmp(CallBlock);
}
L(FullLookup);
@@ -141,19 +122,19 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
je(NoBlock);
// Update L1
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(rcx, rdx);
and_(rcx, LookupCache::L1_ENTRIES_MASK);
shl(rcx, 1);
mov(qword[r13 + rcx*8 + 8], rdx);
mov(qword[r13 + rcx*8 + 0], rax);
if (config.ExecuteBlocksWithCall) {
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(rcx, rdx);
and_(rcx, LookupCache::L1_ENTRIES_MASK);
shl(rcx, 1);
mov(qword[r13 + rcx*8 + 8], rdx);
mov(qword[r13 + rcx*8 + 0], rax);
}
// Real block if we made it here
if (!config.ExecuteBlocksWithCall) {
jmp(rax);
} else {
L(CallBlock);
mov(rdi, STATE);
call(rax);
@@ -196,10 +177,19 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
{
L(NoBlock);
using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
PtrCast Ptr;
Ptr.ClassPtr = &FEXCore::Context::Context::CompileBlock;
// {rdi, rsi, rdx}
mov(rdi, reinterpret_cast<uint64_t>(CTX));
mov(rsi, STATE);
mov(rax, GetCompileBlockPtr());
mov(rax, Ptr.Data);
call(rax);
@@ -274,19 +264,14 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
{
// Signal return handler
SignalHandlerReturnAddress = getCurr<uint64_t>();
ud2();
}
{
// Guest SIGILL handler
// Needs to be distinct from the SignalHandlerReturnAddress
UnimplementedInstructionAddress = getCurr<uint64_t>();
ud2();
}
{
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
// using CallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
// rdi = thread
// rsi = rsp
@@ -311,17 +296,14 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
Start = reinterpret_cast<uint64_t>(getCode());
End = Start + getSize();
if (CTX->Config.BlockJITNaming()) {
#if ENABLE_JITSYMBOLS
std::string Name = "Dispatch_" + std::to_string(::gettid());
CTX->Symbols.Register(reinterpret_cast<void*>(Start), End-Start, Name);
}
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
}
CTX->Symbols.Register(Start, End-Start, Name);
#endif
}
X86Dispatcher::~X86Dispatcher() {
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
}
#ifdef _M_X86_64
@@ -330,7 +312,7 @@ void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore:
DispatcherConfig config;
config.ExecuteBlocksWithCall = true;
Dispatcher = std::make_unique<X86Dispatcher>(ctx, Thread, config);
Dispatcher = new X86Dispatcher(ctx, Thread, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
@@ -5,14 +5,6 @@
#define XBYAK64
#include <xbyak/xbyak.h>
namespace FEXCore::Context {
struct Context;
}
namespace FEXCore::Core {
struct InternalThreadState;
}
namespace FEXCore::CPU {
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
@@ -22,4 +14,4 @@ class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
virtual ~X86Dispatcher() override;
};
}
}
+121 -283
View File
@@ -7,30 +7,20 @@ $end_info$
#include "Interface/Context/Context.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/InternalThreadState.h"
#include <array>
#include <assert.h>
#include <algorithm>
#include <cstring>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/X86Tables.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Telemetry.h>
#include <set>
#include <sys/mman.h>
namespace FEXCore::Frontend {
#include "Interface/Core/VSyscall/VSyscall.inc"
using namespace FEXCore::X86Tables;
static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool HasREX, bool HasXMM, bool HasMM, uint8_t InvalidOffset = 16) {
using GPRArray = std::array<uint32_t, 16>;
static constexpr GPRArray GPRIndexes = {
constexpr std::array<uint64_t, 16> GPRIndexes = {
// Classical ordering?
FEXCore::X86State::REG_RAX,
FEXCore::X86State::REG_RCX,
@@ -50,7 +40,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
FEXCore::X86State::REG_R15,
};
static constexpr GPRArray GPR8BitHighIndexes = {
constexpr std::array<uint64_t, 16> GPR8BitHighIndexes = {
// Classical ordering?
FEXCore::X86State::REG_RAX,
FEXCore::X86State::REG_RCX,
@@ -70,7 +60,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
FEXCore::X86State::REG_R15,
};
static constexpr GPRArray XMMIndexes = {
constexpr std::array<uint64_t, 16> XMMIndexes = {
FEXCore::X86State::REG_XMM_0,
FEXCore::X86State::REG_XMM_1,
FEXCore::X86State::REG_XMM_2,
@@ -89,7 +79,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
FEXCore::X86State::REG_XMM_15,
};
static constexpr GPRArray MMIndexes = {
constexpr std::array<uint64_t, 16> MMIndexes = {
FEXCore::X86State::REG_MM_0,
FEXCore::X86State::REG_MM_1,
FEXCore::X86State::REG_MM_2,
@@ -108,7 +98,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
FEXCore::X86State::REG_INVALID
};
const GPRArray *GPRs = &GPRIndexes;
const std::array<uint64_t, 16> *GPRs = &GPRIndexes;
if (HasXMM) {
GPRs = &XMMIndexes;
}
@@ -127,78 +117,20 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
return (*GPRs)[(REX << 3) | bits];
}
static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
using GPRArray = std::array<uint32_t, 16>;
static constexpr GPRArray GPRIndexes = {
FEXCore::X86State::REG_RAX,
FEXCore::X86State::REG_RCX,
FEXCore::X86State::REG_RDX,
FEXCore::X86State::REG_RBX,
FEXCore::X86State::REG_RSP,
FEXCore::X86State::REG_RBP,
FEXCore::X86State::REG_RSI,
FEXCore::X86State::REG_RDI,
FEXCore::X86State::REG_R8,
FEXCore::X86State::REG_R9,
FEXCore::X86State::REG_R10,
FEXCore::X86State::REG_R11,
FEXCore::X86State::REG_R12,
FEXCore::X86State::REG_R13,
FEXCore::X86State::REG_R14,
FEXCore::X86State::REG_R15,
};
static constexpr GPRArray XMMIndexes = {
FEXCore::X86State::REG_XMM_0,
FEXCore::X86State::REG_XMM_1,
FEXCore::X86State::REG_XMM_2,
FEXCore::X86State::REG_XMM_3,
FEXCore::X86State::REG_XMM_4,
FEXCore::X86State::REG_XMM_5,
FEXCore::X86State::REG_XMM_6,
FEXCore::X86State::REG_XMM_7,
FEXCore::X86State::REG_XMM_8,
FEXCore::X86State::REG_XMM_9,
FEXCore::X86State::REG_XMM_10,
FEXCore::X86State::REG_XMM_11,
FEXCore::X86State::REG_XMM_12,
FEXCore::X86State::REG_XMM_13,
FEXCore::X86State::REG_XMM_14,
FEXCore::X86State::REG_XMM_15,
};
if (HasXMM) {
return XMMIndexes[vvvv];
} else {
return GPRIndexes[vvvv];
}
}
Decoder::Decoder(FEXCore::Context::Context *ctx)
: CTX {ctx}
, OSABI { ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN } {
// Using mmap is a start-up time optimization
// Take advantage of page faulting to reduce startup time for minimal runtime cost
DecodedBuffer =
reinterpret_cast<FEXCore::X86Tables::DecodedInst *>(
FEXCore::Allocator::mmap(0, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize,
PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
}
Decoder::~Decoder() {
FEXCore::Allocator::munmap(DecodedBuffer, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize);
: CTX {ctx} {
DecodedBuffer.resize(DefaultDecodedBufferSize);
}
uint8_t Decoder::ReadByte() {
uint8_t Byte = InstStream[InstructionSize];
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
LogMan::Throw::A(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
Instruction[InstructionSize] = Byte;
InstructionSize++;
return Byte;
}
uint8_t Decoder::PeekByte(uint8_t Offset) const {
uint8_t Decoder::PeekByte(uint8_t Offset) {
uint8_t Byte = InstStream[InstructionSize + Offset];
return Byte;
}
@@ -209,7 +141,7 @@ uint64_t Decoder::ReadData(uint8_t Size) {
}
if (Size > sizeof(uint64_t)) {
LOGMAN_MSG_A_FMT("Unknown data size to read");
LogMan::Msg::A("Unknown data size to read");
return 0;
}
@@ -264,9 +196,9 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModR
}
}
Operand->Type = DecodedOperand::OpType::SIB;
Operand->Data.SIB.Scale = 1;
Operand->Data.SIB.Offset = Literal;
Operand->TypeSIB.Type = DecodedOperand::TYPE_SIB;
Operand->TypeSIB.Scale = 1;
Operand->TypeSIB.Offset = Literal;
// Only called when ModRM.mod != 0b11
struct Encodings {
@@ -279,34 +211,34 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModR
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_INVALID, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RSI, 255},
{FEXCore::X86State::REG_RDI, 255},
{255, 255},
{FEXCore::X86State::REG_RBX, 255},
// Mod = 0b01
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RSI, 255},
{FEXCore::X86State::REG_RDI, 255},
{FEXCore::X86State::REG_RBP, 255},
{FEXCore::X86State::REG_RBX, 255},
// Mod = 0b10
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RSI},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_RDI},
{FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBP, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RBX, FEXCore::X86State::REG_INVALID},
{FEXCore::X86State::REG_RSI, 255},
{FEXCore::X86State::REG_RDI, 255},
{FEXCore::X86State::REG_RBP, 255},
{FEXCore::X86State::REG_RBX, 255},
}};
uint8_t LookupIndex = ModRM.mod << 3 | ModRM.rm;
auto it = Lookup[LookupIndex];
Operand->Data.SIB.Base = it.Base;
Operand->Data.SIB.Index = it.Index;
Operand->TypeSIB.Base = it.Base;
Operand->TypeSIB.Index = it.Index;
}
void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM) {
@@ -344,77 +276,79 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
}
// SIB
Operand->Type = DecodedOperand::OpType::SIB;
Operand->Data.SIB.Scale = 1 << SIB.scale;
Operand->TypeSIB.Type = DecodedOperand::TYPE_SIB;
Operand->TypeSIB.Scale = 1 << SIB.scale;
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
Operand->TypeSIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
Operand->TypeSIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
LOGMAN_THROW_A_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
uint64_t Literal {0};
LogMan::Throw::A(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
uint64_t Literal = ReadData(Displacement);
Literal = ReadData(Displacement);
if (Displacement == 1) {
Literal = static_cast<int8_t>(Literal);
}
Operand->Data.SIB.Offset = Literal;
Operand->TypeSIB.Offset = Literal;
}
else if (ModRM.mod == 0) {
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
if (ModRM.rm == 0b101) {
// 32bit Displacement
const uint32_t Literal = ReadData(4);
uint32_t Literal;
Literal = ReadData(4);
Operand->Type = DecodedOperand::OpType::RIPRelative;
Operand->Data.RIPLiteral.Value.u = Literal;
Operand->TypeRIPLiteral.Type = DecodedOperand::TYPE_RIP_RELATIVE;
Operand->TypeRIPLiteral.Literal.u = Literal;
}
else {
// Register-direct addressing
Operand->Type = DecodedOperand::OpType::GPRDirect;
Operand->Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
Operand->TypeGPR.Type = DecodedOperand::TYPE_GPR_DIRECT;
Operand->TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
}
}
else {
uint8_t DisplacementSize = ModRM.mod == 1 ? 1 : 4;
uint32_t Literal = ReadData(DisplacementSize);
uint32_t Literal{};
Literal = ReadData(DisplacementSize);
if (DisplacementSize == 1) {
Literal = static_cast<int8_t>(Literal);
}
Displacement = DisplacementSize;
Operand->Type = DecodedOperand::OpType::GPRIndirect;
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
Operand->Data.GPRIndirect.Displacement = Literal;
Operand->TypeGPRIndirect.Type = DecodedOperand::TYPE_GPR_INDIRECT;
Operand->TypeGPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
Operand->TypeGPRIndirect.Displacement = Literal;
}
}
bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options) {
bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op) {
DecodeInst->OP = Op;
DecodeInst->TableInfo = Info;
// XXX: Once we support 32bit x86 then this will be necessary to support
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
LogMan::Msg::DFmt("Legacy Prefix");
LogMan::Msg::D("Legacy Prefix");
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name, Op, DecodeInst->PC);
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name, Op, DecodeInst->PC);
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
return false;
}
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
"Group Ops should have been decoded before this!");
LogMan::Throw::A(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
"Group Ops should have been decoded before this!");
uint8_t DestSize{};
const bool HasWideningDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST) != 0 ||
(Options.w && CTX->Config.Is64BitMode);
const bool HasNarrowingDisplacement = (FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST) != 0;
bool HasWideningDisplacement = FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST;
bool HasNarrowingDisplacement = FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST;
bool HasXMMSrc = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_XMM_FLAGS) &&
!HAS_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_GPR) &&
@@ -447,8 +381,8 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
// New instruction size decoding
{
// Decode destinations first
const auto DstSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeDstFlags(Info->Flags);
const auto SrcSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeSrcFlags(Info->Flags);
uint32_t DstSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeDstFlags(Info->Flags);
uint32_t SrcSizeFlag = FEXCore::X86Tables::InstFlags::GetSizeSrcFlags(Info->Flags);
if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_8BIT) {
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_8BIT);
@@ -525,25 +459,22 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RDX)) {
// Some instructions hardcode their destination as RAX
CurrentDest->Type = DecodedOperand::OpType::GPR;
CurrentDest->Data.GPR.HighBits = false;
CurrentDest->Data.GPR.GPR = HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
CurrentDest->TypeGPR.Type = DecodedOperand::TYPE_GPR;
CurrentDest->TypeGPR.HighBits = false;
CurrentDest->TypeGPR.GPR = HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
CurrentDest = &DecodeInst->Src[0];
}
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
LogMan::Throw::A(!HasMODRM, "This instruction shouldn't have ModRM!");
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
// This also means that the destination is always a GPR on these ones
// ADDITIONALLY:
// If there is a REX prefix then that allows extended GPR usage
CurrentDest->Type = DecodedOperand::OpType::GPR;
DecodeInst->Dest.Data.GPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100) || HasHighXMM;
CurrentDest->Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
if (CurrentDest->Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
return false;
CurrentDest->TypeGPR.Type = DecodedOperand::TYPE_GPR;
DecodeInst->Dest.TypeGPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100) || HasHighXMM;
CurrentDest->TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
}
uint8_t Bytes = Info->MoreBytes;
@@ -566,83 +497,55 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
ModRM.Hex = DecodeInst->ModRM;
// Decode the GPR source first
GPR.Type = DecodedOperand::OpType::GPR;
GPR.Data.GPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX) || HasHighXMM;
GPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_R ? 1 : 0, ModRM.reg, GPR8Bit, HasREX, HasXMMGPR, HasMMGPR);
if (GPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
return false;
GPR.TypeGPR.Type = DecodedOperand::TYPE_GPR;
GPR.TypeGPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX) || HasHighXMM;
GPR.TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_R ? 1 : 0, ModRM.reg, GPR8Bit, HasREX, HasXMMGPR, HasMMGPR);
// ModRM.mod == 0b11 == Register
// ModRM.Mod != 0b11 == Register-direct addressing
if (ModRM.mod == 0b11) {
NonGPR.Type = DecodedOperand::OpType::GPR;
NonGPR.Data.GPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX) || HasHighXMM;
NonGPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, NonGPR8Bit, HasREX, HasXMMNonGPR, HasMMNonGPR);
if (NonGPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
return false;
NonGPR.TypeGPR.Type = DecodedOperand::TYPE_GPR;
NonGPR.TypeGPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX) || HasHighXMM;
NonGPR.TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, NonGPR8Bit, HasREX, HasXMMNonGPR, HasMMNonGPR);
}
else {
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
(this->*Disp)(&NonGPR, ModRM);
}
return true;
};
size_t CurrentSrc = 0;
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) != 0) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
++CurrentSrc;
}
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM) {
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SF_MOD_DST) {
if (!ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest))
return false;
ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest);
}
else {
if (!ModRMOperand(DecodeInst->Dest, DecodeInst->Src[CurrentSrc], HasXMMDst, HasXMMSrc, HasMMDst, HasMMSrc, Is8BitDest, Is8BitSrc))
return false;
ModRMOperand(DecodeInst->Dest, DecodeInst->Src[CurrentSrc], HasXMMDst, HasXMMSrc, HasMMDst, HasMMSrc, Is8BitDest, Is8BitSrc);
}
++CurrentSrc;
}
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) != 0) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
++CurrentSrc;
}
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RAX)) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RAX;
DecodeInst->Src[CurrentSrc].TypeGPR.Type = DecodedOperand::TYPE_GPR;
DecodeInst->Src[CurrentSrc].TypeGPR.HighBits = false;
DecodeInst->Src[CurrentSrc].TypeGPR.GPR = FEXCore::X86State::REG_RAX;
++CurrentSrc;
}
else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RCX)) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RCX;
DecodeInst->Src[CurrentSrc].TypeGPR.Type = DecodedOperand::TYPE_GPR;
DecodeInst->Src[CurrentSrc].TypeGPR.HighBits = false;
DecodeInst->Src[CurrentSrc].TypeGPR.GPR = FEXCore::X86State::REG_RCX;
++CurrentSrc;
}
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) != 0) {
CurrentDest->Type = DecodedOperand::OpType::GPR;
CurrentDest->Data.GPR.HighBits = false;
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
}
if (Bytes != 0) {
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
LogMan::Throw::A(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
DecodeInst->Src[CurrentSrc].TypeLiteral.Size = Bytes;
uint64_t Literal = ReadData(Bytes);
uint64_t Literal {0};
Literal = ReadData(Bytes);
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT) ||
(DecodeFlags::GetSizeDstFlags(DecodeInst->Flags) == DecodeFlags::SIZE_64BIT && Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT64BIT)) {
@@ -655,16 +558,15 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
else {
Literal = static_cast<int32_t>(Literal);
}
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
DecodeInst->Src[CurrentSrc].TypeLiteral.Size = DestSize;
}
Bytes = 0;
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
DecodeInst->Src[CurrentSrc].TypeLiteral.Type = DecodedOperand::TYPE_LITERAL;
DecodeInst->Src[CurrentSrc].TypeLiteral.Literal = Literal;
}
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name, InstructionSize, Bytes);
LogMan::Throw::A(Bytes == 0, "Inst at 0x%lx: 0x%04x '%s' Had an instruction of size %d with %d remaining", DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name, InstructionSize, Bytes);
DecodeInst->InstSize = InstructionSize;
return true;
}
@@ -675,22 +577,21 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
// XXX: Once we support 32bit x86 then this will be necessary to support
if (Info->Type == FEXCore::X86Tables::TYPE_LEGACY_PREFIX) {
LogMan::Msg::DFmt("Legacy Prefix");
LogMan::Msg::D("Legacy Prefix");
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name, Op, DecodeInst->PC);
LogMan::Msg::D("Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
return false;
}
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name, Op, DecodeInst->PC);
LogMan::Msg::D("Invalid or Unknown instruction: %s 0x%04x 0x%lx", Info->Name, Op, DecodeInst->PC);
return false;
}
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
"REX PREFIX should have been decoded before this!");
LogMan::Throw::A(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
@@ -746,7 +647,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
3,
};
uint8_t Field = RegToField[ModRM.reg];
LOGMAN_THROW_A_FMT(Field != 255, "Invalid field selected!");
LogMan::Throw::A(Field != 255, "Invalid field selected!");
LocalOp = (Field << 3) | ModRM.rm;
return NormalOp(&SecondModRMTableOps[LocalOp], LocalOp);
@@ -768,38 +669,19 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
return NormalOp(&X87Ops[X87Op], X87Op);
}
else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
uint16_t map_select = 1;
uint16_t pp = 0;
const uint8_t Byte1 = ReadByte();
DecodedHeader options{};
if ((Byte1 & 0b10000000) == 0) {
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.R shouldn't be 0 in 32-bit mode!");
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
}
uint8_t Byte1 = ReadByte();
if (Op == 0xC5) { // Two byte VEX
pp = Byte1 & 0b11;
options.vvvv = 15 - ((Byte1 & 0b01111000) >> 3);
}
else { // 0xC4 = Three byte VEX
const uint8_t Byte2 = ReadByte();
uint8_t Byte2 = ReadByte();
pp = Byte2 & 0b11;
map_select = Byte1 & 0b11111;
options.vvvv = 15 - ((Byte2 & 0b01111000) >> 3);
options.w = (Byte2 & 0b10000000) != 0;
if ((Byte1 & 0b01000000) == 0) {
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "VEX.X shouldn't be 0 in 32-bit mode!");
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
}
if (CTX->Config.Is64BitMode && (Byte1 & 0b00100000) == 0) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
}
if (!(map_select >= 1 && map_select <= 3)) {
LogMan::Msg::EFmt("We don't understand a map_select of: {}", map_select);
return false;
}
LogMan::Throw::A(map_select >= 1 && map_select <= 3, "We don't understand a map_select of: %d", map_select);
}
uint16_t VEXOp = ReadByte();
@@ -811,7 +693,6 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
if (LocalInfo->Type >= FEXCore::X86Tables::TYPE_VEX_GROUP_12 &&
LocalInfo->Type <= FEXCore::X86Tables::TYPE_VEX_GROUP_17) {
FEXCORE_TELEMETRY_SET(VEXOpTelem, 1);
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
@@ -823,14 +704,12 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
Op = OPD(LocalInfo->Type, pp, ModRM.reg);
#undef OPD
return NormalOp(&VEXTableGroupOps[Op], Op, options);
} else {
return NormalOp(LocalInfo, Op, options);
return NormalOp(&VEXTableGroupOps[Op], Op);
}
else
return NormalOp(LocalInfo, Op);
}
else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
/* uint8_t P1 = */ ReadByte();
/* uint8_t P2 = */ ReadByte();
/* uint8_t P3 = */ ReadByte();
@@ -850,8 +729,6 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
DecodeInst->PC = PC;
for(;;) {
if (InstructionSize >= MAX_INST_SIZE)
return false;
uint8_t Op = ReadByte();
switch (Op) {
case 0x0F: {// Escape Op
@@ -882,19 +759,12 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
constexpr uint16_t PF_38_NONE = 0;
constexpr uint16_t PF_38_66 = 1;
constexpr uint16_t PF_38_F2 = 2;
constexpr uint16_t PF_38_F3 = 3;
uint16_t Prefix = PF_38_NONE;
if (DecodeInst->LastEscapePrefix == 0xF2) {
// Repeat prefix or instruction-specific
if (DecodeInst->LastEscapePrefix == 0xF2) // REPNE
Prefix = PF_38_F2;
} else if (DecodeInst->LastEscapePrefix == 0xF3) {
// Repeat prefix or instruction-specific
Prefix = PF_38_F3;
} else if (DecodeInst->LastEscapePrefix == 0x66) {
// Operand size
else if (DecodeInst->LastEscapePrefix == 0x66) // Operand Size
Prefix = PF_38_66;
}
uint16_t LocalOp = (Prefix << 8) | ReadByte();
return NormalOpHeader(&FEXCore::X86Tables::H0F38TableOps[LocalOp], LocalOp);
@@ -1009,7 +879,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
auto Info = &FEXCore::X86Tables::BaseOps[Op];
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
LOGMAN_THROW_A_FMT(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
LogMan::Throw::A(CTX->Config.Is64BitMode, "Got REX prefix in 32bit mode");
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
// Widening displacement
@@ -1039,10 +909,6 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
}
if (DecodeInst->Dest.IsGPR()) {
assert(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID);
}
return true;
}
@@ -1052,7 +918,7 @@ void Decoder::BranchTargetInMultiblockRange() {
// If the RIP setting is conditional AND within our symbol range then it can be considered for multiblock
uint64_t TargetRIP = 0;
const uint8_t GPRSize = CTX->GetGPRSize();
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
bool Conditional = true;
switch (DecodeInst->OP) {
@@ -1062,23 +928,19 @@ void Decoder::BranchTargetInMultiblockRange() {
// auto RIPOffset = LoadSource(Op, Op->Src[0], Op->Flags);
// auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
// Target offset is PC + InstSize + Literal
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
LogMan::Throw::A(DecodeInst->Src[0].TypeNone.Type == DecodedOperand::TYPE_LITERAL, "Had wrong operand type");
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].TypeLiteral.Literal;
break;
}
case 0xE9:
case 0xEB: // Both are unconditional JMP instructions
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
LogMan::Throw::A(DecodeInst->Src[0].TypeNone.Type == DecodedOperand::TYPE_LITERAL, "Had wrong operand type");
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].TypeLiteral.Literal;
Conditional = false;
break;
case 0xE8: // Call - Immediate target, We don't want to inline calls
if (ExternalBranches) {
ExternalBranches->insert(DecodeInst->PC + DecodeInst->InstSize);
}
[[fallthrough]];
case 0xC2: // RET imm
case 0xC3: // RET
case 0xE8: // Call - Immediate target, We don't want to inline calls
default:
return;
break;
@@ -1108,33 +970,10 @@ void Decoder::BranchTargetInMultiblockRange() {
BlocksToDecode.find(TargetRIP) == BlocksToDecode.end()) {
BlocksToDecode.emplace(TargetRIP);
}
} else {
if (ExternalBranches) {
ExternalBranches->insert(TargetRIP);
}
}
}
const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64 &&
RIP >= VSyscall_Base &&
RIP < VSyscall_End) {
// VSyscall
// This doesn't exist on AArch64 and on x86_64 hosts this is emulated with faults to a region mapped with --xp permissions
// Offset 0: vgettimeofday
// Offset 0x400: vtime
// Offset 0x800: vgetcpu
uint64_t Offset = RIP - VSyscall_Base;
return VSyscallData + Offset;
}
return _InstStream - EntryPoint + RIP;
}
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC) {
bool Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC) {
Blocks.clear();
BlocksToDecode.clear();
HasBlocks.clear();
@@ -1148,12 +987,13 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
EntryPoint = PC;
InstStream = _InstStream;
bool ErrorDuringDecoding = false;
uint64_t TotalInstructions{};
// If we don't have symbols available then we become a bit optimistic about multiblock ranges
if (!SymbolAvailable) {
// If we don't have a symbol available then assume all branches are valid for multiblock
SymbolMaxAddress = SectionMaxAddress;
SymbolMaxAddress = ~0ULL;
SymbolMinAddress = EntryPoint;
}
@@ -1176,18 +1016,20 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
uint64_t BlockStartOffset = DecodedSize;
// Do a bit of pointer math to figure out where we are in code
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
InstStream = _InstStream - EntryPoint + RIPToDecode;
while (1) {
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
if (ErrorDuringDecoding) {
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", PC + PCOffset, PC);
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
LogMan::Msg::D("Couldn't Decode something at 0x%lx, Started at 0x%lx", PC + PCOffset, PC);
LogMan::Throw::A(Blocks.size() != 1, "Decode Error in entry block");
CurrentBlockDecoding.HasInvalidInstruction = true;
// Error while decoding instruction. We don't know the table or instruction size
DecodeInst->TableInfo = nullptr;
DecodeInst->InstSize = 0;
if (ErrorDuringDecoding && Blocks.size() != 1) {
ErrorDuringDecoding = false;
}
break;
}
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
@@ -1196,11 +1038,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
++BlockNumberOfInstructions;
++DecodedSize;
// Can not continue this block at all on invalid instruction
if (CurrentBlockDecoding.HasInvalidInstruction) {
break;
}
bool CanContinue = false;
if (!(DecodeInst->TableInfo->Flags &
(FEXCore::X86Tables::InstFlags::FLAGS_BLOCK_END | FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP))) {
@@ -1220,7 +1057,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
}
if (DecodedSize >= CTX->Config.MaxInstPerBlock ||
DecodedSize >= DefaultDecodedBufferSize) {
DecodedSize >= DecodedBuffer.size()) {
break;
}
@@ -1237,7 +1074,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
// Copy over only the number of instructions we decoded
CurrentBlockDecoding.NumInstructions = BlockNumberOfInstructions;
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer[BlockStartOffset];
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer.at(BlockStartOffset);
}
@@ -1245,6 +1082,7 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
std::sort(Blocks.begin(), Blocks.end(), [](const FEXCore::Frontend::Decoder::DecodedBlocks& a, const FEXCore::Frontend::Decoder::DecodedBlocks& b) {
return a.Entry < b.Entry;
});
return !ErrorDuringDecoding;
}
}
+8 -29
View File
@@ -1,13 +1,11 @@
#pragma once
#include <FEXCore/Debug/X86Tables.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <array>
#include <cstdint>
#include <utility>
#include <set>
#include <stddef.h>
#include <stack>
#include <vector>
namespace FEXCore::Context {
@@ -26,43 +24,31 @@ public:
};
Decoder(FEXCore::Context::Context *ctx);
~Decoder();
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
bool DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
std::vector<DecodedBlocks> const *GetDecodedBlocks() {
return &Blocks;
}
uint64_t DecodedMinAddress {};
uint64_t DecodedMaxAddress {~0ULL};
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
private:
// To pass any information from instruction prefixes
// down into the actual instruction handling machinery.
struct DecodedHeader {
uint8_t vvvv; // Encoded operand in a VEX prefix.
bool w; // VEX.W bit.
};
FEXCore::Context::Context *CTX;
const FEXCore::HLE::SyscallOSABI OSABI{};
bool DecodeInstruction(uint64_t PC);
void BranchTargetInMultiblockRange();
uint8_t ReadByte();
uint8_t PeekByte(uint8_t Offset) const;
uint8_t PeekByte(uint8_t Offset);
uint64_t ReadData(uint8_t Size);
void SkipBytes(uint8_t Size) { InstructionSize += Size; }
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
bool NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
std::vector<FEXCore::X86Tables::DecodedInst> DecodedBuffer;
size_t DecodedSize {};
uint8_t const *InstStream;
@@ -79,26 +65,19 @@ private:
uint64_t MaxCondBranchBackwards {~0ULL};
uint64_t SymbolMaxAddress {};
uint64_t SymbolMinAddress {~0ULL};
uint64_t SectionMaxAddress {~0ULL};
std::vector<DecodedBlocks> Blocks;
std::set<uint64_t> BlocksToDecode;
std::set<uint64_t> HasBlocks;
std::set<uint64_t> *ExternalBranches {nullptr};
// ModRM rm decoding
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp{
const std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp {
&FEXCore::Frontend::Decoder::DecodeModRM_64,
&FEXCore::Frontend::Decoder::DecodeModRM_16,
};
const uint8_t *AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP);
FEXCORE_TELEMETRY_INIT(VEXOpTelem, TYPE_USES_VEX_OPS);
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
};
}
+170 -322
View File
@@ -8,62 +8,46 @@ $end_info$
#include <cstdlib>
#include <cstdio>
#include <iomanip>
#include <iostream>
#include <sstream>
#include <string>
#include <memory>
#include <optional>
#include <vector>
#include "Common/NetStream.h"
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/Linux/ThreadManagement.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/NetStream.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Threads.h>
#include <atomic>
#include <cstring>
#include <errno.h>
#include <fcntl.h>
#include <fstream>
#include <fmt/format.h>
#include <netdb.h>
#include <signal.h>
#include <stddef.h>
#include <string_view>
#include <sys/types.h>
#include <sys/socket.h>
#include <netdb.h>
#include <string.h>
#include <fcntl.h>
#include <unistd.h>
#include <utility>
#include <vector>
#include <fstream>
#include "GdbServer.h"
#include <FEXCore/Core/CodeLoader.h>
#include <FEXCore/Core/X86Enums.h>
namespace FEXCore
{
void GdbServer::Break(int signal) {
std::lock_guard lk(sendMutex);
if (!CommsStream) {
return;
}
const auto str = fmt::format("S{:02x}", signal);
SendPacket(*CommsStream, str);
std::ostringstream ss;
ss << "S" << std::setfill('0') << std::setw(2) << std::hex << signal;
if (CommsStream)
SendPacket(*CommsStream, ss.str());
}
GdbServer::GdbServer(FEXCore::Context::Context *ctx) : CTX(ctx) {
Context::SetExitHandler(ctx, [this](uint64_t ThreadId, FEXCore::Context::ExitReason ExitReason) {
ctx->CustomExitHandler = [this](uint64_t ThreadId, FEXCore::Context::ExitReason ExitReason) {
if (ExitReason == FEXCore::Context::ExitReason::EXIT_DEBUG) {
this->Break(SIGTRAP);
}
});
};
// This is a total hack as there is currently no way to resume once hitting a segfault
// But it's semi-useful for debugging.
@@ -76,12 +60,12 @@ GdbServer::GdbServer(FEXCore::Context::Context *ctx) : CTX(ctx) {
usleep(100000);
return true;
}, true);
});
StartThread();
}
static int calculateChecksum(const std::string &packet) {
static int calculateChecksum(std::string &packet) {
unsigned char checksum = 0;
for (const char &c : packet) {
checksum += c;
@@ -115,9 +99,11 @@ static std::string encodeHex(unsigned char *data, size_t length) {
}
static std::string getThreadName(uint32_t ThreadID) {
const auto ThreadFile = fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
std::fstream fs(ThreadFile, std::fstream::in | std::fstream::binary);
std::fstream fs;
std::ostringstream ThreadFile;
ThreadFile << "/proc/" << getpid() << "/task/" << ThreadID << "/comm";
fs.open(ThreadFile.str(), std::fstream::in | std::fstream::binary);
if (fs.is_open()) {
std::string ThreadName;
fs >> ThreadName;
@@ -149,7 +135,7 @@ std::string GdbServer::ReadPacket(std::iostream &stream) {
switch(c) {
case '$': // start of packet
if (packet.size() != 0)
LogMan::Msg::EFmt("Dropping unexpected data: \"{}\"", packet);
LogMan::Msg::E("Dropping unexpected data: \"%s\"", packet.c_str());
// clear any existing data, must have been a mistake.
packet = std::string();
@@ -170,7 +156,7 @@ std::string GdbServer::ReadPacket(std::iostream &stream) {
if (calculateChecksum(packet) == expected_checksum) {
return packet;
} else {
LogMan::Msg::EFmt("Received Invalid Packet: ${}#{:02x}", packet, expected_checksum);
LogMan::Msg::E("Received Invalid Packet: $%s#%02x %c%c", packet.c_str(), expected_checksum);
}
break;
}
@@ -183,10 +169,10 @@ std::string GdbServer::ReadPacket(std::iostream &stream) {
return "";
}
static std::string escapePacket(const std::string& packet) {
static std::string escapePacket(std::string packet) {
std::ostringstream ss;
for(const auto &c : packet) {
for(auto &c : packet) {
switch (c) {
case '$':
case '#':
@@ -205,11 +191,13 @@ static std::string escapePacket(const std::string& packet) {
return ss.str();
}
void GdbServer::SendPacket(std::ostream &stream, const std::string& packet) {
const auto escaped = escapePacket(packet);
const auto str = fmt::format("${}#{:02x}", escaped, calculateChecksum(escaped));
void GdbServer::SendPacket(std::ostream &stream, std::string packet) {
auto escaped = escapePacket(packet);
std::ostringstream ss;
stream << str << std::flush;
ss << '$' << escaped << '#';
ss << std::setfill('0') << std::setw(2) << std::hex << (int)calculateChecksum(escaped);
stream << ss.str() << std::flush;
}
void GdbServer::SendACK(std::ostream &stream, bool NACK) {
@@ -230,7 +218,7 @@ void GdbServer::SendACK(std::ostream &stream, bool NACK) {
}
}
struct FEX_PACKED GDBContextDefinition {
struct __attribute__((packed)) GDBContextDefinition {
uint64_t gregs[16];
uint64_t rip;
uint32_t eflags;
@@ -291,7 +279,7 @@ std::string GdbServer::readRegs() {
return encodeHex((unsigned char *)&GDB, sizeof(GDBContextDefinition));
}
GdbServer::HandledPacketType GdbServer::readReg(const std::string& packet) {
GdbServer::HandledPacketType GdbServer::readReg(std::string& packet) {
size_t addr;
auto ss = std::istringstream(packet);
ss.get(); // Drop first letter
@@ -369,7 +357,7 @@ GdbServer::HandledPacketType GdbServer::readReg(const std::string& packet) {
return {encodeHex((unsigned char *)(&Empty), sizeof(uint32_t)), HandledPacketType::TYPE_ACK};
}
LogMan::Msg::EFmt("Unknown GDB register 0x{:x}", addr);
LogMan::Msg::E("Unknown GDB register 0x%lx", addr);
return {"E00", HandledPacketType::TYPE_ACK};
}
@@ -474,49 +462,7 @@ std::string buildTargetXML() {
return xml.str();
}
std::string buildMemoryMap() {
std::ostringstream xml;
xml << "<?xml version='1.0'?>\n";
xml << "<!DOCTYPE memory-map>\n";
xml << "<memory-map>";
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
std::string Line;
while (std::getline(fs, Line)) {
if (fs.eof()) break;
uint64_t Begin, End;
char r,w,x,p;
if (sscanf(Line.c_str(), "%lx-%lx %c%c%c%c", &Begin, &End, &r, &w, &x, &p) == 6) {
xml << "<memory type=\"ram\" start=\"0x" << std::hex << Begin << "\" length=\"0x" << (End - Begin) << "\"/>\n";
}
}
xml << "</memory-map>";
xml << std::flush;
return xml.str();
}
std::string buildOSData() {
std::ostringstream xml;
xml << "<?xml version='1.0'?>\n";
xml << "<!DOCTYPE target SYSTEM \"osdata.dtd\">\n";
xml << "<osdata type=\"processes\">";
// XXX
xml << "</osdata>";
xml << std::flush;
return xml.str();
}
GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
GdbServer::HandledPacketType GdbServer::handleXfer(std::string &packet) {
std::string object;
std::string rw;
std::string annex;
@@ -582,11 +528,11 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
ThreadString.clear();
std::ostringstream ss;
ss << "<?xml version=\"1.0\"?>\n";
ss << "<?xml version=\"1.0\?>\n";
ss << "<threads>\n";
for (auto &Thread : *Threads) {
// Thread id is in hex without 0x prefix
ss << "\t<thread id=\"" << std::hex << Thread->ThreadManager.GetTID() << "\" name=\"" << getThreadName(Thread->ThreadManager.GetTID()) << "\">\n";
for (size_t i = 0; i < Threads->size(); ++i) {
auto Thread = Threads->at(i);
ss << "\t<thread id=\"" << std::hex << Thread->ThreadManager.GetTID() << "\" core=\"" << i << "\" name=\"" << getThreadName(Thread->ThreadManager.GetTID()) << "\">\n";
ss << "\t</thread>\n";
}
@@ -597,28 +543,15 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
return {encode(ThreadString.substr(offset, length)), HandledPacketType::TYPE_ACK};
}
if (object == "memory-map") {
if (offset == 0) {
MemoryMapString = buildMemoryMap();
}
return {encode(MemoryMapString.substr(offset, length)), HandledPacketType::TYPE_ACK};
}
if (object == "osdata") {
if (offset == 0) {
OSDataString = buildOSData();
}
return {encode(OSDataString.substr(offset, length)), HandledPacketType::TYPE_ACK};
}
return {"", HandledPacketType::TYPE_UNKNOWN};
}
static size_t CheckMemMapping(uint64_t Address, size_t Size) {
uint64_t AddressEnd = Address + Size;
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
std::string Line;
std::fstream fs;
fs.open("/proc/self/maps", std::fstream::in | std::fstream::binary);
std::string Line;
while (std::getline(fs, Line)) {
if (fs.eof()) break;
uint64_t Begin, End;
@@ -635,29 +568,32 @@ static size_t CheckMemMapping(uint64_t Address, size_t Size) {
}
}
fs.close();
return 0;
}
GdbServer::HandledPacketType GdbServer::handleProgramOffsets() {
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
std::fstream fs;
fs.open("/proc/self/maps", std::fstream::in | std::fstream::binary);
std::string Line;
std::string const &RuntimeExecutable = Filename();
while (std::getline(fs, Line)) {
uint64_t Begin, End;
char Filename[255];
if (sscanf(Line.c_str(), "%lx-%lx %*c%*c%*c%*c %*x %*x:%*x %*d%s", &Begin, &End, Filename) == 3) {
if (RuntimeExecutable == Filename) {
auto str = fmt::format("Text={:x};Data={:x};Bss={:x}", Begin, Begin, Begin);
return {std::move(str), HandledPacketType::TYPE_ACK};
std::ostringstream ss;
ss << "Text=" << std::hex << Begin << ";Data=" << std::hex << Begin << ";Bss=" << std::hex << Begin;
ss << std::flush;
return {ss.str(), HandledPacketType::TYPE_ACK};
}
}
}
fs.close();
return {"Text=0;Data=0;Bss=0", HandledPacketType::TYPE_ACK};
}
GdbServer::HandledPacketType GdbServer::handleMemory(const std::string &packet) {
GdbServer::HandledPacketType GdbServer::handleMemory(std::string &packet) {
bool write;
size_t addr;
size_t length;
@@ -698,84 +634,14 @@ GdbServer::HandledPacketType GdbServer::handleMemory(const std::string &packet)
}
GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
const auto MatchStr = [](const std::string &Str, const char *str) -> bool { return Str.rfind(str, 0) == 0; };
GdbServer::HandledPacketType GdbServer::handleQuery(std::string &packet) {
auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
const auto split = [](const std::string &Str, char deliminator) -> std::vector<std::string> {
std::vector<std::string> Elements;
std::istringstream Input(Str);
for (std::string line;
std::getline(Input, line);
Elements.emplace_back(line));
return Elements;
};
if (match("QNonStop:")) {
auto ss = std::istringstream(packet);
ss.seekg(std::string("QNonStop:").size());
ss.get(); // discard colon
ss >> NonStopMode;
return {"OK", HandledPacketType::TYPE_ACK};
}
if (match("qSupported:")) {
// eg: qSupported:multiprocess+;swbreak+;hwbreak+;qRelocInsn+;fork-events+;vfork-events+;exec-events+;vContSupported+;QThreadEvents+;no-resumed+;memory-tagging+;xmlRegisters=i386
auto Features = split(packet.substr(strlen("qSupported:")), ';');
// For feature documentation
// https://sourceware.org/gdb/current/onlinedocs/gdb/General-Query-Packets.html#qSupported
std::string SupportedFeatures{};
// Required features
SupportedFeatures += "PacketSize=5000;";
SupportedFeatures += "xmlRegisters=i386;";
// XXX: Not yet supported, would be easy
// SupportedFeatures += "qXfer:auxv-file:read+";
SupportedFeatures += "qXfer:exec-file:read+;";
SupportedFeatures += "qXfer:features:read+;";
// XXX: Requires parsing the ELF and watching the library list
// SupportedFeatures += "qXfer:libraries:read+;";
SupportedFeatures += "qXfer:memory-map:read+;";
SupportedFeatures += "qXfer:siginfo:read+;";
SupportedFeatures += "qXfer:siginfo:write+;";
// XXX: Allowing this causes GDB to crash
SupportedFeatures += "qXfer:threads:read+;";
// QCatchSignals
// QPassSignals
SupportedFeatures += "QNonStop+;";
SupportedFeatures += "qXfer:osdata:read+;";
// Causes GDB to crash?
// SupportedFeatures += "QStartNoAckMode+;";
for (auto &Feature : Features) {
if (MatchStr(Feature, "swbreak+")) {
SupportedFeatures += "swbreak+;";
}
if (MatchStr(Feature, "hwbreak+")) {
SupportedFeatures += "hwbreak+;";
}
if (MatchStr(Feature, "vContSupported+")) {
SupportedFeatures += "vContSupported+;";
}
// Unsupported:
// multiprocess
// qRelocInsn
// fork-events
// vfork-events
// exec-events
// QThreadEvents
// no-resumed
// memory-tagging
}
return {SupportedFeatures, HandledPacketType::TYPE_ACK};
if (match("qSupported")) {
return {"PacketSize=5000;xmlRegisters=i386;qXfer:exec-file:read+;qXfer:features:read+;", HandledPacketType::TYPE_ACK};
}
if (match("qAttached")) {
return {"tnotrun:0", HandledPacketType::TYPE_ACK}; // We don't currently support launching executables from gdb.
return {"1", HandledPacketType::TYPE_ACK}; // We don't currently support launching executables from gdb.
}
if (match("qXfer")) {
return handleXfer(packet);
@@ -794,10 +660,7 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
ss << "m";
for (size_t i = 0; i < Threads->size(); ++i) {
auto Thread = Threads->at(i);
ss << std::hex << Thread->ThreadManager.TID;
if (i != (Threads->size() - 1)) {
ss << ",";
}
ss << std::hex << Thread->ThreadManager.TID << ",";
}
return {ss.str(), HandledPacketType::TYPE_ACK};
}
@@ -830,32 +693,8 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
return {"", HandledPacketType::TYPE_UNKNOWN};
}
GdbServer::HandledPacketType GdbServer::ThreadAction(char action, uint32_t tid) {
switch (action) {
case 'c': {
CTX->Run();
CTX->WaitForThreadsToRun();
return {"", HandledPacketType::TYPE_ONLYACK};
}
case 's': {
CTX->Step();
SendPacketPair({"OK", HandledPacketType::TYPE_ACK});
auto str = fmt::format("T05thread:{:02x};core:2c;", getpid());
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
return {"OK", HandledPacketType::TYPE_ACK};
}
case 't':
// This thread isn't part of the thread pool
CTX->Stop(false /* Ignore current thread */);
return {"OK", HandledPacketType::TYPE_ACK};
default:
return {"E00", HandledPacketType::TYPE_ACK};
}
}
GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
const auto match = [&](const std::string& str) -> std::optional<std::istringstream> {
GdbServer::HandledPacketType GdbServer::handleV(std::string& packet) {
auto match = [&](std::string str) -> std::optional<std::istringstream> {
if (packet.rfind(str, 0) == 0) {
auto ss = std::istringstream(packet);
ss.seekg(str.size());
@@ -864,11 +703,18 @@ GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
return std::nullopt;
};
const auto F = [](int result) { return fmt::format("F{:x}", result); };
const auto F_error = [] { return fmt::format("F-1,{:x}", errno); };
const auto F_data = [](int result, const std::string& data) {
return fmt::format("F{:x};{}", result, data);
};
auto F = [](int result) {
std::ostringstream ss;
ss << "F" << std::hex << result;
return ss.str(); };
auto F_error = [&]() {
std::ostringstream ss;
ss << "F-1," << std::hex << errno;
return ss.str(); };
auto F_data = [&](int result, std::string data) {
std::ostringstream ss;
ss << "F" << std::hex << result << ";" << data;
return ss.str(); };
std::optional<std::istringstream> ss;
if((ss = match("vFile:open:"))) {
@@ -890,11 +736,11 @@ GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
return {F(pid == 0 ? 0 : -1), HandledPacketType::TYPE_ACK}; // Only support the common filesystem
}
if((ss = match("vFile:close:"))) {
int fd;
*ss >> std::hex >> fd;
close(fd);
return {F(0), HandledPacketType::TYPE_ACK};
}
int fd;
*ss >> std::hex >> fd;
close(fd);
return {F(0), HandledPacketType::TYPE_ACK};
}
if((ss = match("vFile:pread:"))) {
int fd, count, offset;
@@ -916,31 +762,52 @@ GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
return {F_data(ret, data), HandledPacketType::TYPE_ACK};
}
if ((ss = match("vCont?"))) {
return {"vCont;c;t;s;r", HandledPacketType::TYPE_ACK}; // We support continue, step and terminate
// FIXME: We also claim to support continue with signal... because it's compulsory
return {"vCont;c;C;t;s;S;r", HandledPacketType::TYPE_ACK}; // We support continue, step and terminate
// FIXME: We also claim to support continue with signal... because it's compulsory
}
if ((ss = match("vCont;"))) {
char action{};
int thread{};
char action;
int thread;
action = ss->get();
action = ss->get();
if (ss->peek() == ':') {
ss->get();
*ss >> std::hex >> thread;
}
if (ss->peek() == ':') {
ss->get();
*ss >> std::hex >> thread;
}
if (ss->fail()) {
return {"E00", HandledPacketType::TYPE_ACK};
}
if (ss->fail()) {
return {"E00", HandledPacketType::TYPE_ACK};
}
switch (action) {
case 'c': {
CTX->Run();
return {"", HandledPacketType::TYPE_ONLYACK};
}
case 's': {
CTX->Step();
SendPacketPair({"OK", HandledPacketType::TYPE_ACK});
std::ostringstream ss;
ss << "T05thread:" << std::setfill('0') << std::setw(2) << std::hex << getpid() << ";core:2c;";
SendPacketPair({ss.str(), HandledPacketType::TYPE_ACK});
return {"OK", HandledPacketType::TYPE_ACK};
}
case 't':
// This thread isn't part of the thread pool
CTX->Stop(false /* Ignore current thread */);
return {"OK", HandledPacketType::TYPE_ACK};
default:
return {"E00", HandledPacketType::TYPE_ACK};
}
return ThreadAction(action, thread);
}
return {"", HandledPacketType::TYPE_ACK};
return {"", HandledPacketType::TYPE_ACK};
}
GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet) {
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
GdbServer::HandledPacketType GdbServer::handleThreadOp(std::string &packet) {
auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
if (match("Hc")) {
// Sets thread to this ID for stepping
@@ -956,7 +823,7 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet
if (match("Hg")) {
// Sets thread for "other" operations
auto ss = std::istringstream(packet);
ss.seekg(std::string_view("Hg").size());
ss.seekg(std::string("Hg").size());
ss >> std::hex >> CurrentDebuggingThread;
// This must return quick otherwise IDA complains
@@ -967,7 +834,7 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet
return {"", HandledPacketType::TYPE_UNKNOWN};
}
GdbServer::HandledPacketType GdbServer::handleBreakpoint(const std::string &packet) {
GdbServer::HandledPacketType GdbServer::handleBreakpoint(std::string &packet) {
auto ss = std::istringstream(packet);
bool Set{};
@@ -983,30 +850,19 @@ GdbServer::HandledPacketType GdbServer::handleBreakpoint(const std::string &pack
return {"OK", HandledPacketType::TYPE_ACK};
}
GdbServer::HandledPacketType GdbServer::ProcessPacket(const std::string &packet) {
GdbServer::HandledPacketType GdbServer::ProcessPacket(std::string &packet) {
switch (packet[0]) {
case '?': {
// Indicates the reason that the thread has stopped
// Behaviour changes if the target is in non-stop mode
// Binja doesn't support S response here
auto str = fmt::format("T00thread:{:x};", getpid());
return {std::move(str), HandledPacketType::TYPE_ACK};
//return {"S00", HandledPacketType::TYPE_ACK};
std::ostringstream ss;
ss << "T00thread:" << std::setfill('0') << std::setw(2) << std::hex << getpid() << ";core:2c;";
return {ss.str(), HandledPacketType::TYPE_ACK};
}
case 'c':
// Continue
CTX->Run();
CTX->WaitForThreadsToRun();
return {"OK", HandledPacketType::TYPE_ACK};
case 'D':
// Detach
// Ensure the threads are back in running state on detach
CTX->Run();
CTX->WaitForThreadsToRun();
return {"OK", HandledPacketType::TYPE_ACK};
case 'g':
// We might be running while we try reading
// Pause up front
CTX->Pause();
return {readRegs(), HandledPacketType::TYPE_ACK};
case 'p':
return readReg(packet);
@@ -1023,8 +879,6 @@ GdbServer::HandledPacketType GdbServer::ProcessPacket(const std::string &packet)
case '!': // Enable extended mode
case 'T': // Is a thread alive?
return {"OK", HandledPacketType::TYPE_ACK};
case 's': // Step
return ThreadAction('s', 0);
case 'Z': // Inserts breakpoint or watchpoint
return handleBreakpoint(packet);
case 'k': // Kill the process
@@ -1036,14 +890,14 @@ GdbServer::HandledPacketType GdbServer::ProcessPacket(const std::string &packet)
}
}
void GdbServer::SendPacketPair(const HandledPacketType& response) {
void GdbServer::SendPacketPair(HandledPacketType response) {
std::lock_guard lk(sendMutex);
if (response.TypeResponse == HandledPacketType::TYPE_ACK ||
response.TypeResponse == HandledPacketType::TYPE_ONLYACK) {
SendACK(*CommsStream, false);
}
else if (response.TypeResponse == HandledPacketType::TYPE_NACK ||
response.TypeResponse == HandledPacketType::TYPE_ONLYNACK) {
response.TypeResponse == HandledPacketType::TYPE_ONLYNACK) {
SendACK(*CommsStream, true);
}
@@ -1051,15 +905,13 @@ void GdbServer::SendPacketPair(const HandledPacketType& response) {
SendPacket(*CommsStream, "");
}
else if (response.TypeResponse != HandledPacketType::TYPE_ONLYNACK &&
response.TypeResponse != HandledPacketType::TYPE_ONLYACK &&
response.TypeResponse != HandledPacketType::TYPE_NONE) {
response.TypeResponse != HandledPacketType::TYPE_ONLYACK &&
response.TypeResponse != HandledPacketType::TYPE_NONE) {
SendPacket(*CommsStream, response.Response);
}
}
void GdbServer::GdbServerLoop() {
OpenListenSocket();
while (!CTX->CoreShuttingDown.load()) {
CommsStream = OpenSocket();
@@ -1075,7 +927,7 @@ void GdbServer::GdbServerLoop() {
response = ProcessPacket(packet);
SendPacketPair(response);
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
LogMan::Msg::DFmt("Unknown packet {}", packet);
LogMan::Msg::D("Unknown packet %s", packet.c_str());
}
break;
}
@@ -1091,22 +943,21 @@ void GdbServer::GdbServerLoop() {
break;
case '\x03': { // ASCII EOT
CTX->Pause();
auto str = fmt::format("T02thread:{:02x};core:2c;", getpid());
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
std::ostringstream ss;
ss << "T02thread:" << std::setfill('0') << std::setw(2) << std::hex << getpid() << ";core:2c;";
SendPacketPair({ss.str(), HandledPacketType::TYPE_ACK});
break;
}
default:
LogMan::Msg::DFmt("GdbServer: Unexpected byte {} ({:02x})", static_cast<char>(c), c);
LogMan::Msg::D("GdbServer: Unexpected byte %c (%02x)", c, c);
}
}
{
std::lock_guard lk(sendMutex);
CommsStream.reset();
CommsStream.release();
}
}
close(ListenSocket);
}
static void* ThreadHandler(void *Arg) {
FEXCore::GdbServer *This = reinterpret_cast<FEXCore::GdbServer*>(Arg);
@@ -1115,52 +966,49 @@ static void* ThreadHandler(void *Arg) {
}
void GdbServer::StartThread() {
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
gdbServerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
FEXCore::Threads::SetSignalMask(OldMask);
}
void GdbServer::OpenListenSocket() {
// open socket
struct addrinfo hints, *res;
memset(&hints, 0, sizeof(hints));
hints.ai_family = AF_UNSPEC;
hints.ai_socktype = SOCK_STREAM;
hints.ai_flags = AI_PASSIVE;
if(getaddrinfo(NULL, "8086", &hints, &res) < 0) {
perror("getaddrinfo");
}
int on = 1;
ListenSocket = socket(res->ai_family, res->ai_socktype, res->ai_protocol);
if (ListenSocket < 0) {
perror("socket");
}
if(setsockopt(ListenSocket, SOL_SOCKET, SO_REUSEADDR, (char*)&on, sizeof(on)) < 0) {
perror("setsockopt");
close(ListenSocket);
}
if (bind(ListenSocket, res->ai_addr, res->ai_addrlen) < 0) {
perror("bind");
close(ListenSocket);
}
listen(ListenSocket, 1);
gdbServerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
}
std::unique_ptr<std::iostream> GdbServer::OpenSocket() {
// Block until a connection arrives
struct sockaddr_storage their_addr;
socklen_t addr_size;
// open socket
int sockfd, new_fd;
LogMan::Msg::IFmt("GdbServer, waiting for connection on localhost:8086");
int new_fd = accept(ListenSocket, (struct sockaddr *)&their_addr, &addr_size);
struct addrinfo hints, *res;
struct sockaddr_storage their_addr;
socklen_t addr_size;
return std::make_unique<FEXCore::Utils::NetStream>(new_fd);
memset(&hints, 0, sizeof(hints));
hints.ai_family = AF_UNSPEC;
hints.ai_socktype = SOCK_STREAM;
hints.ai_flags = AI_PASSIVE;
if(getaddrinfo(NULL, "8086", &hints, &res) < 0) {
perror("getaddrinfo");
}
int on = 1;
sockfd = socket(res->ai_family, res->ai_socktype, res->ai_protocol);
if (sockfd < 0) {
perror("socket");
}
if(setsockopt(sockfd, SOL_SOCKET, SO_REUSEADDR, (char*)&on, sizeof(on)) < 0) {
perror("setsockopt");
}
if (bind(sockfd, res->ai_addr, res->ai_addrlen) < 0) {
perror("bind");
}
// Block until a connection arrives
LogMan::Msg::I("GdbServer, waiting for connection on localhost:8086");
listen(sockfd, 1);
new_fd = accept(sockfd, (struct sockaddr *)&their_addr, &addr_size);
return std::make_unique<NetStream>(new_fd);
}
+16 -27
View File
@@ -5,21 +5,18 @@ $end_info$
*/
#pragma once
#include <FEXCore/Config/Config.h>
#include <mutex>
#include <thread>
#include "Interface/Context/Context.h"
#include "Common/NetStream.h"
#include <FEXCore/Utils/Threads.h>
#include <istream>
#include <memory>
#include <mutex>
#include <stdint.h>
#include <string>
namespace FEXCore {
namespace Context {
struct Context;
}
class GdbServer {
public:
GdbServer(FEXCore::Context::Context *ctx);
@@ -30,11 +27,10 @@ public:
private:
void Break(int signal);
void OpenListenSocket();
std::unique_ptr<std::iostream> OpenSocket();
void StartThread();
std::string ReadPacket(std::iostream &stream);
void SendPacket(std::ostream &stream, const std::string& packet);
void SendPacket(std::ostream &stream, std::string packet);
void SendACK(std::ostream &stream, bool NACK);
@@ -51,20 +47,18 @@ private:
ResponseType TypeResponse{};
};
void SendPacketPair(const HandledPacketType& packetPair);
HandledPacketType ProcessPacket(const std::string &packet);
HandledPacketType handleQuery(const std::string &packet);
HandledPacketType handleXfer(const std::string &packet);
HandledPacketType handleMemory(const std::string &packet);
HandledPacketType handleV(const std::string& packet);
HandledPacketType handleThreadOp(const std::string &packet);
HandledPacketType handleBreakpoint(const std::string &packet);
void SendPacketPair(HandledPacketType packetPair);
HandledPacketType ProcessPacket(std::string &packet);
HandledPacketType handleQuery(std::string &packet);
HandledPacketType handleXfer(std::string &packet);
HandledPacketType handleMemory(std::string &packet);
HandledPacketType handleV(std::string& packet);
HandledPacketType handleThreadOp(std::string &packet);
HandledPacketType handleBreakpoint(std::string &packet);
HandledPacketType handleProgramOffsets();
HandledPacketType ThreadAction(char action, uint32_t tid);
std::string readRegs();
HandledPacketType readReg(const std::string& packet);
HandledPacketType readReg(std::string& packet);
FEXCore::Context::Context *CTX;
std::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
@@ -72,13 +66,8 @@ private:
std::mutex sendMutex;
bool SettingNoAckMode{false};
bool NoAckMode{false};
bool NonStopMode{false};
std::string ThreadString{};
std::string MemoryMapString{};
std::string OSDataString{};
uint32_t CurrentDebuggingThread{};
int ListenSocket{};
FEX_CONFIG_OPT(Filename, APP_FILENAME);
};
@@ -1,977 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <FEXCore/Utils/BitUtils.h>
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(TruncElementPair) {
auto Op = IROp->C<IR::IROp_TruncElementPair>();
switch (Op->Size) {
case 4: {
uint64_t *Src = GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Result{};
Result = Src[0] & ~0U;
Result |= Src[1] << 32;
GD = Result;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", Op->Size); break;
}
}
DEF_OP(Constant) {
auto Op = IROp->C<IR::IROp_Constant>();
GD = Op->Constant;
}
DEF_OP(EntrypointOffset) {
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
GD = Data->CurrentEntry + Op->Offset;
}
DEF_OP(InlineConstant) {
//nop
}
DEF_OP(InlineEntrypointOffset) {
//nop
}
DEF_OP(CycleCounter) {
#ifdef DEBUG_CYCLES
GD = 0;
#else
timespec time;
clock_gettime(CLOCK_REALTIME, &time);
GD = time.tv_nsec + time.tv_sec * 1000000000;
#endif
}
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
DEF_OP(Add) {
auto Op = IROp->C<IR::IROp_Add>();
uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
auto Func = [](auto a, auto b) { return a + b; };
switch (OpSize) {
DO_OP(4, uint32_t, Func)
DO_OP(8, uint64_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown size: {}", OpSize); break;
}
}
DEF_OP(Sub) {
auto Op = IROp->C<IR::IROp_Sub>();
uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
auto Func = [](auto a, auto b) { return a - b; };
switch (OpSize) {
DO_OP(4, uint32_t, Func)
DO_OP(8, uint64_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown size: {}", OpSize); break;
}
}
DEF_OP(Neg) {
auto Op = IROp->C<IR::IROp_Neg>();
uint8_t OpSize = IROp->Size;
uint64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
switch (OpSize) {
case 4:
GD = -static_cast<int32_t>(Src);
break;
case 8:
GD = -static_cast<int64_t>(Src);
break;
default: LOGMAN_MSG_A_FMT("Unknown NEG Size: {}\n", OpSize); break;
}
}
DEF_OP(Mul) {
auto Op = IROp->C<IR::IROp_Mul>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
switch (OpSize) {
case 4:
GD = static_cast<int64_t>(static_cast<int32_t>(Src1)) * static_cast<int64_t>(static_cast<int32_t>(Src2));
break;
case 8:
GD = static_cast<int64_t>(Src1) * static_cast<int64_t>(Src2);
break;
case 16: {
__int128_t Tmp = static_cast<__int128_t>(static_cast<int64_t>(Src1)) * static_cast<__int128_t>(static_cast<int64_t>(Src2));
memcpy(GDP, &Tmp, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Mul Size: {}\n", OpSize); break;
}
}
DEF_OP(UMul) {
auto Op = IROp->C<IR::IROp_UMul>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
switch (OpSize) {
case 4:
GD = static_cast<uint32_t>(Src1) * static_cast<uint32_t>(Src2);
break;
case 8:
GD = static_cast<uint64_t>(Src1) * static_cast<uint64_t>(Src2);
break;
case 16: {
__uint128_t Tmp = static_cast<__uint128_t>(static_cast<uint64_t>(Src1)) * static_cast<__uint128_t>(static_cast<uint64_t>(Src2));
memcpy(GDP, &Tmp, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown UMul Size: {}\n", OpSize); break;
}
}
DEF_OP(Div) {
auto Op = IROp->C<IR::IROp_Div>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
switch (OpSize) {
case 1:
GD = static_cast<int64_t>(static_cast<int8_t>(Src1)) / static_cast<int64_t>(static_cast<int8_t>(Src2));
break;
case 2:
GD = static_cast<int64_t>(static_cast<int16_t>(Src1)) / static_cast<int64_t>(static_cast<int16_t>(Src2));
break;
case 4:
GD = static_cast<int64_t>(static_cast<int32_t>(Src1)) / static_cast<int64_t>(static_cast<int32_t>(Src2));
break;
case 8:
GD = static_cast<int64_t>(Src1) / static_cast<int64_t>(Src2);
break;
case 16: {
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[0]) / *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[1]);
memcpy(GDP, &Tmp, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Mul Size: {}\n", OpSize); break;
}
}
DEF_OP(UDiv) {
auto Op = IROp->C<IR::IROp_UDiv>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
switch (OpSize) {
case 1:
GD = static_cast<uint64_t>(static_cast<uint8_t>(Src1)) / static_cast<uint64_t>(static_cast<uint8_t>(Src2));
break;
case 2:
GD = static_cast<uint64_t>(static_cast<uint16_t>(Src1)) / static_cast<uint64_t>(static_cast<uint16_t>(Src2));
break;
case 4:
GD = static_cast<uint64_t>(static_cast<uint32_t>(Src1)) / static_cast<uint64_t>(static_cast<uint32_t>(Src2));
break;
case 8:
GD = static_cast<uint64_t>(Src1) / static_cast<uint64_t>(Src2);
break;
case 16: {
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]) / *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
memcpy(GDP, &Tmp, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Mul Size: {}\n", OpSize); break;
}
}
DEF_OP(Rem) {
auto Op = IROp->C<IR::IROp_Rem>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
switch (OpSize) {
case 1:
GD = static_cast<int64_t>(static_cast<int8_t>(Src1)) % static_cast<int64_t>(static_cast<int8_t>(Src2));
break;
case 2:
GD = static_cast<int64_t>(static_cast<int16_t>(Src1)) % static_cast<int64_t>(static_cast<int16_t>(Src2));
break;
case 4:
GD = static_cast<int64_t>(static_cast<int32_t>(Src1)) % static_cast<int64_t>(static_cast<int32_t>(Src2));
break;
case 8:
GD = static_cast<int64_t>(Src1) % static_cast<int64_t>(Src2);
break;
case 16: {
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[0]) % *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[1]);
memcpy(GDP, &Tmp, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Mul Size: {}\n", OpSize); break;
}
}
DEF_OP(URem) {
auto Op = IROp->C<IR::IROp_URem>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
switch (OpSize) {
case 1:
GD = static_cast<uint64_t>(static_cast<uint8_t>(Src1)) % static_cast<uint64_t>(static_cast<uint8_t>(Src2));
break;
case 2:
GD = static_cast<uint64_t>(static_cast<uint16_t>(Src1)) % static_cast<uint64_t>(static_cast<uint16_t>(Src2));
break;
case 4:
GD = static_cast<uint64_t>(static_cast<uint32_t>(Src1)) % static_cast<uint64_t>(static_cast<uint32_t>(Src2));
break;
case 8:
GD = static_cast<uint64_t>(Src1) % static_cast<uint64_t>(Src2);
break;
case 16: {
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]) % *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
memcpy(GDP, &Tmp, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Mul Size: {}\n", OpSize); break;
}
}
DEF_OP(MulH) {
auto Op = IROp->C<IR::IROp_MulH>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
switch (OpSize) {
case 4: {
int64_t Tmp = static_cast<int64_t>(static_cast<int32_t>(Src1)) * static_cast<int64_t>(static_cast<int32_t>(Src2));
GD = Tmp >> 32;
break;
}
case 8: {
__int128_t Tmp = static_cast<__int128_t>(static_cast<int64_t>(Src1)) * static_cast<__int128_t>(static_cast<int64_t>(Src2));
GD = Tmp >> 64;
break;
}
default: LOGMAN_MSG_A_FMT("Unknown MulH Size: {}\n", OpSize); break;
}
}
DEF_OP(UMulH) {
auto Op = IROp->C<IR::IROp_UMulH>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
switch (OpSize) {
case 4:
GD = static_cast<uint64_t>(Src1) * static_cast<uint64_t>(Src2);
GD >>= 32;
break;
case 8: {
__uint128_t Tmp = static_cast<__uint128_t>(Src1) * static_cast<__uint128_t>(Src2);
GD = Tmp >> 64;
break;
}
case 16: {
// XXX: This is incorrect
__uint128_t Tmp = static_cast<__uint128_t>(Src1) * static_cast<__uint128_t>(Src2);
GD = Tmp >> 64;
break;
}
default: LOGMAN_MSG_A_FMT("Unknown UMulH Size: {}\n", OpSize); break;
}
}
DEF_OP(Or) {
auto Op = IROp->C<IR::IROp_Or>();
uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
auto Func = [](auto a, auto b) { return a | b; };
switch (OpSize) {
DO_OP(1, uint8_t, Func)
DO_OP(2, uint16_t, Func)
DO_OP(4, uint32_t, Func)
DO_OP(8, uint64_t, Func)
DO_OP(16, __uint128_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown size: {}", OpSize); break;
}
}
DEF_OP(And) {
auto Op = IROp->C<IR::IROp_And>();
uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
auto Func = [](auto a, auto b) { return a & b; };
switch (OpSize) {
DO_OP(1, uint8_t, Func)
DO_OP(2, uint16_t, Func)
DO_OP(4, uint32_t, Func)
DO_OP(8, uint64_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown size: {}", OpSize); break;
}
}
DEF_OP(Andn) {
auto Op = IROp->C<IR::IROp_Andn>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
constexpr auto Func = [](auto a, auto b) {
using Type = decltype(a);
return static_cast<Type>(a & static_cast<Type>(~b));
};
switch (OpSize) {
DO_OP(1, uint8_t, Func)
DO_OP(2, uint16_t, Func)
DO_OP(4, uint32_t, Func)
DO_OP(8, uint64_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown size: {}", OpSize); break;
}
}
DEF_OP(Xor) {
auto Op = IROp->C<IR::IROp_Xor>();
uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
auto Func = [](auto a, auto b) { return a ^ b; };
switch (OpSize) {
DO_OP(1, uint8_t, Func)
DO_OP(2, uint16_t, Func)
DO_OP(4, uint32_t, Func)
DO_OP(8, uint64_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown size: {}", OpSize); break;
}
}
DEF_OP(Lshl) {
auto Op = IROp->C<IR::IROp_Lshl>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Mask = OpSize * 8 - 1;
switch (OpSize) {
case 4:
GD = static_cast<uint32_t>(Src1) << (Src2 & Mask);
break;
case 8:
GD = static_cast<uint64_t>(Src1) << (Src2 & Mask);
break;
default: LOGMAN_MSG_A_FMT("Unknown LSHL Size: {}\n", OpSize); break;
}
}
DEF_OP(Lshr) {
auto Op = IROp->C<IR::IROp_Lshr>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Mask = OpSize * 8 - 1;
switch (OpSize) {
case 4:
GD = static_cast<uint32_t>(Src1) >> (Src2 & Mask);
break;
case 8:
GD = static_cast<uint64_t>(Src1) >> (Src2 & Mask);
break;
default: LOGMAN_MSG_A_FMT("Unknown LSHR Size: {}\n", OpSize); break;
}
}
DEF_OP(Ashr) {
auto Op = IROp->C<IR::IROp_Ashr>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Mask = OpSize * 8 - 1;
switch (OpSize) {
case 4:
GD = (uint32_t)(static_cast<int32_t>(Src1) >> (Src2 & Mask));
break;
case 8:
GD = (uint64_t)(static_cast<int64_t>(Src1) >> (Src2 & Mask));
break;
default: LOGMAN_MSG_A_FMT("Unknown ASHR Size: {}\n", OpSize); break;
}
}
DEF_OP(Ror) {
auto Op = IROp->C<IR::IROp_Ror>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
auto Ror = [] (auto In, auto R) {
auto RotateMask = sizeof(In) * 8 - 1;
R &= RotateMask;
return (In >> R) | (In << (sizeof(In) * 8 - R));
};
switch (OpSize) {
case 4:
GD = Ror(static_cast<uint32_t>(Src1), static_cast<uint32_t>(Src2));
break;
case 8: {
GD = Ror(static_cast<uint64_t>(Src1), static_cast<uint64_t>(Src2));
break;
}
default: LOGMAN_MSG_A_FMT("Unknown ROR Size: {}\n", OpSize); break;
}
}
DEF_OP(Extr) {
auto Op = IROp->C<IR::IROp_Extr>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
auto Extr = [] (auto Src1, auto Src2, uint8_t lsb) -> decltype(Src1) {
__uint128_t Result{};
Result = Src1;
Result <<= sizeof(Src1) * 8;
Result |= Src2;
Result >>= lsb;
return Result;
};
switch (OpSize) {
case 4:
GD = Extr(static_cast<uint32_t>(Src1), static_cast<uint32_t>(Src2), Op->LSB);
break;
case 8: {
GD = Extr(static_cast<uint64_t>(Src1), static_cast<uint64_t>(Src2), Op->LSB);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown EXTR Size: {}\n", OpSize); break;
}
}
DEF_OP(LDiv) {
auto Op = IROp->C<IR::IROp_LDiv>();
uint8_t OpSize = IROp->Size;
// Each source is OpSize in size
// So you can have up to a 128bit divide from x86-64
switch (OpSize) {
case 2: {
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
int32_t Res = Source / Divisor;
// We only store the lower bits of the result
GD = static_cast<int16_t>(Res);
break;
}
case 4: {
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
int64_t Res = Source / Divisor;
// We only store the lower bits of the result
GD = static_cast<int32_t>(Res);
break;
}
case 8: {
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[2]);
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
__int128_t Res = Source / Divisor;
// We only store the lower bits of the result
memcpy(GDP, &Res, OpSize);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LDIV Size: {}", OpSize); break;
}
}
DEF_OP(LUDiv) {
auto Op = IROp->C<IR::IROp_LUDiv>();
uint8_t OpSize = IROp->Size;
// Each source is OpSize in size
// So you can have up to a 128bit divide from x86-64
switch (OpSize) {
case 2: {
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
uint32_t Res = Source / Divisor;
// We only store the lower bits of the result
GD = static_cast<uint16_t>(Res);
break;
}
case 4: {
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
uint64_t Res = Source / Divisor;
// We only store the lower bits of the result
GD = static_cast<uint32_t>(Res);
break;
}
case 8: {
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
__uint128_t Res = Source / Divisor;
// We only store the lower bits of the result
memcpy(GDP, &Res, OpSize);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUDIV Size: {}", OpSize); break;
}
}
DEF_OP(LRem) {
auto Op = IROp->C<IR::IROp_LRem>();
uint8_t OpSize = IROp->Size;
// Each source is OpSize in size
// So you can have up to a 128bit Remainder from x86-64
switch (OpSize) {
case 2: {
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
int32_t Res = Source % Divisor;
// We only store the lower bits of the result
GD = static_cast<int16_t>(Res);
break;
}
case 4: {
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
int64_t Res = Source % Divisor;
// We only store the lower bits of the result
GD = static_cast<int32_t>(Res);
break;
}
case 8: {
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[2]);
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
__int128_t Res = Source % Divisor;
// We only store the lower bits of the result
memcpy(GDP, &Res, OpSize);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LREM Size: {}", OpSize); break;
}
}
DEF_OP(LURem) {
auto Op = IROp->C<IR::IROp_LURem>();
uint8_t OpSize = IROp->Size;
// Each source is OpSize in size
// So you can have up to a 128bit Remainder from x86-64
switch (OpSize) {
case 2: {
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
uint32_t Res = Source % Divisor;
// We only store the lower bits of the result
GD = static_cast<uint16_t>(Res);
break;
}
case 4: {
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
uint64_t Res = Source % Divisor;
// We only store the lower bits of the result
GD = static_cast<uint32_t>(Res);
break;
}
case 8: {
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
__uint128_t Res = Source % Divisor;
// We only store the lower bits of the result
memcpy(GDP, &Res, OpSize);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUREM Size: {}", OpSize); break;
}
}
DEF_OP(Not) {
auto Op = IROp->C<IR::IROp_Not>();
uint8_t OpSize = IROp->Size;
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
const uint64_t mask[9]= { 0, 0xFF, 0xFFFF, 0, 0xFFFFFFFF, 0, 0, 0, 0xFFFFFFFFFFFFFFFFULL };
uint64_t Mask = mask[OpSize];
GD = (~Src) & Mask;
}
DEF_OP(Popcount) {
auto Op = IROp->C<IR::IROp_Popcount>();
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
GD = std::popcount(Src);
}
DEF_OP(FindLSB) {
auto Op = IROp->C<IR::IROp_FindLSB>();
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Result = FindFirstSetBit(Src);
GD = Result - 1;
}
DEF_OP(FindMSB) {
auto Op = IROp->C<IR::IROp_FindMSB>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
case 1: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
case 2: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
case 4: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
case 8: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
default: LOGMAN_MSG_A_FMT("Unknown FindMSB size: {}", OpSize); break;
}
}
DEF_OP(FindTrailingZeros) {
auto Op = IROp->C<IR::IROp_FindTrailingZeros>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
case 1: {
auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
GD = std::countr_zero(Src);
break;
}
case 2: {
auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
GD = std::countr_zero(Src);
break;
}
case 4: {
auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
GD = std::countr_zero(Src);
break;
}
case 8: {
auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
GD = std::countr_zero(Src);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown size: {}", OpSize); break;
}
}
DEF_OP(CountLeadingZeroes) {
auto Op = IROp->C<IR::IROp_CountLeadingZeroes>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
case 1: {
auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
GD = std::countl_zero(Src);
break;
}
case 2: {
auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
GD = std::countl_zero(Src);
break;
}
case 4: {
auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
GD = std::countl_zero(Src);
break;
}
case 8: {
auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
GD = std::countl_zero(Src);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown size: {}", OpSize); break;
}
}
DEF_OP(Rev) {
auto Op = IROp->C<IR::IROp_Rev>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
case 2: GD = BSwap16(*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0])); break;
case 4: GD = BSwap32(*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0])); break;
case 8: GD = BSwap64(*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0])); break;
default: LOGMAN_MSG_A_FMT("Unknown REV size: {}", OpSize); break;
}
}
DEF_OP(Bfi) {
auto Op = IROp->C<IR::IROp_Bfi>();
uint64_t SourceMask = (1ULL << Op->Width) - 1;
if (Op->Width == 64)
SourceMask = ~0ULL;
uint64_t DestMask = ~(SourceMask << Op->lsb);
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Res = (Src1 & DestMask) | ((Src2 & SourceMask) << Op->lsb);
GD = Res;
}
DEF_OP(Bfe) {
auto Op = IROp->C<IR::IROp_Bfe>();
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_A_FMT(OpSize <= 8, "OpSize is too large for BFE: {}", OpSize);
uint64_t SourceMask = (1ULL << Op->Width) - 1;
if (Op->Width == 64)
SourceMask = ~0ULL;
SourceMask <<= Op->lsb;
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
GD = (Src & SourceMask) >> Op->lsb;
}
DEF_OP(Sbfe) {
auto Op = IROp->C<IR::IROp_Sbfe>();
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_A_FMT(OpSize <= 8, "OpSize is too large for SBFE: {}", OpSize);
int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t ShiftLeftAmount = (64 - (Op->Width + Op->lsb));
uint64_t ShiftRightAmount = ShiftLeftAmount + Op->lsb;
Src <<= ShiftLeftAmount;
Src >>= ShiftRightAmount;
GD = Src;
}
DEF_OP(Select) {
auto Op = IROp->C<IR::IROp_Select>();
uint8_t OpSize = IROp->Size;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t ArgTrue;
uint64_t ArgFalse;
if (OpSize == 4) {
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[3]);
} else {
ArgTrue = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
ArgFalse = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[3]);
}
bool CompResult;
if (Op->CompareSize == 4)
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
else
CompResult = IsConditionTrue<uint64_t, int64_t, double>(Op->Cond.Val, Src1, Src2);
GD = CompResult ? ArgTrue : ArgFalse;
}
DEF_OP(VExtractToGPR) {
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
uint8_t OpSize = IROp->Size;
uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Header.Args[0]);
LOGMAN_THROW_A_FMT(OpSize <= 16, "OpSize is too large for VExtractToGPR: {}", OpSize);
if (SourceSize == 16) {
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
uint64_t Shift = Op->Header.ElementSize * Op->Idx * 8;
if (Op->Header.ElementSize == 8)
SourceMask = ~0ULL;
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
Src >>= Shift;
Src &= SourceMask;
memcpy(GDP, &Src, Op->Header.ElementSize);
}
else {
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
uint64_t Shift = Op->Header.ElementSize * Op->Idx * 8;
if (Op->Header.ElementSize == 8)
SourceMask = ~0ULL;
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
Src >>= Shift;
Src &= SourceMask;
GD = Src;
}
}
DEF_OP(Float_ToGPR_ZS) {
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0804: { // int64_t <- float
int64_t Dst = (int64_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
memcpy(GDP, &Dst, IROp->Size);
break;
}
case 0x0808: { // int64_t <- double
int64_t Dst = (int64_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
memcpy(GDP, &Dst, IROp->Size);
break;
}
case 0x0404: { // int32_t <- float
int32_t Dst = (int32_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
memcpy(GDP, &Dst, IROp->Size);
break;
}
case 0x0408: { // int32_t <- double
int32_t Dst = (int32_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
memcpy(GDP, &Dst, IROp->Size);
break;
}
}
}
DEF_OP(Float_ToGPR_S) {
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0804: { // int64_t <- float
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
memcpy(GDP, &Dst, IROp->Size);
break;
}
case 0x0808: { // int64_t <- double
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
memcpy(GDP, &Dst, IROp->Size);
break;
}
case 0x0404: { // int32_t <- float
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
memcpy(GDP, &Dst, IROp->Size);
break;
}
case 0x0408: { // int32_t <- double
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
memcpy(GDP, &Dst, IROp->Size);
break;
}
}
}
DEF_OP(FCmp) {
auto Op = IROp->C<IR::IROp_FCmp>();
uint32_t ResultFlags{};
if (Op->ElementSize == 4) {
float Src1 = *GetSrc<float*>(Data->SSAData, Op->Header.Args[0]);
float Src2 = *GetSrc<float*>(Data->SSAData, Op->Header.Args[1]);
bool Unordered = std::isnan(Src1) || std::isnan(Src2);
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
if (Unordered || (Src1 < Src2)) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
}
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED)) {
if (Unordered) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
}
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ)) {
if (Unordered || (Src1 == Src2)) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
}
}
else {
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
bool Unordered = std::isnan(Src1) || std::isnan(Src2);
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
if (Unordered || (Src1 < Src2)) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
}
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED)) {
if (Unordered) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
}
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ)) {
if (Unordered || (Src1 == Src2)) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
}
}
GD = ResultFlags;
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,778 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <FEXCore/Utils/BitUtils.h>
#include <cstdint>
namespace FEXCore::CPU {
#ifdef _M_X86_64
uint8_t AtomicFetchNeg(uint8_t *Addr) {
using Type = uint8_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
uint16_t AtomicFetchNeg(uint16_t *Addr) {
using Type = uint16_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
uint32_t AtomicFetchNeg(uint32_t *Addr) {
using Type = uint32_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
uint64_t AtomicFetchNeg(uint64_t *Addr) {
using Type = uint64_t;
std::atomic<Type> *MemData = reinterpret_cast<std::atomic<Type>*>(Addr);
Type Expected = MemData->load();
Type Desired = -Expected;
do {
Desired = -Expected;
} while (!MemData->compare_exchange_strong(Expected, Desired, std::memory_order_seq_cst));
return Expected;
}
template<typename T>
T AtomicCompareAndSwap(T expected, T desired, T *addr)
{
std::atomic<T> *MemData = reinterpret_cast<std::atomic<T>*>(addr);
T Src1 = expected;
T Src2 = desired;
T Expected = Src1;
bool Result = MemData->compare_exchange_strong(Expected, Src2);
return Result ? Src1 : Expected;
}
template uint8_t AtomicCompareAndSwap<uint8_t>(uint8_t expected, uint8_t desired, uint8_t *addr);
template uint16_t AtomicCompareAndSwap<uint16_t>(uint16_t expected, uint16_t desired, uint16_t *addr);
template uint32_t AtomicCompareAndSwap<uint32_t>(uint32_t expected, uint32_t desired, uint32_t *addr);
template uint64_t AtomicCompareAndSwap<uint64_t>(uint64_t expected, uint64_t desired, uint64_t *addr);
#else
// Needs to match what the AArch64 JIT and unaligned signal handler expects
uint8_t AtomicFetchNeg(uint8_t *Addr) {
using Type = uint8_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxrb %w[Result], [%[Memory]];
neg %w[Tmp], %w[Result];
stlxrb %w[TmpStatus], %w[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
uint16_t AtomicFetchNeg(uint16_t *Addr) {
using Type = uint16_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxrh %w[Result], [%[Memory]];
neg %w[Tmp], %w[Result];
stlxrh %w[TmpStatus], %w[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
uint32_t AtomicFetchNeg(uint32_t *Addr) {
using Type = uint32_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxr %w[Result], [%[Memory]];
neg %w[Tmp], %w[Result];
stlxr %w[TmpStatus], %w[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
uint64_t AtomicFetchNeg(uint64_t *Addr) {
using Type = uint64_t;
Type Result{};
Type Tmp{};
Type TmpStatus{};
__asm__ volatile(
R"(
1:
ldaxr %[Result], [%[Memory]];
neg %[Tmp], %[Result];
stlxr %w[TmpStatus], %[Tmp], [%[Memory]];
cbnz %w[TmpStatus], 1b;
)"
: [Result] "=r" (Result)
, [Tmp] "=r" (Tmp)
, [TmpStatus] "=r" (TmpStatus)
, [Memory] "+r" (Addr)
:: "memory"
);
return Result;
}
template<>
uint8_t AtomicCompareAndSwap(uint8_t expected, uint8_t desired, uint8_t *addr) {
using Type = uint8_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxrb %w[Tmp], [%[Memory]];
cmp %w[Tmp], %w[Expected], uxtb;
b.ne 2f;
stlxrb %w[Tmp2], %w[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %w[Result], %w[Expected];
b 3f;
2:
mov %w[Result], %w[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
template<>
uint16_t AtomicCompareAndSwap(uint16_t expected, uint16_t desired, uint16_t *addr) {
using Type = uint16_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxrh %w[Tmp], [%[Memory]];
cmp %w[Tmp], %w[Expected], uxth;
b.ne 2f;
stlxrh %w[Tmp2], %w[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %w[Result], %w[Expected];
b 3f;
2:
mov %w[Result], %w[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
template<>
uint32_t AtomicCompareAndSwap(uint32_t expected, uint32_t desired, uint32_t *addr) {
using Type = uint32_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxr %w[Tmp], [%[Memory]];
cmp %w[Tmp], %w[Expected];
b.ne 2f;
stlxr %w[Tmp2], %w[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %w[Result], %w[Expected];
b 3f;
2:
mov %w[Result], %w[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
template<>
uint64_t AtomicCompareAndSwap(uint64_t expected, uint64_t desired, uint64_t *addr) {
using Type = uint64_t;
//force Result to r9 (scratch register) or clang spills to stack
register Type Result asm("r9"){};
Type Tmp{};
Type Tmp2{};
__asm__ volatile(
R"(
1:
ldaxr %[Tmp], [%[Memory]];
cmp %[Tmp], %[Expected];
b.ne 2f;
stlxr %w[Tmp2], %[Desired], [%[Memory]];
cbnz %w[Tmp2], 1b;
mov %[Result], %[Expected];
b 3f;
2:
mov %[Result], %[Tmp];
clrex;
3:
)"
: [Tmp] "=r" (Tmp)
, [Tmp2] "=r" (Tmp2)
, [Desired] "+r" (desired)
, [Expected] "+r" (expected)
, [Result] "=r" (Result)
, [Memory] "+r" (addr)
:: "memory"
);
return Result;
}
#endif
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CASPair>();
uint8_t OpSize = IROp->Size;
// Size is the size of each pair element
switch (OpSize) {
case 4: {
GD = AtomicCompareAndSwap(
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint64_t**>(Data->SSAData, Op->Header.Args[2])
);
break;
}
case 8: {
std::atomic<__uint128_t> *MemData = *GetSrc<std::atomic<__uint128_t> **>(Data->SSAData, Op->Header.Args[2]);
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
__uint128_t Expected = Src1;
bool Result = MemData->compare_exchange_strong(Expected, Src2);
memcpy(GDP, Result ? &Src1 : &Expected, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", OpSize); break;
}
}
DEF_OP(CAS) {
auto Op = IROp->C<IR::IROp_CAS>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
case 1: {
GD = AtomicCompareAndSwap(
*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint8_t**>(Data->SSAData, Op->Header.Args[2])
);
break;
}
case 2: {
GD = AtomicCompareAndSwap(
*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint16_t**>(Data->SSAData, Op->Header.Args[2])
);
break;
}
case 4: {
GD = AtomicCompareAndSwap(
*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint32_t**>(Data->SSAData, Op->Header.Args[2])
);
break;
}
case 8: {
GD = AtomicCompareAndSwap(
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint64_t**>(Data->SSAData, Op->Header.Args[2])
);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", OpSize); break;
}
}
DEF_OP(AtomicAdd) {
auto Op = IROp->C<IR::IROp_AtomicAdd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData += Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData += Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData += Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData += Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicSub) {
auto Op = IROp->C<IR::IROp_AtomicSub>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData -= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData -= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData -= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData -= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicAnd) {
auto Op = IROp->C<IR::IROp_AtomicAnd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData &= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData &= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData &= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData &= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicOr) {
auto Op = IROp->C<IR::IROp_AtomicOr>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData |= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData |= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData |= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData |= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicXor) {
auto Op = IROp->C<IR::IROp_AtomicXor>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData ^= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData ^= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData ^= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
*MemData ^= Src;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicSwap) {
auto Op = IROp->C<IR::IROp_AtomicSwap>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
uint16_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
uint32_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchAdd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
uint16_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
uint32_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchSub) {
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
uint16_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
uint32_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchAnd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
uint16_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
uint32_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
uint16_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
uint32_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchXor) {
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
uint16_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
uint32_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchNeg) {
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
switch (IROp->Size) {
case 1: {
using Type = uint8_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
break;
}
case 2: {
using Type = uint16_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
break;
}
case 4: {
using Type = uint32_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
break;
}
case 8: {
using Type = uint64_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,170 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include "Interface/HLE/Thunks/Thunks.h"
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <cstdint>
#include <unistd.h>
namespace FEXCore::CPU {
[[noreturn]]
static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
Thread->CTX->SignalThread(Thread, FEXCore::Core::SignalEvent::Return);
LOGMAN_MSG_A_FMT("unreachable");
FEX_UNREACHABLE;
}
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(GuestCallDirect) {
LogMan::Msg::DFmt("Unimplemented");
}
DEF_OP(GuestCallIndirect) {
LogMan::Msg::DFmt("Unimplemented");
}
DEF_OP(GuestReturn) {
LogMan::Msg::DFmt("Unimplemented");
}
DEF_OP(SignalReturn) {
SignalReturn(Data->State);
}
DEF_OP(CallbackReturn) {
Data->State->CTX->InterpreterCallbackReturn(Data->State, Data->StackEntry);
}
DEF_OP(ExitFunction) {
auto Op = IROp->C<IR::IROp_ExitFunction>();
uint8_t OpSize = IROp->Size;
uintptr_t* ContextPtr = reinterpret_cast<uintptr_t*>(Data->State->CurrentFrame);
void *ContextData = reinterpret_cast<void*>(ContextPtr);
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
memcpy(ContextData, Src, OpSize);
Data->BlockResults.Quit = true;
}
DEF_OP(Jump) {
auto Op = IROp->C<IR::IROp_Jump>();
uintptr_t ListBegin = Data->CurrentIR->GetListData();
uintptr_t DataBegin = Data->CurrentIR->GetData();
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->Header.Args[0]);
Data->BlockResults.Redo = true;
}
DEF_OP(CondJump) {
auto Op = IROp->C<IR::IROp_CondJump>();
uintptr_t ListBegin = Data->CurrentIR->GetListData();
uintptr_t DataBegin = Data->CurrentIR->GetData();
bool CompResult;
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
if (Op->CompareSize == 4)
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
else
CompResult = IsConditionTrue<uint64_t, int64_t, double>(Op->Cond.Val, Src1, Src2);
if (CompResult) {
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TrueBlock);
}
else {
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->FalseBlock);
}
Data->BlockResults.Redo = true;
}
DEF_OP(Syscall) {
auto Op = IROp->C<IR::IROp_Syscall>();
FEXCore::HLE::SyscallArguments Args;
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
if (Op->Header.Args[j].IsInvalid()) break;
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
}
uint64_t Res = FEXCore::Context::HandleSyscall(Data->State->CTX->SyscallHandler, Data->State->CurrentFrame, &Args);
GD = Res;
}
DEF_OP(InlineSyscall) {
auto Op = IROp->C<IR::IROp_InlineSyscall>();
FEXCore::HLE::SyscallArguments Args;
for (size_t j = 0; j < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++j) {
if (Op->Header.Args[j].IsInvalid()) break;
Args.Argument[j] = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[j]);
}
// We don't want the errno handling but I also don't want to write inline ASM atm
uint64_t Res = syscall(
Op->HostSyscallNumber,
Args.Argument[0],
Args.Argument[1],
Args.Argument[2],
Args.Argument[3],
Args.Argument[4],
Args.Argument[5],
Args.Argument[6]
);
if (Res == -1) {
Res = -errno;
}
GD = Res;
}
DEF_OP(Thunk) {
auto Op = IROp->C<IR::IROp_Thunk>();
auto thunkFn = Data->State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
thunkFn(*GetSrc<void**>(Data->SSAData, Op->Header.Args[0]));
}
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
auto CodePtr = Data->CurrentEntry + Op->Offset;
if (memcmp((void*)CodePtr, &Op->CodeOriginalLow, Op->CodeLength) != 0) {
GD = 1;
} else {
GD = 0;
}
}
DEF_OP(RemoveCodeEntry) {
Data->State->CTX->RemoveCodeEntry(Data->State, Data->CurrentEntry);
}
DEF_OP(CPUID) {
auto Op = IROp->C<IR::IROp_CPUID>();
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
auto Results = Data->State->CTX->CPUID.RunFunction(Arg, Leaf);
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,224 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(VInsGPR) {
auto Op = IROp->C<IR::IROp_VInsGPR>();
uint8_t OpSize = IROp->Size;
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Offset = Op->Index * Op->Header.ElementSize * 8;
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
if (Op->Header.ElementSize == 8) {
Mask = ~0ULL;
}
Src2 = Src2 & Mask;
Mask <<= Offset;
Mask = ~Mask;
__uint128_t Dst = Src1 & Mask;
Dst |= Src2 << Offset;
memcpy(GDP, &Dst, OpSize);
}
DEF_OP(VCastFromGPR) {
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), Op->Header.ElementSize);
}
DEF_OP(Float_FromGPR_S) {
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0404: { // Float <- int32_t
float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
case 0x0408: { // Float <- int64_t
float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
case 0x0804: { // Double <- int32_t
double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
case 0x0808: { // Double <- int64_t
double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
memcpy(GDP, &Dst, Op->Header.ElementSize);
break;
}
}
}
DEF_OP(Float_FToF) {
auto Op = IROp->C<IR::IROp_Float_FToF>();
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0804: { // Double <- Float
double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]);
memcpy(GDP, &Dst, 8);
break;
}
case 0x0408: { // Float <- Double
float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
memcpy(GDP, &Dst, 4);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
}
}
DEF_OP(Vector_SToF) {
auto Op = IROp->C<IR::IROp_Vector_SToF>();
uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
uint8_t Tmp[16]{};
uint8_t Elements = OpSize / Op->Header.ElementSize;
auto Func = [](auto a, auto min, auto max) { return a; };
switch (Op->Header.ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToZS) {
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
uint8_t Tmp[16]{};
uint8_t Elements = OpSize / Op->Header.ElementSize;
auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
switch (Op->Header.ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToS) {
auto Op = IROp->C<IR::IROp_Vector_FToS>();
uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
uint8_t Tmp[16]{};
uint8_t Elements = OpSize / Op->Header.ElementSize;
auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
switch (Op->Header.ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToF) {
auto Op = IROp->C<IR::IROp_Vector_FToF>();
uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
uint8_t Tmp[16]{};
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
auto Func = [](auto a, auto min, auto max) { return a; };
switch (Conv) {
case 0x0804: { // Double <- float
// Only the lower elements from the source
// This uses half the source elements
uint8_t Elements = OpSize / 8;
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(double, float, Func, 0, 0)
break;
}
case 0x0408: { // Float <- Double
// Little bit tricky here
// Sometimes is used to convert from a 128bit vector register
// in to a 64bit vector register with different sized elements
// eg: %ssa5 i32v2 = Vector_FToF %ssa4 i128, #0x8
uint8_t Elements = (OpSize << 1) / Op->SrcElementSize;
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv); break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToI) {
auto Op = IROp->C<IR::IROp_Vector_FToI>();
uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
uint8_t Tmp[16]{};
uint8_t Elements = OpSize / Op->Header.ElementSize;
auto Func_Nearest = [](auto a) { return std::rint(a); };
auto Func_Neg = [](auto a) { return std::floor(a); };
auto Func_Pos = [](auto a) { return std::ceil(a); };
auto Func_Trunc = [](auto a) { return std::trunc(a); };
auto Func_Host = [](auto a) { return std::rint(a); };
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (Op->Header.ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
}
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (Op->Header.ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
}
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (Op->Header.ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
}
break;
case FEXCore::IR::Round_Towards_Zero.Val:
switch (Op->Header.ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
}
break;
case FEXCore::IR::Round_Host.Val:
switch (Op->Header.ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Host)
DO_VECTOR_1SRC_OP(8, double, Func_Host)
}
break;
}
memcpy(GDP, Tmp, OpSize);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,434 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace AES {
static __uint128_t InvShiftRows(uint8_t *State) {
uint8_t Shifted[16] = {
State[0], State[13], State[10], State[7],
State[4], State[1], State[14], State[11],
State[8], State[5], State[2], State[15],
State[12], State[9], State[6], State[3],
};
__uint128_t Res{};
memcpy(&Res, Shifted, 16);
return Res;
}
static __uint128_t InvSubBytes(uint8_t *State) {
// 16x16 matrix table
static const uint8_t InvSubstitutionTable[256] = {
0x52, 0x09, 0x6a, 0xd5, 0x30, 0x36, 0xa5, 0x38, 0xbf, 0x40, 0xa3, 0x9e, 0x81, 0xf3, 0xd7, 0xfb,
0x7c, 0xe3, 0x39, 0x82, 0x9b, 0x2f, 0xff, 0x87, 0x34, 0x8e, 0x43, 0x44, 0xc4, 0xde, 0xe9, 0xcb,
0x54, 0x7b, 0x94, 0x32, 0xa6, 0xc2, 0x23, 0x3d, 0xee, 0x4c, 0x95, 0x0b, 0x42, 0xfa, 0xc3, 0x4e,
0x08, 0x2e, 0xa1, 0x66, 0x28, 0xd9, 0x24, 0xb2, 0x76, 0x5b, 0xa2, 0x49, 0x6d, 0x8b, 0xd1, 0x25,
0x72, 0xf8, 0xf6, 0x64, 0x86, 0x68, 0x98, 0x16, 0xd4, 0xa4, 0x5c, 0xcc, 0x5d, 0x65, 0xb6, 0x92,
0x6c, 0x70, 0x48, 0x50, 0xfd, 0xed, 0xb9, 0xda, 0x5e, 0x15, 0x46, 0x57, 0xa7, 0x8d, 0x9d, 0x84,
0x90, 0xd8, 0xab, 0x00, 0x8c, 0xbc, 0xd3, 0x0a, 0xf7, 0xe4, 0x58, 0x05, 0xb8, 0xb3, 0x45, 0x06,
0xd0, 0x2c, 0x1e, 0x8f, 0xca, 0x3f, 0x0f, 0x02, 0xc1, 0xaf, 0xbd, 0x03, 0x01, 0x13, 0x8a, 0x6b,
0x3a, 0x91, 0x11, 0x41, 0x4f, 0x67, 0xdc, 0xea, 0x97, 0xf2, 0xcf, 0xce, 0xf0, 0xb4, 0xe6, 0x73,
0x96, 0xac, 0x74, 0x22, 0xe7, 0xad, 0x35, 0x85, 0xe2, 0xf9, 0x37, 0xe8, 0x1c, 0x75, 0xdf, 0x6e,
0x47, 0xf1, 0x1a, 0x71, 0x1d, 0x29, 0xc5, 0x89, 0x6f, 0xb7, 0x62, 0x0e, 0xaa, 0x18, 0xbe, 0x1b,
0xfc, 0x56, 0x3e, 0x4b, 0xc6, 0xd2, 0x79, 0x20, 0x9a, 0xdb, 0xc0, 0xfe, 0x78, 0xcd, 0x5a, 0xf4,
0x1f, 0xdd, 0xa8, 0x33, 0x88, 0x07, 0xc7, 0x31, 0xb1, 0x12, 0x10, 0x59, 0x27, 0x80, 0xec, 0x5f,
0x60, 0x51, 0x7f, 0xa9, 0x19, 0xb5, 0x4a, 0x0d, 0x2d, 0xe5, 0x7a, 0x9f, 0x93, 0xc9, 0x9c, 0xef,
0xa0, 0xe0, 0x3b, 0x4d, 0xae, 0x2a, 0xf5, 0xb0, 0xc8, 0xeb, 0xbb, 0x3c, 0x83, 0x53, 0x99, 0x61,
0x17, 0x2b, 0x04, 0x7e, 0xba, 0x77, 0xd6, 0x26, 0xe1, 0x69, 0x14, 0x63, 0x55, 0x21, 0x0c, 0x7d,
};
// Uses a byte substitution table with a constant set of values
// Needs to do a table look up
uint8_t Substituted[16];
for (size_t i = 0; i < 16; ++i) {
Substituted[i] = InvSubstitutionTable[State[i]];
}
__uint128_t Res{};
memcpy(&Res, Substituted, 16);
return Res;
}
static __uint128_t ShiftRows(uint8_t *State) {
uint8_t Shifted[16] = {
State[0], State[5], State[10], State[15],
State[4], State[9], State[14], State[3],
State[8], State[13], State[2], State[7],
State[12], State[1], State[6], State[11],
};
__uint128_t Res{};
memcpy(&Res, Shifted, 16);
return Res;
}
static __uint128_t SubBytes(uint8_t *State, size_t Bytes) {
// 16x16 matrix table
static const uint8_t SubstitutionTable[256] = {
0x63, 0x7c, 0x77, 0x7b, 0xf2, 0x6b, 0x6f, 0xc5, 0x30, 0x01, 0x67, 0x2b, 0xfe, 0xd7, 0xab, 0x76,
0xca, 0x82, 0xc9, 0x7d, 0xfa, 0x59, 0x47, 0xf0, 0xad, 0xd4, 0xa2, 0xaf, 0x9c, 0xa4, 0x72, 0xc0,
0xb7, 0xfd, 0x93, 0x26, 0x36, 0x3f, 0xf7, 0xcc, 0x34, 0xa5, 0xe5, 0xf1, 0x71, 0xd8, 0x31, 0x15,
0x04, 0xc7, 0x23, 0xc3, 0x18, 0x96, 0x05, 0x9a, 0x07, 0x12, 0x80, 0xe2, 0xeb, 0x27, 0xb2, 0x75,
0x09, 0x83, 0x2c, 0x1a, 0x1b, 0x6e, 0x5a, 0xa0, 0x52, 0x3b, 0xd6, 0xb3, 0x29, 0xe3, 0x2f, 0x84,
0x53, 0xd1, 0x00, 0xed, 0x20, 0xfc, 0xb1, 0x5b, 0x6a, 0xcb, 0xbe, 0x39, 0x4a, 0x4c, 0x58, 0xcf,
0xd0, 0xef, 0xaa, 0xfb, 0x43, 0x4d, 0x33, 0x85, 0x45, 0xf9, 0x02, 0x7f, 0x50, 0x3c, 0x9f, 0xa8,
0x51, 0xa3, 0x40, 0x8f, 0x92, 0x9d, 0x38, 0xf5, 0xbc, 0xb6, 0xda, 0x21, 0x10, 0xff, 0xf3, 0xd2,
0xcd, 0x0c, 0x13, 0xec, 0x5f, 0x97, 0x44, 0x17, 0xc4, 0xa7, 0x7e, 0x3d, 0x64, 0x5d, 0x19, 0x73,
0x60, 0x81, 0x4f, 0xdc, 0x22, 0x2a, 0x90, 0x88, 0x46, 0xee, 0xb8, 0x14, 0xde, 0x5e, 0x0b, 0xdb,
0xe0, 0x32, 0x3a, 0x0a, 0x49, 0x06, 0x24, 0x5c, 0xc2, 0xd3, 0xac, 0x62, 0x91, 0x95, 0xe4, 0x79,
0xe7, 0xc8, 0x37, 0x6d, 0x8d, 0xd5, 0x4e, 0xa9, 0x6c, 0x56, 0xf4, 0xea, 0x65, 0x7a, 0xae, 0x08,
0xba, 0x78, 0x25, 0x2e, 0x1c, 0xa6, 0xb4, 0xc6, 0xe8, 0xdd, 0x74, 0x1f, 0x4b, 0xbd, 0x8b, 0x8a,
0x70, 0x3e, 0xb5, 0x66, 0x48, 0x03, 0xf6, 0x0e, 0x61, 0x35, 0x57, 0xb9, 0x86, 0xc1, 0x1d, 0x9e,
0xe1, 0xf8, 0x98, 0x11, 0x69, 0xd9, 0x8e, 0x94, 0x9b, 0x1e, 0x87, 0xe9, 0xce, 0x55, 0x28, 0xdf,
0x8c, 0xa1, 0x89, 0x0d, 0xbf, 0xe6, 0x42, 0x68, 0x41, 0x99, 0x2d, 0x0f, 0xb0, 0x54, 0xbb, 0x16,
};
// Uses a byte substitution table with a constant set of values
// Needs to do a table look up
uint8_t Substituted[16];
Bytes = std::min(Bytes, (size_t)16);
for (size_t i = 0; i < Bytes; ++i) {
Substituted[i] = SubstitutionTable[State[i]];
}
__uint128_t Res{};
memcpy(&Res, Substituted, Bytes);
return Res;
}
static uint8_t FFMul02(uint8_t in) {
static const uint8_t FFMul02[256] = {
0x00, 0x02, 0x04, 0x06, 0x08, 0x0a, 0x0c, 0x0e, 0x10, 0x12, 0x14, 0x16, 0x18, 0x1a, 0x1c, 0x1e,
0x20, 0x22, 0x24, 0x26, 0x28, 0x2a, 0x2c, 0x2e, 0x30, 0x32, 0x34, 0x36, 0x38, 0x3a, 0x3c, 0x3e,
0x40, 0x42, 0x44, 0x46, 0x48, 0x4a, 0x4c, 0x4e, 0x50, 0x52, 0x54, 0x56, 0x58, 0x5a, 0x5c, 0x5e,
0x60, 0x62, 0x64, 0x66, 0x68, 0x6a, 0x6c, 0x6e, 0x70, 0x72, 0x74, 0x76, 0x78, 0x7a, 0x7c, 0x7e,
0x80, 0x82, 0x84, 0x86, 0x88, 0x8a, 0x8c, 0x8e, 0x90, 0x92, 0x94, 0x96, 0x98, 0x9a, 0x9c, 0x9e,
0xa0, 0xa2, 0xa4, 0xa6, 0xa8, 0xaa, 0xac, 0xae, 0xb0, 0xb2, 0xb4, 0xb6, 0xb8, 0xba, 0xbc, 0xbe,
0xc0, 0xc2, 0xc4, 0xc6, 0xc8, 0xca, 0xcc, 0xce, 0xd0, 0xd2, 0xd4, 0xd6, 0xd8, 0xda, 0xdc, 0xde,
0xe0, 0xe2, 0xe4, 0xe6, 0xe8, 0xea, 0xec, 0xee, 0xf0, 0xf2, 0xf4, 0xf6, 0xf8, 0xfa, 0xfc, 0xfe,
0x1b, 0x19, 0x1f, 0x1d, 0x13, 0x11, 0x17, 0x15, 0x0b, 0x09, 0x0f, 0x0d, 0x03, 0x01, 0x07, 0x05,
0x3b, 0x39, 0x3f, 0x3d, 0x33, 0x31, 0x37, 0x35, 0x2b, 0x29, 0x2f, 0x2d, 0x23, 0x21, 0x27, 0x25,
0x5b, 0x59, 0x5f, 0x5d, 0x53, 0x51, 0x57, 0x55, 0x4b, 0x49, 0x4f, 0x4d, 0x43, 0x41, 0x47, 0x45,
0x7b, 0x79, 0x7f, 0x7d, 0x73, 0x71, 0x77, 0x75, 0x6b, 0x69, 0x6f, 0x6d, 0x63, 0x61, 0x67, 0x65,
0x9b, 0x99, 0x9f, 0x9d, 0x93, 0x91, 0x97, 0x95, 0x8b, 0x89, 0x8f, 0x8d, 0x83, 0x81, 0x87, 0x85,
0xbb, 0xb9, 0xbf, 0xbd, 0xb3, 0xb1, 0xb7, 0xb5, 0xab, 0xa9, 0xaf, 0xad, 0xa3, 0xa1, 0xa7, 0xa5,
0xdb, 0xd9, 0xdf, 0xdd, 0xd3, 0xd1, 0xd7, 0xd5, 0xcb, 0xc9, 0xcf, 0xcd, 0xc3, 0xc1, 0xc7, 0xc5,
0xfb, 0xf9, 0xff, 0xfd, 0xf3, 0xf1, 0xf7, 0xf5, 0xeb, 0xe9, 0xef, 0xed, 0xe3, 0xe1, 0xe7, 0xe5,
};
return FFMul02[in];
}
static uint8_t FFMul03(uint8_t in) {
static const uint8_t FFMul03[256] = {
0x00, 0x03, 0x06, 0x05, 0x0c, 0x0f, 0x0a, 0x09, 0x18, 0x1b, 0x1e, 0x1d, 0x14, 0x17, 0x12, 0x11,
0x30, 0x33, 0x36, 0x35, 0x3c, 0x3f, 0x3a, 0x39, 0x28, 0x2b, 0x2e, 0x2d, 0x24, 0x27, 0x22, 0x21,
0x60, 0x63, 0x66, 0x65, 0x6c, 0x6f, 0x6a, 0x69, 0x78, 0x7b, 0x7e, 0x7d, 0x74, 0x77, 0x72, 0x71,
0x50, 0x53, 0x56, 0x55, 0x5c, 0x5f, 0x5a, 0x59, 0x48, 0x4b, 0x4e, 0x4d, 0x44, 0x47, 0x42, 0x41,
0xc0, 0xc3, 0xc6, 0xc5, 0xcc, 0xcf, 0xca, 0xc9, 0xd8, 0xdb, 0xde, 0xdd, 0xd4, 0xd7, 0xd2, 0xd1,
0xf0, 0xf3, 0xf6, 0xf5, 0xfc, 0xff, 0xfa, 0xf9, 0xe8, 0xeb, 0xee, 0xed, 0xe4, 0xe7, 0xe2, 0xe1,
0xa0, 0xa3, 0xa6, 0xa5, 0xac, 0xaf, 0xaa, 0xa9, 0xb8, 0xbb, 0xbe, 0xbd, 0xb4, 0xb7, 0xb2, 0xb1,
0x90, 0x93, 0x96, 0x95, 0x9c, 0x9f, 0x9a, 0x99, 0x88, 0x8b, 0x8e, 0x8d, 0x84, 0x87, 0x82, 0x81,
0x9b, 0x98, 0x9d, 0x9e, 0x97, 0x94, 0x91, 0x92, 0x83, 0x80, 0x85, 0x86, 0x8f, 0x8c, 0x89, 0x8a,
0xab, 0xa8, 0xad, 0xae, 0xa7, 0xa4, 0xa1, 0xa2, 0xb3, 0xb0, 0xb5, 0xb6, 0xbf, 0xbc, 0xb9, 0xba,
0xfb, 0xf8, 0xfd, 0xfe, 0xf7, 0xf4, 0xf1, 0xf2, 0xe3, 0xe0, 0xe5, 0xe6, 0xef, 0xec, 0xe9, 0xea,
0xcb, 0xc8, 0xcd, 0xce, 0xc7, 0xc4, 0xc1, 0xc2, 0xd3, 0xd0, 0xd5, 0xd6, 0xdf, 0xdc, 0xd9, 0xda,
0x5b, 0x58, 0x5d, 0x5e, 0x57, 0x54, 0x51, 0x52, 0x43, 0x40, 0x45, 0x46, 0x4f, 0x4c, 0x49, 0x4a,
0x6b, 0x68, 0x6d, 0x6e, 0x67, 0x64, 0x61, 0x62, 0x73, 0x70, 0x75, 0x76, 0x7f, 0x7c, 0x79, 0x7a,
0x3b, 0x38, 0x3d, 0x3e, 0x37, 0x34, 0x31, 0x32, 0x23, 0x20, 0x25, 0x26, 0x2f, 0x2c, 0x29, 0x2a,
0x0b, 0x08, 0x0d, 0x0e, 0x07, 0x04, 0x01, 0x02, 0x13, 0x10, 0x15, 0x16, 0x1f, 0x1c, 0x19, 0x1a,
};
return FFMul03[in];
}
static __uint128_t MixColumns(uint8_t *State) {
uint8_t In0[16] = {
State[0], State[4], State[8], State[12],
State[1], State[5], State[9], State[13],
State[2], State[6], State[10], State[14],
State[3], State[7], State[11], State[15],
};
uint8_t Out0[4]{};
uint8_t Out1[4]{};
uint8_t Out2[4]{};
uint8_t Out3[4]{};
for (size_t i = 0; i < 4; ++i) {
Out0[i] = FFMul02(In0[0 + i]) ^ FFMul03(In0[4 + i]) ^ In0[8 + i] ^ In0[12 + i];
Out1[i] = In0[0 + i] ^ FFMul02(In0[4 + i]) ^ FFMul03(In0[8 + i]) ^ In0[12 + i];
Out2[i] = In0[0 + i] ^ In0[4 + i] ^ FFMul02(In0[8 + i]) ^ FFMul03(In0[12 + i]);
Out3[i] = FFMul03(In0[0 + i]) ^ In0[4 + i] ^ In0[8 + i] ^ FFMul02(In0[12 + i]);
}
uint8_t OutArray[16] = {
Out0[0], Out1[0], Out2[0], Out3[0],
Out0[1], Out1[1], Out2[1], Out3[1],
Out0[2], Out1[2], Out2[2], Out3[2],
Out0[3], Out1[3], Out2[3], Out3[3],
};
__uint128_t Res{};
memcpy(&Res, OutArray, 16);
return Res;
}
static uint8_t FFMul09(uint8_t in) {
static const uint8_t FFMul09[256] = {
0x00, 0x09, 0x12, 0x1b, 0x24, 0x2d, 0x36, 0x3f, 0x48, 0x41, 0x5a, 0x53, 0x6c, 0x65, 0x7e, 0x77,
0x90, 0x99, 0x82, 0x8b, 0xb4, 0xbd, 0xa6, 0xaf, 0xd8, 0xd1, 0xca, 0xc3, 0xfc, 0xf5, 0xee, 0xe7,
0x3b, 0x32, 0x29, 0x20, 0x1f, 0x16, 0x0d, 0x04, 0x73, 0x7a, 0x61, 0x68, 0x57, 0x5e, 0x45, 0x4c,
0xab, 0xa2, 0xb9, 0xb0, 0x8f, 0x86, 0x9d, 0x94, 0xe3, 0xea, 0xf1, 0xf8, 0xc7, 0xce, 0xd5, 0xdc,
0x76, 0x7f, 0x64, 0x6d, 0x52, 0x5b, 0x40, 0x49, 0x3e, 0x37, 0x2c, 0x25, 0x1a, 0x13, 0x08, 0x01,
0xe6, 0xef, 0xf4, 0xfd, 0xc2, 0xcb, 0xd0, 0xd9, 0xae, 0xa7, 0xbc, 0xb5, 0x8a, 0x83, 0x98, 0x91,
0x4d, 0x44, 0x5f, 0x56, 0x69, 0x60, 0x7b, 0x72, 0x05, 0x0c, 0x17, 0x1e, 0x21, 0x28, 0x33, 0x3a,
0xdd, 0xd4, 0xcf, 0xc6, 0xf9, 0xf0, 0xeb, 0xe2, 0x95, 0x9c, 0x87, 0x8e, 0xb1, 0xb8, 0xa3, 0xaa,
0xec, 0xe5, 0xfe, 0xf7, 0xc8, 0xc1, 0xda, 0xd3, 0xa4, 0xad, 0xb6, 0xbf, 0x80, 0x89, 0x92, 0x9b,
0x7c, 0x75, 0x6e, 0x67, 0x58, 0x51, 0x4a, 0x43, 0x34, 0x3d, 0x26, 0x2f, 0x10, 0x19, 0x02, 0x0b,
0xd7, 0xde, 0xc5, 0xcc, 0xf3, 0xfa, 0xe1, 0xe8, 0x9f, 0x96, 0x8d, 0x84, 0xbb, 0xb2, 0xa9, 0xa0,
0x47, 0x4e, 0x55, 0x5c, 0x63, 0x6a, 0x71, 0x78, 0x0f, 0x06, 0x1d, 0x14, 0x2b, 0x22, 0x39, 0x30,
0x9a, 0x93, 0x88, 0x81, 0xbe, 0xb7, 0xac, 0xa5, 0xd2, 0xdb, 0xc0, 0xc9, 0xf6, 0xff, 0xe4, 0xed,
0x0a, 0x03, 0x18, 0x11, 0x2e, 0x27, 0x3c, 0x35, 0x42, 0x4b, 0x50, 0x59, 0x66, 0x6f, 0x74, 0x7d,
0xa1, 0xa8, 0xb3, 0xba, 0x85, 0x8c, 0x97, 0x9e, 0xe9, 0xe0, 0xfb, 0xf2, 0xcd, 0xc4, 0xdf, 0xd6,
0x31, 0x38, 0x23, 0x2a, 0x15, 0x1c, 0x07, 0x0e, 0x79, 0x70, 0x6b, 0x62, 0x5d, 0x54, 0x4f, 0x46,
};
return FFMul09[in];
}
static uint8_t FFMul0B(uint8_t in) {
static const uint8_t FFMul0B[256] = {
0x00, 0x0b, 0x16, 0x1d, 0x2c, 0x27, 0x3a, 0x31, 0x58, 0x53, 0x4e, 0x45, 0x74, 0x7f, 0x62, 0x69,
0xb0, 0xbb, 0xa6, 0xad, 0x9c, 0x97, 0x8a, 0x81, 0xe8, 0xe3, 0xfe, 0xf5, 0xc4, 0xcf, 0xd2, 0xd9,
0x7b, 0x70, 0x6d, 0x66, 0x57, 0x5c, 0x41, 0x4a, 0x23, 0x28, 0x35, 0x3e, 0x0f, 0x04, 0x19, 0x12,
0xcb, 0xc0, 0xdd, 0xd6, 0xe7, 0xec, 0xf1, 0xfa, 0x93, 0x98, 0x85, 0x8e, 0xbf, 0xb4, 0xa9, 0xa2,
0xf6, 0xfd, 0xe0, 0xeb, 0xda, 0xd1, 0xcc, 0xc7, 0xae, 0xa5, 0xb8, 0xb3, 0x82, 0x89, 0x94, 0x9f,
0x46, 0x4d, 0x50, 0x5b, 0x6a, 0x61, 0x7c, 0x77, 0x1e, 0x15, 0x08, 0x03, 0x32, 0x39, 0x24, 0x2f,
0x8d, 0x86, 0x9b, 0x90, 0xa1, 0xaa, 0xb7, 0xbc, 0xd5, 0xde, 0xc3, 0xc8, 0xf9, 0xf2, 0xef, 0xe4,
0x3d, 0x36, 0x2b, 0x20, 0x11, 0x1a, 0x07, 0x0c, 0x65, 0x6e, 0x73, 0x78, 0x49, 0x42, 0x5f, 0x54,
0xf7, 0xfc, 0xe1, 0xea, 0xdb, 0xd0, 0xcd, 0xc6, 0xaf, 0xa4, 0xb9, 0xb2, 0x83, 0x88, 0x95, 0x9e,
0x47, 0x4c, 0x51, 0x5a, 0x6b, 0x60, 0x7d, 0x76, 0x1f, 0x14, 0x09, 0x02, 0x33, 0x38, 0x25, 0x2e,
0x8c, 0x87, 0x9a, 0x91, 0xa0, 0xab, 0xb6, 0xbd, 0xd4, 0xdf, 0xc2, 0xc9, 0xf8, 0xf3, 0xee, 0xe5,
0x3c, 0x37, 0x2a, 0x21, 0x10, 0x1b, 0x06, 0x0d, 0x64, 0x6f, 0x72, 0x79, 0x48, 0x43, 0x5e, 0x55,
0x01, 0x0a, 0x17, 0x1c, 0x2d, 0x26, 0x3b, 0x30, 0x59, 0x52, 0x4f, 0x44, 0x75, 0x7e, 0x63, 0x68,
0xb1, 0xba, 0xa7, 0xac, 0x9d, 0x96, 0x8b, 0x80, 0xe9, 0xe2, 0xff, 0xf4, 0xc5, 0xce, 0xd3, 0xd8,
0x7a, 0x71, 0x6c, 0x67, 0x56, 0x5d, 0x40, 0x4b, 0x22, 0x29, 0x34, 0x3f, 0x0e, 0x05, 0x18, 0x13,
0xca, 0xc1, 0xdc, 0xd7, 0xe6, 0xed, 0xf0, 0xfb, 0x92, 0x99, 0x84, 0x8f, 0xbe, 0xb5, 0xa8, 0xa3,
};
return FFMul0B[in];
}
static uint8_t FFMul0D(uint8_t in) {
static const uint8_t FFMul0D[256] = {
0x00, 0x0d, 0x1a, 0x17, 0x34, 0x39, 0x2e, 0x23, 0x68, 0x65, 0x72, 0x7f, 0x5c, 0x51, 0x46, 0x4b,
0xd0, 0xdd, 0xca, 0xc7, 0xe4, 0xe9, 0xfe, 0xf3, 0xb8, 0xb5, 0xa2, 0xaf, 0x8c, 0x81, 0x96, 0x9b,
0xbb, 0xb6, 0xa1, 0xac, 0x8f, 0x82, 0x95, 0x98, 0xd3, 0xde, 0xc9, 0xc4, 0xe7, 0xea, 0xfd, 0xf0,
0x6b, 0x66, 0x71, 0x7c, 0x5f, 0x52, 0x45, 0x48, 0x03, 0x0e, 0x19, 0x14, 0x37, 0x3a, 0x2d, 0x20,
0x6d, 0x60, 0x77, 0x7a, 0x59, 0x54, 0x43, 0x4e, 0x05, 0x08, 0x1f, 0x12, 0x31, 0x3c, 0x2b, 0x26,
0xbd, 0xb0, 0xa7, 0xaa, 0x89, 0x84, 0x93, 0x9e, 0xd5, 0xd8, 0xcf, 0xc2, 0xe1, 0xec, 0xfb, 0xf6,
0xd6, 0xdb, 0xcc, 0xc1, 0xe2, 0xef, 0xf8, 0xf5, 0xbe, 0xb3, 0xa4, 0xa9, 0x8a, 0x87, 0x90, 0x9d,
0x06, 0x0b, 0x1c, 0x11, 0x32, 0x3f, 0x28, 0x25, 0x6e, 0x63, 0x74, 0x79, 0x5a, 0x57, 0x40, 0x4d,
0xda, 0xd7, 0xc0, 0xcd, 0xee, 0xe3, 0xf4, 0xf9, 0xb2, 0xbf, 0xa8, 0xa5, 0x86, 0x8b, 0x9c, 0x91,
0x0a, 0x07, 0x10, 0x1d, 0x3e, 0x33, 0x24, 0x29, 0x62, 0x6f, 0x78, 0x75, 0x56, 0x5b, 0x4c, 0x41,
0x61, 0x6c, 0x7b, 0x76, 0x55, 0x58, 0x4f, 0x42, 0x09, 0x04, 0x13, 0x1e, 0x3d, 0x30, 0x27, 0x2a,
0xb1, 0xbc, 0xab, 0xa6, 0x85, 0x88, 0x9f, 0x92, 0xd9, 0xd4, 0xc3, 0xce, 0xed, 0xe0, 0xf7, 0xfa,
0xb7, 0xba, 0xad, 0xa0, 0x83, 0x8e, 0x99, 0x94, 0xdf, 0xd2, 0xc5, 0xc8, 0xeb, 0xe6, 0xf1, 0xfc,
0x67, 0x6a, 0x7d, 0x70, 0x53, 0x5e, 0x49, 0x44, 0x0f, 0x02, 0x15, 0x18, 0x3b, 0x36, 0x21, 0x2c,
0x0c, 0x01, 0x16, 0x1b, 0x38, 0x35, 0x22, 0x2f, 0x64, 0x69, 0x7e, 0x73, 0x50, 0x5d, 0x4a, 0x47,
0xdc, 0xd1, 0xc6, 0xcb, 0xe8, 0xe5, 0xf2, 0xff, 0xb4, 0xb9, 0xae, 0xa3, 0x80, 0x8d, 0x9a, 0x97,
};
return FFMul0D[in];
}
static uint8_t FFMul0E(uint8_t in) {
static const uint8_t FFMul0E[256] = {
0x00, 0x0e, 0x1c, 0x12, 0x38, 0x36, 0x24, 0x2a, 0x70, 0x7e, 0x6c, 0x62, 0x48, 0x46, 0x54, 0x5a,
0xe0, 0xee, 0xfc, 0xf2, 0xd8, 0xd6, 0xc4, 0xca, 0x90, 0x9e, 0x8c, 0x82, 0xa8, 0xa6, 0xb4, 0xba,
0xdb, 0xd5, 0xc7, 0xc9, 0xe3, 0xed, 0xff, 0xf1, 0xab, 0xa5, 0xb7, 0xb9, 0x93, 0x9d, 0x8f, 0x81,
0x3b, 0x35, 0x27, 0x29, 0x03, 0x0d, 0x1f, 0x11, 0x4b, 0x45, 0x57, 0x59, 0x73, 0x7d, 0x6f, 0x61,
0xad, 0xa3, 0xb1, 0xbf, 0x95, 0x9b, 0x89, 0x87, 0xdd, 0xd3, 0xc1, 0xcf, 0xe5, 0xeb, 0xf9, 0xf7,
0x4d, 0x43, 0x51, 0x5f, 0x75, 0x7b, 0x69, 0x67, 0x3d, 0x33, 0x21, 0x2f, 0x05, 0x0b, 0x19, 0x17,
0x76, 0x78, 0x6a, 0x64, 0x4e, 0x40, 0x52, 0x5c, 0x06, 0x08, 0x1a, 0x14, 0x3e, 0x30, 0x22, 0x2c,
0x96, 0x98, 0x8a, 0x84, 0xae, 0xa0, 0xb2, 0xbc, 0xe6, 0xe8, 0xfa, 0xf4, 0xde, 0xd0, 0xc2, 0xcc,
0x41, 0x4f, 0x5d, 0x53, 0x79, 0x77, 0x65, 0x6b, 0x31, 0x3f, 0x2d, 0x23, 0x09, 0x07, 0x15, 0x1b,
0xa1, 0xaf, 0xbd, 0xb3, 0x99, 0x97, 0x85, 0x8b, 0xd1, 0xdf, 0xcd, 0xc3, 0xe9, 0xe7, 0xf5, 0xfb,
0x9a, 0x94, 0x86, 0x88, 0xa2, 0xac, 0xbe, 0xb0, 0xea, 0xe4, 0xf6, 0xf8, 0xd2, 0xdc, 0xce, 0xc0,
0x7a, 0x74, 0x66, 0x68, 0x42, 0x4c, 0x5e, 0x50, 0x0a, 0x04, 0x16, 0x18, 0x32, 0x3c, 0x2e, 0x20,
0xec, 0xe2, 0xf0, 0xfe, 0xd4, 0xda, 0xc8, 0xc6, 0x9c, 0x92, 0x80, 0x8e, 0xa4, 0xaa, 0xb8, 0xb6,
0x0c, 0x02, 0x10, 0x1e, 0x34, 0x3a, 0x28, 0x26, 0x7c, 0x72, 0x60, 0x6e, 0x44, 0x4a, 0x58, 0x56,
0x37, 0x39, 0x2b, 0x25, 0x0f, 0x01, 0x13, 0x1d, 0x47, 0x49, 0x5b, 0x55, 0x7f, 0x71, 0x63, 0x6d,
0xd7, 0xd9, 0xcb, 0xc5, 0xef, 0xe1, 0xf3, 0xfd, 0xa7, 0xa9, 0xbb, 0xb5, 0x9f, 0x91, 0x83, 0x8d,
};
return FFMul0E[in];
}
static __uint128_t InvMixColumns(uint8_t *State) {
uint8_t In0[16] = {
State[0], State[4], State[8], State[12],
State[1], State[5], State[9], State[13],
State[2], State[6], State[10], State[14],
State[3], State[7], State[11], State[15],
};
uint8_t Out0[4]{};
uint8_t Out1[4]{};
uint8_t Out2[4]{};
uint8_t Out3[4]{};
for (size_t i = 0; i < 4; ++i) {
Out0[i] = FFMul0E(In0[0 + i]) ^ FFMul0B(In0[4 + i]) ^ FFMul0D(In0[8 + i]) ^ FFMul09(In0[12 + i]);
Out1[i] = FFMul09(In0[0 + i]) ^ FFMul0E(In0[4 + i]) ^ FFMul0B(In0[8 + i]) ^ FFMul0D(In0[12 + i]);
Out2[i] = FFMul0D(In0[0 + i]) ^ FFMul09(In0[4 + i]) ^ FFMul0E(In0[8 + i]) ^ FFMul0B(In0[12 + i]);
Out3[i] = FFMul0B(In0[0 + i]) ^ FFMul0D(In0[4 + i]) ^ FFMul09(In0[8 + i]) ^ FFMul0E(In0[12 + i]);
}
uint8_t OutArray[16] = {
Out0[0], Out1[0], Out2[0], Out3[0],
Out0[1], Out1[1], Out2[1], Out3[1],
Out0[2], Out1[2], Out2[2], Out3[2],
Out0[3], Out1[3], Out2[3], Out3[3],
};
__uint128_t Res{};
memcpy(&Res, OutArray, 16);
return Res;
}
}
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(AESImc) {
auto Op = IROp->C<IR::IROp_VAESImc>();
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
// Pseudo-code
// Dst = InvMixColumns(STATE)
__uint128_t Tmp{};
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Src1));
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESEnc) {
auto Op = IROp->C<IR::IROp_VAESEnc>();
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = ShiftRows(STATE)
// STATE = SubBytes(STATE)
// STATE = MixColumns(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
Tmp = AES::MixColumns(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESEncLast) {
auto Op = IROp->C<IR::IROp_VAESEncLast>();
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = ShiftRows(STATE)
// STATE = SubBytes(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::ShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::SubBytes(reinterpret_cast<uint8_t*>(&Tmp), 16);
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESDec) {
auto Op = IROp->C<IR::IROp_VAESDec>();
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = InvShiftRows(STATE)
// STATE = InvSubBytes(STATE)
// STATE = InvMixColumns(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = AES::InvMixColumns(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESDecLast) {
auto Op = IROp->C<IR::IROp_VAESDecLast>();
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
// Pseudo-code
// STATE = Src1
// RoundKey = Src2
// STATE = InvShiftRows(STATE)
// STATE = InvSubBytes(STATE)
// Dst = STATE XOR RoundKey
__uint128_t Tmp{};
Tmp = AES::InvShiftRows(reinterpret_cast<uint8_t*>(&Src1));
Tmp = AES::InvSubBytes(reinterpret_cast<uint8_t*>(&Tmp));
Tmp = Tmp ^ Src2;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
DEF_OP(AESKeyGenAssist) {
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
// Pseudo-code
// X3 = Src1[127:96]
// X2 = Src1[95:64]
// X1 = Src1[63:32]
// X0 = Src1[31:30]
// RCON = (Zext)rcon
// Dest[31:0] = SubWord(X1)
// Dest[63:32] = RotWord(SubWord(X1)) XOR RCON
// Dest[95:64] = SubWord(X3)
// Dest[127:96] = RotWord(SubWord(X3)) XOR RCON
__uint128_t Tmp{};
uint32_t X1{};
uint32_t X3{};
memcpy(&X1, &Src1[4], 4);
memcpy(&X3, &Src1[12], 4);
uint32_t SubWord_X1 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X1), 4);
uint32_t SubWord_X3 = AES::SubBytes(reinterpret_cast<uint8_t*>(&X3), 4);
auto Ror = [] (auto In, auto R) {
auto RotateMask = sizeof(In) * 8 - 1;
R &= RotateMask;
return (In >> R) | (In << (sizeof(In) * 8 - R));
};
uint32_t Rot_X1 = Ror(SubWord_X1, 8);
uint32_t Rot_X3 = Ror(SubWord_X3, 8);
Tmp = Rot_X3 ^ Op->RCON;
Tmp <<= 32;
Tmp |= SubWord_X3;
Tmp <<= 32;
Tmp |= Rot_X1 ^ Op->RCON;
Tmp <<= 32;
Tmp |= SubWord_X1;
memcpy(GDP, &Tmp, sizeof(Tmp));
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,361 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include "F80Ops.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(F80LOADFCW) {
FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle(*GetSrc<uint16_t*>(Data->SSAData, IROp->Args[0]));
}
DEF_OP(F80ADD) {
auto Op = IROp->C<IR::IROp_F80Add>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FADD(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SUB) {
auto Op = IROp->C<IR::IROp_F80Sub>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FSUB(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80MUL) {
auto Op = IROp->C<IR::IROp_F80Mul>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FMUL(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80DIV) {
auto Op = IROp->C<IR::IROp_F80Div>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FDIV(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80FYL2X) {
auto Op = IROp->C<IR::IROp_F80FYL2X>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FYL2X(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80ATAN) {
auto Op = IROp->C<IR::IROp_F80ATAN>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FATAN(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80FPREM1) {
auto Op = IROp->C<IR::IROp_F80FPREM1>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FREM1(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80FPREM) {
auto Op = IROp->C<IR::IROp_F80FPREM>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FREM(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SCALE) {
auto Op = IROp->C<IR::IROp_F80SCALE>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FSCALE(Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80CVT) {
auto Op = IROp->C<IR::IROp_F80CVT>();
uint8_t OpSize = IROp->Size;
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
switch (OpSize) {
case 4: {
float Tmp = Src;
memcpy(GDP, &Tmp, OpSize);
break;
}
case 8: {
double Tmp = Src;
memcpy(GDP, &Tmp, OpSize);
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
}
DEF_OP(F80CVTINT) {
auto Op = IROp->C<IR::IROp_F80CVTInt>();
uint8_t OpSize = IROp->Size;
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
switch (OpSize) {
case 2: {
int16_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2)(Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
case 4: {
int32_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4)(Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
case 8: {
int64_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8)(Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
}
DEF_OP(F80CVTTO) {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch (Op->Size) {
case 4: {
float Src = *GetSrc<float *>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
case 8: {
double Src = *GetSrc<double *>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->Size);
}
}
DEF_OP(F80CVTTOINT) {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->Size) {
case 2: {
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
case 4: {
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp = Src;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->Size);
}
}
DEF_OP(F80ROUND) {
auto Op = IROp->C<IR::IROp_F80Round>();
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FRNDINT(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80F2XM1) {
auto Op = IROp->C<IR::IROp_F80F2XM1>();
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::F2XM1(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80TAN) {
auto Op = IROp->C<IR::IROp_F80TAN>();
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FTAN(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SQRT) {
auto Op = IROp->C<IR::IROp_F80SQRT>();
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FSQRT(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80SIN) {
auto Op = IROp->C<IR::IROp_F80SIN>();
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FSIN(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80COS) {
auto Op = IROp->C<IR::IROp_F80COS>();
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FCOS(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80XTRACT_EXP) {
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FXTRACT_EXP(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80XTRACT_SIG) {
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp;
Tmp = X80SoftFloat::FXTRACT_SIG(Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80CMP) {
auto Op = IROp->C<IR::IROp_F80Cmp>();
uint32_t ResultFlags{};
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
bool eq, lt, nan;
X80SoftFloat::FCMP(Src1, Src2, &eq, &lt, &nan);
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
nan) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ) &&
eq) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
GD = ResultFlags;
}
DEF_OP(F80BCDLOAD) {
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
// Each 4bit split is a digit
// Only 0-9 is supported, A-F results in undefined data
// | 4 bit | 4 bit |
// | 10s place | 1s place |
// EG 0x48 = 48
// EG 0x4847 = 4847
// This gives us an 18digit value encoded in BCD
// The last byte lets us know if it negative or not
for (size_t i = 0; i < 9; ++i) {
uint8_t Digit = Src1[8 - i];
// First shift our last value over
BCD *= 100;
// Add the tens place digit
BCD += (Digit >> 4) * 10;
// Add the ones place digit
BCD += Digit & 0xF;
}
// Set negative flag once converted to x87
bool Negative = Src1[9] & 0x80;
X80SoftFloat Tmp;
Tmp = BCD;
Tmp.Sign = Negative;
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80BCDSTORE) {
auto Op = IROp->C<IR::IROp_F80BCDStore>();
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
bool Negative = Src1.Sign;
// Clear the Sign bit
Src1.Sign = 0;
uint64_t Tmp = Src1;
uint8_t BCD[10]{};
for (size_t i = 0; i < 9; ++i) {
if (Tmp == 0) {
// Nothing left? Just leave
break;
}
// Extract the lower 100 values
uint8_t Digit = Tmp % 100;
// Now divide it for the next iteration
Tmp /= 100;
uint8_t UpperNibble = Digit / 10;
uint8_t LowerNibble = Digit % 10;
// Now store the BCD
BCD[i] = (UpperNibble << 4) | LowerNibble;
}
// Set negative flag once converted to x87
BCD[9] = Negative ? 0x80 : 0;
memcpy(GDP, BCD, 10);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,330 +0,0 @@
#pragma once
#include "Common/SoftFloat.h"
#include "Common/SoftFloat-3e/softfloat.h"
#include <FEXCore/IR/IR.h>
namespace FEXCore::CPU {
template<IR::IROps Op>
struct OpHandlers {
};
template<>
struct OpHandlers<IR::OP_F80CVTTO> {
static X80SoftFloat handle4(float src) {
return src;
}
static X80SoftFloat handle8(double src) {
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80CMP> {
template<uint32_t Flags>
static uint64_t handle(X80SoftFloat Src1, X80SoftFloat Src2) {
bool eq, lt, nan;
uint64_t ResultFlags = 0;
X80SoftFloat::FCMP(Src1, Src2, &eq, &lt, &nan);
if (Flags & (1 << IR::FCMP_FLAG_LT) &&
lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
nan) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
if (Flags & (1 << IR::FCMP_FLAG_EQ) &&
eq) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
return ResultFlags;
}
};
template<>
struct OpHandlers<IR::OP_F80CVT> {
static float handle4(X80SoftFloat src) {
return src;
}
static double handle8(X80SoftFloat src) {
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80CVTINT> {
static int16_t handle2(X80SoftFloat src) {
return src;
}
static int32_t handle4(X80SoftFloat src) {
return src;
}
static int64_t handle8(X80SoftFloat src) {
return src;
}
static int16_t handle2t(X80SoftFloat src) {
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
if (rv > INT16_MAX) {
return INT16_MAX;
} else if (rv < INT16_MIN) {
return INT16_MIN;
} else {
return rv;
}
}
static int32_t handle4t(X80SoftFloat src) {
return extF80_to_i32(src, softfloat_round_minMag, false);
}
static int64_t handle8t(X80SoftFloat src) {
return extF80_to_i64(src, softfloat_round_minMag, false);
}
};
template<>
struct OpHandlers<IR::OP_F80CVTTOINT> {
static X80SoftFloat handle2(int16_t src) {
return src;
}
static X80SoftFloat handle4(int32_t src) {
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80ROUND> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FRNDINT(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80F2XM1> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::F2XM1(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80TAN> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FTAN(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80SQRT> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FSQRT(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80SIN> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FSIN(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80COS> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FCOS(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FXTRACT_EXP(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
static X80SoftFloat handle(X80SoftFloat Src1) {
return X80SoftFloat::FXTRACT_SIG(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80ADD> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FADD(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80SUB> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FSUB(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80MUL> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FMUL(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80DIV> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FDIV(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FYL2X> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FYL2X(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80ATAN> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FATAN(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FPREM1> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FREM1(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FPREM> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FREM(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80SCALE> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
return X80SoftFloat::FSCALE(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80BCDSTORE> {
static X80SoftFloat handle(X80SoftFloat Src1) {
bool Negative = Src1.Sign;
// Clear the Sign bit
Src1.Sign = 0;
uint64_t Tmp = Src1;
X80SoftFloat Rv;
uint8_t *BCD = reinterpret_cast<uint8_t*>(&Rv);
memset(BCD, 0, 10);
for (size_t i = 0; i < 9; ++i) {
if (Tmp == 0) {
// Nothing left? Just leave
break;
}
// Extract the lower 100 values
uint8_t Digit = Tmp % 100;
// Now divide it for the next iteration
Tmp /= 100;
uint8_t UpperNibble = Digit / 10;
uint8_t LowerNibble = Digit % 10;
// Now store the BCD
BCD[i] = (UpperNibble << 4) | LowerNibble;
}
// Set negative flag once converted to x87
BCD[9] = Negative ? 0x80 : 0;
return Rv;
}
};
template<>
struct OpHandlers<IR::OP_F80BCDLOAD> {
static X80SoftFloat handle(X80SoftFloat Src) {
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
// Each 4bit split is a digit
// Only 0-9 is supported, A-F results in undefined data
// | 4 bit | 4 bit |
// | 10s place | 1s place |
// EG 0x48 = 48
// EG 0x4847 = 4847
// This gives us an 18digit value encoded in BCD
// The last byte lets us know if it negative or not
for (size_t i = 0; i < 9; ++i) {
uint8_t Digit = Src1[8 - i];
// First shift our last value over
BCD *= 100;
// Add the tens place digit
BCD += (Digit >> 4) * 10;
// Add the ones place digit
BCD += Digit & 0xF;
}
// Set negative flag once converted to x87
bool Negative = Src1[9] & 0x80;
X80SoftFloat Tmp;
Tmp = BCD;
Tmp.Sign = Negative;
return Tmp;
}
};
template<>
struct OpHandlers<IR::OP_F80LOADFCW> {
static void handle(uint16_t NewFCW) {
auto PC = (NewFCW >> 8) & 3;
switch(PC) {
case 0: extF80_roundingPrecision = 32; break;
case 2: extF80_roundingPrecision = 64; break;
case 3: extF80_roundingPrecision = 80; break;
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
}
auto RC = (NewFCW >> 10) & 3;
switch(RC) {
case 0:
softfloat_roundingMode = softfloat_round_near_even;
break;
case 1:
softfloat_roundingMode = softfloat_round_min;
break;
case 2:
softfloat_roundingMode = softfloat_round_max;
break;
case 3:
softfloat_roundingMode = softfloat_round_minMag;
break;
}
}
};
}
@@ -1,21 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(GetHostFlag) {
auto Op = IROp->C<IR::IROp_GetHostFlag>();
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]) >> Op->Flag) & 1;
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,5 +1,6 @@
#pragma once
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
@@ -20,36 +21,32 @@ using DestMapType = std::vector<uint32_t>;
class InterpreterCore final : public CPUBackend {
public:
explicit InterpreterCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread,
bool CompileThread);
explicit InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
~InterpreterCore() override;
std::string GetName() override { return "Interpreter"; }
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]] void *CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
bool NeedsOpDispatch() override { return true; }
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
bool HandleSIGBUS(int Signal, void *info, void *ucontext);
private:
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *State;
std::unique_ptr<Dispatcher> Dispatcher{};
uint32_t AllocateTmpSpace(size_t Size);
template<typename Res>
Res GetDest(void* SSAData, IR::OrderedNodeWrapper Op);
template<typename Res>
Res GetSrc(void* SSAData, IR::OrderedNodeWrapper Src);
Dispatcher *Dispatcher{};
};
template<typename T>
T AtomicCompareAndSwap(T expected, T desired, T *addr);
uint8_t AtomicFetchNeg(uint8_t *Addr);
uint16_t AtomicFetchNeg(uint16_t *Addr);
uint32_t AtomicFetchNeg(uint32_t *Addr);
uint64_t AtomicFetchNeg(uint64_t *Addr);
} // namespace FEXCore::CPU
}
@@ -1,43 +1,91 @@
#include "Common/MathUtils.h"
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/Arm64.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/DebugData.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/LogManager.h>
#include <memory>
#include <bits/types/stack_t.h>
#include <signal.h>
#include <stdint.h>
#include <unordered_map>
#include <utility>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include "Interface/HLE/Thunks/Thunks.h"
#include <atomic>
#include <cmath>
#include <limits>
#include <vector>
#include "InterpreterOps.h"
namespace FEXCore::IR {
class IRListView;
class RegisterAllocationData;
}
namespace FEXCore::CPU {
class CPUBackend;
static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
auto Thread = Frame->Thread;
auto LocalEntry = Thread->LocalIRCache.find(Thread->CurrentFrame->State.rip);
InterpreterOps::InterpretIR(Thread, Thread->CurrentFrame->State.rip, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
InterpreterOps::InterpretIR(Thread, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
}
bool InterpreterCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
#ifdef _M_ARM_64
constexpr bool is_arm64 = true;
#else
constexpr bool is_arm64 = false;
#endif
if constexpr (is_arm64) {
uint32_t *PC = reinterpret_cast<uint32_t*>(ArchHelpers::Context::GetPc(ucontext));
uint32_t Instr = PC[0];
if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
LogMan::Msg::E("Unhandled JIT SIGBUS CASPAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
return false;
}
}
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
LogMan::Msg::E("Unhandled JIT SIGBUS CASAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
return false;
}
}
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
uint8_t Op = (PC[0] >> 12) & 0xF;
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x: PC: %p Instruction: 0x%08x\n", Op, PC, PC[0]);
return false;
}
}
}
return false;
}
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
: CTX {ctx}
, State {Thread} {
// Grab our space for temporary data
if (!CompileThread &&
CTX->Config.Core == FEXCore::Config::CONFIG_INTERPRETER) {
@@ -45,31 +93,36 @@ InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
}, true);
});
#ifdef _M_ARM_64
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
}, true);
#endif
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->HandleSIGBUS(Signal, info, ucontext);
});
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
};
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
}
}
}
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
InterpreterCore::~InterpreterCore() {
delete Dispatcher;
}
void *InterpreterCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
return reinterpret_cast<void*>(InterpreterExecution);
}
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
return std::make_unique<InterpreterCore>(ctx, Thread, CompileThread);
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
return new InterpreterCore(ctx, Thread, CompileThread);
}
}
@@ -1,7 +1,5 @@
#pragma once
#include <memory>
namespace FEXCore::Context {
struct Context;
}
@@ -13,8 +11,6 @@ namespace FEXCore::Core {
namespace FEXCore::CPU {
class CPUBackend;
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread,
bool CompileThread);
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
} // namespace FEXCore::CPU
}
@@ -1,179 +0,0 @@
#pragma once
#include <FEXCore/IR/IR.h>
#define GD *GetDest<uint64_t*>(Data->SSAData, Node)
#define GDP GetDest<void*>(Data->SSAData, Node)
#define DO_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(GDP); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
*Dst_d = func(*Src1_d, *Src2_d); \
break; \
}
#define DO_SCALAR_COMPARE_OP(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
Dst_d[0] = func(Src1_d[0], Src2_d[0]); \
break; \
}
#define DO_VECTOR_COMPARE_OP(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type2*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
} \
break; \
}
#define DO_VECTOR_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], Src2_d[i]); \
} \
break; \
}
#define DO_VECTOR_PAIR_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i*2], Src1_d[i*2 + 1]); \
Dst_d[i+Elements] = func(Src2_d[i*2], Src2_d[i*2 + 1]); \
} \
break; \
}
#define DO_VECTOR_SCALAR_OP(size, type, func)\
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], *Src2_d); \
} \
break; \
}
#define DO_VECTOR_0SRC_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(); \
} \
break; \
}
#define DO_VECTOR_1SRC_OP(size, type, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src_d[i]); \
} \
break; \
}
#define DO_VECTOR_REDUCE_1SRC_OP(size, type, func, start_val) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type*>(Src); \
type begin = start_val; \
for (uint8_t i = 0; i < Elements; ++i) { \
begin = func(begin, Src_d[i]); \
} \
Dst_d[0] = begin; \
break; \
}
#define DO_VECTOR_SAT_OP(size, type, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = func(Src1_d[i], Src2_d[i], min, max); \
} \
break; \
}
#define DO_VECTOR_1SRC_2TYPE_OP(size, type, type2, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func(Src_d[i], min, max); \
} \
break; \
}
#define DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(type, type2, func, min, max) \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func(Src_d[i], min, max); \
}
#define DO_VECTOR_1SRC_2TYPE_OP_TOP(size, type, type2, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src2); \
memcpy(Dst_d, Src1, Elements * sizeof(type2));\
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i+Elements] = (type)func(Src_d[i], min, max); \
} \
break; \
}
#define DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(size, type, type2, func, min, max) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src_d = reinterpret_cast<type2*>(Src); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func(Src_d[i+Elements], min, max); \
} \
break; \
}
#define DO_VECTOR_2SRC_2TYPE_OP(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func((type)Src1_d[i], (type)Src2_d[i]); \
} \
break; \
}
#define DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(size, type, type2, func) \
case size: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type2*>(Src1); \
auto *Src2_d = reinterpret_cast<type2*>(Src2); \
for (uint8_t i = 0; i < Elements; ++i) { \
Dst_d[i] = (type)func((type)Src1_d[i+Elements], (type)Src2_d[i+Elements]); \
} \
break; \
}
template<typename Res>
Res GetDest(void* SSAData, FEXCore::IR::OrderedNodeWrapper Op) {
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Op.ID().Value];
return reinterpret_cast<Res>(DstPtr);
}
template<typename Res>
Res GetDest(void* SSAData, FEXCore::IR::NodeID Op) {
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Op.Value];
return reinterpret_cast<Res>(DstPtr);
}
template<typename Res>
Res GetSrc(void* SSAData, FEXCore::IR::OrderedNodeWrapper Src) {
auto DstPtr = &reinterpret_cast<__uint128_t*>(SSAData)[Src.ID().Value];
return reinterpret_cast<Res>(DstPtr);
}
File diff suppressed because it is too large. Load diff
@@ -1,16 +1,9 @@
#pragma once
#include <stdint.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::Core {
struct InternalThreadState;
}
namespace FEXCore::IR {
class IRListView;
struct IROp_Header;
}
namespace FEXCore::Core{
@@ -39,358 +32,11 @@ namespace FEXCore::CPU {
FallbackABI ABI;
void *fn;
};
class InterpreterOps {
public:
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
struct IROpData {
FEXCore::Core::InternalThreadState *State{};
uint64_t CurrentEntry{};
FEXCore::IR::IRListView *CurrentIR{};
volatile void *StackEntry{};
void *SSAData{};
struct {
bool Quit;
bool Redo;
} BlockResults{};
IR::NodeIterator BlockIterator{0, 0};
};
#define DEF_OP(x) static void Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
///< Unhandled handler
DEF_OP(Unhandled);
///< No-op Handler
DEF_OP(NoOp);
///< ALU Ops
DEF_OP(TruncElementPair);
DEF_OP(Constant);
DEF_OP(EntrypointOffset);
DEF_OP(InlineConstant);
DEF_OP(InlineEntrypointOffset);
DEF_OP(CycleCounter);
DEF_OP(Add);
DEF_OP(Sub);
DEF_OP(Neg);
DEF_OP(Mul);
DEF_OP(UMul);
DEF_OP(Div);
DEF_OP(UDiv);
DEF_OP(Rem);
DEF_OP(URem);
DEF_OP(MulH);
DEF_OP(UMulH);
DEF_OP(Or);
DEF_OP(And);
DEF_OP(Andn);
DEF_OP(Xor);
DEF_OP(Lshl);
DEF_OP(Lshr);
DEF_OP(Ashr);
DEF_OP(Rol);
DEF_OP(Ror);
DEF_OP(Extr);
DEF_OP(LDiv);
DEF_OP(LUDiv);
DEF_OP(LRem);
DEF_OP(LURem);
DEF_OP(Zext);
DEF_OP(Not);
DEF_OP(Popcount);
DEF_OP(FindLSB);
DEF_OP(FindMSB);
DEF_OP(FindTrailingZeros);
DEF_OP(CountLeadingZeroes);
DEF_OP(Rev);
DEF_OP(Bfi);
DEF_OP(Bfe);
DEF_OP(Sbfe);
DEF_OP(Select);
DEF_OP(VExtractToGPR);
DEF_OP(Float_ToGPR_ZU);
DEF_OP(Float_ToGPR_ZS);
DEF_OP(Float_ToGPR_S);
DEF_OP(FCmp);
///< Atomic ops
DEF_OP(CASPair);
DEF_OP(CAS);
DEF_OP(AtomicAdd);
DEF_OP(AtomicSub);
DEF_OP(AtomicAnd);
DEF_OP(AtomicOr);
DEF_OP(AtomicXor);
DEF_OP(AtomicSwap);
DEF_OP(AtomicFetchAdd);
DEF_OP(AtomicFetchSub);
DEF_OP(AtomicFetchAnd);
DEF_OP(AtomicFetchOr);
DEF_OP(AtomicFetchXor);
DEF_OP(AtomicFetchNeg);
///< Branch ops
DEF_OP(GuestCallDirect);
DEF_OP(GuestCallIndirect);
DEF_OP(GuestReturn);
DEF_OP(SignalReturn);
DEF_OP(CallbackReturn);
DEF_OP(ExitFunction);
DEF_OP(Jump);
DEF_OP(CondJump);
DEF_OP(Syscall);
DEF_OP(InlineSyscall);
DEF_OP(Thunk);
DEF_OP(ValidateCode);
DEF_OP(RemoveCodeEntry);
DEF_OP(CPUID);
///< Conversion ops
DEF_OP(VInsGPR);
DEF_OP(VCastFromGPR);
DEF_OP(Float_FromGPR_S);
DEF_OP(Float_FToF);
DEF_OP(Vector_SToF);
DEF_OP(Vector_FToZS);
DEF_OP(Vector_FToS);
DEF_OP(Vector_FToF);
DEF_OP(Vector_FToI);
///< Flag ops
DEF_OP(GetHostFlag);
///< Memory ops
DEF_OP(LoadContext);
DEF_OP(StoreContext);
DEF_OP(LoadRegister);
DEF_OP(StoreRegister);
DEF_OP(LoadContextIndexed);
DEF_OP(StoreContextIndexed);
DEF_OP(SpillRegister);
DEF_OP(FillRegister);
DEF_OP(LoadFlag);
DEF_OP(StoreFlag);
DEF_OP(LoadMem);
DEF_OP(StoreMem);
DEF_OP(VLoadMemElement);
DEF_OP(VStoreMemElement);
DEF_OP(CacheLineClear);
///< Misc ops
DEF_OP(EndBlock);
DEF_OP(Fence);
DEF_OP(Break);
DEF_OP(Phi);
DEF_OP(PhiValue);
DEF_OP(Print);
DEF_OP(GetRoundingMode);
DEF_OP(SetRoundingMode);
///< Move ops
DEF_OP(ExtractElementPair);
DEF_OP(CreateElementPair);
DEF_OP(Mov);
///< Vector ops
DEF_OP(VectorZero);
DEF_OP(VectorImm);
DEF_OP(CreateVector2);
DEF_OP(CreateVector4);
DEF_OP(SplatVector);
DEF_OP(VMov);
DEF_OP(VAnd);
DEF_OP(VBic);
DEF_OP(VOr);
DEF_OP(VXor);
DEF_OP(VAdd);
DEF_OP(VSub);
DEF_OP(VUQAdd);
DEF_OP(VUQSub);
DEF_OP(VSQAdd);
DEF_OP(VSQSub);
DEF_OP(VAddP);
DEF_OP(VAddV);
DEF_OP(VUMinV);
DEF_OP(VURAvg);
DEF_OP(VAbs);
DEF_OP(VPopcount);
DEF_OP(VFAdd);
DEF_OP(VFAddP);
DEF_OP(VFSub);
DEF_OP(VFMul);
DEF_OP(VFDiv);
DEF_OP(VFMin);
DEF_OP(VFMax);
DEF_OP(VFRecp);
DEF_OP(VFSqrt);
DEF_OP(VFRSqrt);
DEF_OP(VNeg);
DEF_OP(VFNeg);
DEF_OP(VNot);
DEF_OP(VUMin);
DEF_OP(VSMin);
DEF_OP(VUMax);
DEF_OP(VSMax);
DEF_OP(VZip);
DEF_OP(VUnZip);
DEF_OP(VBSL);
DEF_OP(VCMPEQ);
DEF_OP(VCMPEQZ);
DEF_OP(VCMPGT);
DEF_OP(VCMPGTZ);
DEF_OP(VCMPLTZ);
DEF_OP(VFCMPEQ);
DEF_OP(VFCMPNEQ);
DEF_OP(VFCMPLT);
DEF_OP(VFCMPGT);
DEF_OP(VFCMPLE);
DEF_OP(VFCMPORD);
DEF_OP(VFCMPUNO);
DEF_OP(VUShl);
DEF_OP(VUShr);
DEF_OP(VSShr);
DEF_OP(VUShlS);
DEF_OP(VUShrS);
DEF_OP(VSShrS);
DEF_OP(VInsElement);
DEF_OP(VInsScalarElement);
DEF_OP(VExtractElement);
DEF_OP(VDupElement);
DEF_OP(VExtr);
DEF_OP(VSLI);
DEF_OP(VSRI);
DEF_OP(VUShrI);
DEF_OP(VSShrI);
DEF_OP(VShlI);
DEF_OP(VUShrNI);
DEF_OP(VUShrNI2);
DEF_OP(VBitcast);
DEF_OP(VSXTL);
DEF_OP(VSXTL2);
DEF_OP(VUXTL);
DEF_OP(VUXTL2);
DEF_OP(VSQXTN);
DEF_OP(VSQXTN2);
DEF_OP(VSQXTUN);
DEF_OP(VSQXTUN2);
DEF_OP(VUMul);
DEF_OP(VUMull);
DEF_OP(VSMul);
DEF_OP(VSMull);
DEF_OP(VUMull2);
DEF_OP(VSMull2);
DEF_OP(VUABDL);
DEF_OP(VTBL1);
///< Encryption ops
DEF_OP(AESImc);
DEF_OP(AESEnc);
DEF_OP(AESEncLast);
DEF_OP(AESDec);
DEF_OP(AESDecLast);
DEF_OP(AESKeyGenAssist);
///< F80 ops
DEF_OP(F80LOADFCW);
DEF_OP(F80ADD);
DEF_OP(F80SUB);
DEF_OP(F80MUL);
DEF_OP(F80DIV);
DEF_OP(F80FYL2X);
DEF_OP(F80ATAN);
DEF_OP(F80FPREM1);
DEF_OP(F80FPREM);
DEF_OP(F80SCALE);
DEF_OP(F80CVT);
DEF_OP(F80CVTINT);
DEF_OP(F80CVTTO);
DEF_OP(F80CVTTOINT);
DEF_OP(F80ROUND);
DEF_OP(F80F2XM1);
DEF_OP(F80TAN);
DEF_OP(F80SQRT);
DEF_OP(F80SIN);
DEF_OP(F80COS);
DEF_OP(F80XTRACT_EXP);
DEF_OP(F80XTRACT_SIG);
DEF_OP(F80CMP);
DEF_OP(F80BCDLOAD);
DEF_OP(F80BCDSTORE);
#undef DEF_OP
template<typename unsigned_type, typename signed_type, typename float_type>
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
bool CompResult = false;
switch (Cond) {
case FEXCore::IR::COND_EQ:
CompResult = static_cast<unsigned_type>(Src1) == static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_NEQ:
CompResult = static_cast<unsigned_type>(Src1) != static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_SGE:
CompResult = static_cast<signed_type>(Src1) >= static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_SLT:
CompResult = static_cast<signed_type>(Src1) < static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_SGT:
CompResult = static_cast<signed_type>(Src1) > static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_SLE:
CompResult = static_cast<signed_type>(Src1) <= static_cast<signed_type>(Src2);
break;
case FEXCore::IR::COND_UGE:
CompResult = static_cast<unsigned_type>(Src1) >= static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_ULT:
CompResult = static_cast<unsigned_type>(Src1) < static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_UGT:
CompResult = static_cast<unsigned_type>(Src1) > static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_ULE:
CompResult = static_cast<unsigned_type>(Src1) <= static_cast<unsigned_type>(Src2);
break;
case FEXCore::IR::COND_FLU:
CompResult = reinterpret_cast<float_type&>(Src1) < reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FGE:
CompResult = reinterpret_cast<float_type&>(Src1) >= reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FLEU:
CompResult = reinterpret_cast<float_type&>(Src1) <= reinterpret_cast<float_type&>(Src2) || (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FGT:
CompResult = reinterpret_cast<float_type&>(Src1) > reinterpret_cast<float_type&>(Src2) && !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FU:
CompResult = (std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_FNU:
CompResult = !(std::isnan(reinterpret_cast<float_type&>(Src1)) || std::isnan(reinterpret_cast<float_type&>(Src2)));
break;
case FEXCore::IR::COND_MI:
case FEXCore::IR::COND_PL:
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
break;
}
return CompResult;
}
static uint8_t GetOpSize(FEXCore::IR::IRListView *CurrentIR, IR::OrderedNodeWrapper Node) {
auto IROp = CurrentIR->GetOp<FEXCore::IR::IROp_Header>(Node);
return IROp->Size;
}
};
} // namespace FEXCore::CPU
};
@@ -1,268 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
static inline void CacheLineFlush(char *Addr) {
#ifdef _M_X86_64
__asm volatile (
"clflush (%[Addr]);"
:: [Addr] "r" (Addr)
: "memory");
#else
__builtin___clear_cache(Addr, Addr+64);
#endif
}
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(LoadContext) {
auto Op = IROp->C<IR::IROp_LoadContext>();
uint8_t OpSize = IROp->Size;
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += Op->Offset;
#define LOAD_CTX(x, y) \
case x: { \
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
GD = *MemData; \
break; \
}
switch (OpSize) {
LOAD_CTX(1, uint8_t)
LOAD_CTX(2, uint16_t)
LOAD_CTX(4, uint32_t)
LOAD_CTX(8, uint64_t)
case 16: {
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
memcpy(GDP, MemData, OpSize);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
}
#undef LOAD_CTX
}
DEF_OP(StoreContext) {
auto Op = IROp->C<IR::IROp_StoreContext>();
uint8_t OpSize = IROp->Size;
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += Op->Offset;
void *MemData = reinterpret_cast<void*>(ContextPtr);
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
memcpy(MemData, Src, OpSize);
}
DEF_OP(LoadRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(StoreRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(LoadContextIndexed) {
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += Op->BaseOffset;
ContextPtr += Index * Op->Stride;
#define LOAD_CTX(x, y) \
case x: { \
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
GD = *MemData; \
break; \
}
switch (IROp->Size) {
LOAD_CTX(1, uint8_t)
LOAD_CTX(2, uint16_t)
LOAD_CTX(4, uint32_t)
LOAD_CTX(8, uint64_t)
case 16: {
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
memcpy(GDP, MemData, IROp->Size);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
}
#undef LOAD_CTX
}
DEF_OP(StoreContextIndexed) {
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += Op->BaseOffset;
ContextPtr += Index * Op->Stride;
void *MemData = reinterpret_cast<void*>(ContextPtr);
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
memcpy(MemData, Src, IROp->Size);
}
DEF_OP(SpillRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(FillRegister) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(LoadFlag) {
auto Op = IROp->C<IR::IROp_LoadFlag>();
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
ContextPtr += Op->Flag;
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
GD = *MemData;
}
DEF_OP(StoreFlag) {
auto Op = IROp->C<IR::IROp_StoreFlag>();
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
ContextPtr += Op->Flag;
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
*MemData = Arg;
}
DEF_OP(LoadMem) {
auto Op = IROp->C<IR::IROp_LoadMem>();
uint8_t OpSize = IROp->Size;
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
if (!Op->Offset.IsInvalid()) {
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
switch(Op->OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
}
}
memset(GDP, 0, 16);
switch (OpSize) {
case 1: {
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
GD = D->load();
break;
}
case 2: {
auto D = reinterpret_cast<const std::atomic<uint16_t>*>(MemData);
GD = D->load();
break;
}
case 4: {
auto D = reinterpret_cast<const std::atomic<uint32_t>*>(MemData);
GD = D->load();
break;
}
case 8: {
auto D = reinterpret_cast<const std::atomic<uint64_t>*>(MemData);
GD = D->load();
break;
}
default:
memcpy(GDP, MemData, IROp->Size);
break;
}
}
DEF_OP(StoreMem) {
auto Op = IROp->C<IR::IROp_StoreMem>();
uint8_t OpSize = IROp->Size;
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
if (!Op->Offset.IsInvalid()) {
auto Offset = *GetSrc<uintptr_t const*>(Data->SSAData, Op->Offset) * Op->OffsetScale;
switch(Op->OffsetType.Val) {
case IR::MEM_OFFSET_SXTX.Val: MemData += Offset; break;
case IR::MEM_OFFSET_UXTW.Val: MemData += (uint32_t)Offset; break;
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
}
}
switch (OpSize) {
case 1: {
reinterpret_cast<std::atomic<uint8_t>*>(MemData)->store(*GetSrc<uint8_t*>(Data->SSAData, Op->Value));
break;
}
case 2: {
reinterpret_cast<std::atomic<uint16_t>*>(MemData)->store(*GetSrc<uint16_t*>(Data->SSAData, Op->Value));
break;
}
case 4: {
reinterpret_cast<std::atomic<uint32_t>*>(MemData)->store(*GetSrc<uint32_t*>(Data->SSAData, Op->Value));
break;
}
case 8: {
reinterpret_cast<std::atomic<uint64_t>*>(MemData)->store(*GetSrc<uint64_t*>(Data->SSAData, Op->Value));
break;
}
default:
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), IROp->Size);
break;
}
}
DEF_OP(VLoadMemElement) {
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Header.Args[0]);
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[1]), 16);
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
MemData, Op->Header.ElementSize);
}
DEF_OP(VStoreMemElement) {
#define STORE_DATA(x, y) \
case x: { \
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Header.Args[0]); \
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Header.Args[1])[Op->Index], sizeof(y)); \
break; \
}
auto Op = IROp->C<IR::IROp_VStoreMemElement>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
STORE_DATA(1, uint8_t)
STORE_DATA(2, uint16_t)
STORE_DATA(4, uint32_t)
STORE_DATA(8, uint64_t)
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size"); break;
}
#undef STORE_DATA
}
DEF_OP(CacheLineClear) {
auto Op = IROp->C<IR::IROp_CacheLineClear>();
char *MemData = *GetSrc<char **>(Data->SSAData, Op->Addr);
// 64-byte cache line clear
CacheLineFlush(MemData);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,145 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
#ifdef _M_X86_64
#include <xmmintrin.h>
#endif
namespace FEXCore::CPU {
[[noreturn]]
static void StopThread(FEXCore::Core::InternalThreadState *Thread) {
Thread->CTX->StopThread(Thread);
LOGMAN_MSG_A_FMT("unreachable");
FEX_UNREACHABLE;
}
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(Fence) {
auto Op = IROp->C<IR::IROp_Fence>();
switch (Op->Fence) {
case IR::Fence_Load.Val:
std::atomic_thread_fence(std::memory_order_acquire);
break;
case IR::Fence_LoadStore.Val:
std::atomic_thread_fence(std::memory_order_seq_cst);
break;
case IR::Fence_Store.Val:
std::atomic_thread_fence(std::memory_order_release);
break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
}
}
DEF_OP(Break) {
auto Op = IROp->C<IR::IROp_Break>();
switch (Op->Reason) {
case FEXCore::IR::Break_Halt: // HLT
StopThread(Data->State);
break;
case FEXCore::IR::Break_InvalidInstruction:
tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
break;
default: LOGMAN_MSG_A_FMT("Unknown Break Reason: {}", Op->Reason); break;
}
}
DEF_OP(GetRoundingMode) {
uint32_t GuestRounding{};
#ifdef _M_ARM_64
uint64_t Tmp{};
__asm(R"(
mrs %[Tmp], FPCR;
)"
: [Tmp] "=r" (Tmp));
// Extract the rounding
// On ARM the ordering is different than on x86
GuestRounding |= ((Tmp >> 24) & 1) ? IR::ROUND_MODE_FLUSH_TO_ZERO : 0;
uint8_t RoundingMode = (Tmp >> 22) & 0b11;
if (RoundingMode == 0)
GuestRounding |= IR::ROUND_MODE_NEAREST;
else if (RoundingMode == 1)
GuestRounding |= IR::ROUND_MODE_POSITIVE_INFINITY;
else if (RoundingMode == 2)
GuestRounding |= IR::ROUND_MODE_NEGATIVE_INFINITY;
else if (RoundingMode == 3)
GuestRounding |= IR::ROUND_MODE_TOWARDS_ZERO;
#else
GuestRounding = _mm_getcsr();
// Extract the rounding
GuestRounding = (GuestRounding >> 13) & 0b111;
#endif
memcpy(GDP, &GuestRounding, sizeof(GuestRounding));
}
DEF_OP(SetRoundingMode) {
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
uint8_t GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
#ifdef _M_ARM_64
uint64_t HostRounding{};
__asm volatile(R"(
mrs %[Tmp], FPCR;
)"
: [Tmp] "=r" (HostRounding));
// Mask out the rounding
HostRounding &= ~(0b111 << 22);
HostRounding |= (GuestRounding & IR::ROUND_MODE_FLUSH_TO_ZERO) ? (1U << 24) : 0;
uint8_t RoundingMode = GuestRounding & 0b11;
if (RoundingMode == IR::ROUND_MODE_NEAREST)
HostRounding |= (0b00U << 22);
else if (RoundingMode == IR::ROUND_MODE_POSITIVE_INFINITY)
HostRounding |= (0b01U << 22);
else if (RoundingMode == IR::ROUND_MODE_NEGATIVE_INFINITY)
HostRounding |= (0b10U << 22);
else if (RoundingMode == IR::ROUND_MODE_TOWARDS_ZERO)
HostRounding |= (0b11U << 22);
__asm volatile(R"(
msr FPCR, %[Tmp];
)"
:: [Tmp] "r" (HostRounding));
#else
uint32_t HostRounding = _mm_getcsr();
// Cut out the host rounding mode
HostRounding &= ~(0b111 << 13);
// Insert our new rounding mode
HostRounding |= GuestRounding << 13;
_mm_setcsr(HostRounding);
#endif
}
DEF_OP(Print) {
auto Op = IROp->C<IR::IROp_Print>();
uint8_t OpSize = IROp->Size;
if (OpSize <= 8) {
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
LogMan::Msg::IFmt(">>>> Value in Arg: 0x{:x}, {}", Src, Src);
}
else if (OpSize == 16) {
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src0 = Src;
uint64_t Src1 = Src >> 64;
LogMan::Msg::IFmt(">>>> Value[0] in Arg: 0x{:x}, {}", Src0, Src0);
LogMan::Msg::IFmt(" Value[1] in Arg: 0x{:x}, {}", Src1, Src1);
}
else
LOGMAN_MSG_A_FMT("Unknown value size: {}", OpSize);
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -1,42 +0,0 @@
/*
$info$
tags: backend|interpreter
$end_info$
*/
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <cstdint>
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(ExtractElementPair) {
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
uintptr_t Src = GetSrc<uintptr_t>(Data->SSAData, Op->Header.Args[0]);
memcpy(GDP,
reinterpret_cast<void*>(Src + Op->Header.Size * Op->Element), Op->Header.Size);
}
DEF_OP(CreateElementPair) {
auto Op = IROp->C<IR::IROp_CreateElementPair>();
void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
uint8_t *Dst = GetDest<uint8_t*>(Data->SSAData, Node);
memcpy(Dst, Src_Lower, Op->Header.Size);
memcpy(Dst + Op->Header.Size, Src_Upper, Op->Header.Size);
}
DEF_OP(Mov) {
auto Op = IROp->C<IR::IROp_Mov>();
uint8_t OpSize = IROp->Size;
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), OpSize);
}
#undef DEF_OP
} // namespace FEXCore::CPU
File diff suppressed because it is too large. Load diff
+56 -81
View File
@@ -34,7 +34,7 @@ static int64_t LREM(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(TruncElementPair) {
auto Op = IROp->C<IR::IROp_TruncElementPair>();
@@ -46,7 +46,7 @@ DEF_OP(TruncElementPair) {
mov(Dst.second, Src.second);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", Op->Size); break;
default: LogMan::Msg::A("Unhandled Truncation size: %d", Op->Size); break;
}
}
@@ -59,7 +59,7 @@ DEF_OP(Constant) {
DEF_OP(EntrypointOffset) {
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
auto Constant = Entry + Op->Offset;
auto Constant = IR->GetHeader()->Entry + Op->Offset;
auto Dst = GetReg<RA_64>(Node);
LoadConstant(Dst, Constant);
}
@@ -95,7 +95,7 @@ DEF_OP(Add) {
case 8:
add(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), Const);
break;
default: LOGMAN_MSG_A_FMT("Unsupported Add size: {}", OpSize);
default: LogMan::Msg::A("Unsupported Add size: %d", OpSize);
}
} else {
switch (OpSize) {
@@ -105,7 +105,7 @@ DEF_OP(Add) {
case 8:
add(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unsupported Add size: {}", OpSize);
default: LogMan::Msg::A("Unsupported Add size: %d", OpSize);
}
}
}
@@ -121,7 +121,7 @@ DEF_OP(Sub) {
case 8:
sub(GRS(Node), GRS(Op->Header.Args[0].ID()), Const);
break;
default: LOGMAN_MSG_A_FMT("Unsupported Sub size: {}", OpSize);
default: LogMan::Msg::A("Unsupported Sub size: %d", OpSize);
}
} else {
switch (OpSize) {
@@ -131,7 +131,7 @@ DEF_OP(Sub) {
case 8:
sub(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unsupported Sub size: {}", OpSize);
default: LogMan::Msg::A("Unsupported Sub size: %d", OpSize);
}
}
@@ -147,7 +147,7 @@ DEF_OP(Neg) {
case 8:
neg(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unsupported Neg size: {}", OpSize);
default: LogMan::Msg::A("Unsupported Not size: %d", OpSize);
}
}
@@ -159,11 +159,12 @@ DEF_OP(Mul) {
switch (OpSize) {
case 4:
mul(Dst.W(), GetReg<RA_32>(Op->Header.Args[0].ID()), GetReg<RA_32>(Op->Header.Args[1].ID()));
sxtw(Dst, Dst);
break;
case 8:
mul(Dst, GetReg<RA_64>(Op->Header.Args[0].ID()), GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown Mul size: {}", OpSize);
default: LogMan::Msg::A("Unknown Sext size: %d", OpSize);
}
}
@@ -179,7 +180,7 @@ DEF_OP(UMul) {
case 8:
mul(Dst, GetReg<RA_64>(Op->Header.Args[0].ID()), GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown UMul size: {}", OpSize);
default: LogMan::Msg::A("Unknown Sext size: %d", OpSize);
}
}
@@ -216,7 +217,7 @@ DEF_OP(Div) {
sdiv(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unknown DIV Size: {}", Size); break;
default: LogMan::Msg::A("Unknown DIV Size: %d", Size); break;
}
}
@@ -243,7 +244,7 @@ DEF_OP(UDiv) {
udiv(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unknown UDIV Size: {}", Size); break;
default: LogMan::Msg::A("Unknown UDIV Size: %d", Size); break;
}
}
@@ -290,7 +291,7 @@ DEF_OP(Rem) {
msub(GetReg<RA_64>(Node), TMP1, Divisor, Dividend);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown REM Size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown REM Size: %d", OpSize); break;
}
}
@@ -332,7 +333,7 @@ DEF_OP(URem) {
msub(GetReg<RA_64>(Node), TMP1, Divisor, Dividend);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown UREM Size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown UREM Size: %d", OpSize); break;
}
}
@@ -344,12 +345,12 @@ DEF_OP(MulH) {
sxtw(TMP1, GetReg<RA_64>(Op->Header.Args[0].ID()));
sxtw(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
mul(TMP1, TMP1, TMP2);
ubfx(GetReg<RA_64>(Node), TMP1, 32, 32);
sbfx(GetReg<RA_64>(Node), TMP1, 32, 32);
break;
case 8:
smulh(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown Sext size: {}", OpSize);
default: LogMan::Msg::A("Unknown Sext size: %d", OpSize);
}
}
@@ -366,7 +367,7 @@ DEF_OP(UMulH) {
case 8:
umulh(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown Sext size: {}", OpSize);
default: LogMan::Msg::A("Unknown Sext size: %d", OpSize);
}
}
@@ -390,19 +391,6 @@ DEF_OP(And) {
}
}
DEF_OP(Andn) {
auto Op = IROp->C<IR::IROp_Andn>();
const auto& Lhs = Op->Header.Args[0];
const auto& Rhs = Op->Header.Args[1];
uint64_t Const{};
if (IsInlineConstant(Rhs, &Const)) {
bic(GRS(Node), GRS(Lhs.ID()), Const);
} else {
bic(GRS(Node), GRS(Lhs.ID()), GRS(Rhs.ID()));
}
}
DEF_OP(Xor) {
auto Op = IROp->C<IR::IROp_Xor>();
uint64_t Const;
@@ -475,7 +463,7 @@ DEF_OP(Ror) {
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled ROR size: {}", OpSize);
default: LogMan::Msg::A("Unhandled ROR size: %d", OpSize);
}
} else {
switch (OpSize) {
@@ -488,7 +476,7 @@ DEF_OP(Ror) {
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled ROR size: {}", OpSize);
default: LogMan::Msg::A("Unhandled ROR size: %d", OpSize);
}
}
}
@@ -507,7 +495,7 @@ DEF_OP(Extr) {
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled EXTR size: {}", OpSize);
default: LogMan::Msg::A("Unhandled EXTR size: %d", OpSize);
}
}
@@ -552,7 +540,7 @@ DEF_OP(LDiv) {
mov(GetReg<RA_64>(Node), x0);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LDIV Size: {}", Size); break;
default: LogMan::Msg::A("Unknown LDIV Size: %d", Size); break;
}
}
@@ -595,7 +583,7 @@ DEF_OP(LUDiv) {
mov(GetReg<RA_64>(Node), x0);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUDIV Size: {}", Size); break;
default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break;
}
}
@@ -648,7 +636,7 @@ DEF_OP(LRem) {
mov(GetReg<RA_64>(Node), x0);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LREM Size: {}", Size); break;
default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break;
}
}
@@ -698,7 +686,7 @@ DEF_OP(LURem) {
mov(GetReg<RA_64>(Node), x0);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUREM Size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown LUREM Size: %d", OpSize); break;
}
}
@@ -712,7 +700,7 @@ DEF_OP(Not) {
case 8:
mvn(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unsupported Not size: {}", OpSize);
default: LogMan::Msg::A("Unsupported Not size: %d", OpSize);
}
}
@@ -743,7 +731,7 @@ DEF_OP(Popcount) {
// fmov has zero extended, unused bytes are zero
addv(VTMP1.B(), VTMP1.V8B());
break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
default: LogMan::Msg::A("Unsupported Popcount size: %d", OpSize);
}
auto Dst = GetReg<RA_32>(Node);
@@ -792,7 +780,7 @@ DEF_OP(FindMSB) {
clz(Dst, GetReg<RA_64>(Op->Header.Args[0].ID()));
sub(Dst, TMP1, Dst);
break;
default: LOGMAN_MSG_A_FMT("Unknown FindMSB size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown REV size: %d", OpSize); break;
}
}
@@ -813,7 +801,7 @@ DEF_OP(FindTrailingZeros) {
rbit(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()));
clz(GetReg<RA_64>(Node), GetReg<RA_64>(Node));
break;
default: LOGMAN_MSG_A_FMT("Unknown FindTrailingZeros size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown size: %d", OpSize); break;
}
}
@@ -832,7 +820,7 @@ DEF_OP(CountLeadingZeroes) {
case 8:
clz(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown CountLeadingZeroes size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown size: %d", OpSize); break;
}
}
@@ -850,7 +838,7 @@ DEF_OP(Rev) {
case 8:
rev(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown REV size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown REV size: %d", OpSize); break;
}
}
@@ -872,14 +860,15 @@ DEF_OP(Bfi) {
bfi(TMP1, GetReg<RA_64>(Op->Header.Args[1].ID()), Op->lsb, Op->Width);
mov(GetReg<RA_64>(Node), TMP1);
break;
default: LOGMAN_MSG_A_FMT("Unknown BFI size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown BFI size: %d", OpSize); break;
}
}
DEF_OP(Bfe) {
auto Op = IROp->C<IR::IROp_Bfe>();
LOGMAN_THROW_A_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
LOGMAN_THROW_A_FMT(Op->Width != 0, "Invalid BFE width of 0");
uint8_t OpSize = IROp->Size;
LogMan::Throw::A(OpSize <= 8, "OpSize is too large for BFE: %d", OpSize);
LogMan::Throw::A(Op->Width != 0, "Invalid BFE width of 0");
auto Dst = GetReg<RA_64>(Node);
ubfx(Dst, GetReg<RA_64>(Op->Header.Args[0].ID()), Op->lsb, Op->Width);
@@ -893,7 +882,7 @@ DEF_OP(Sbfe) {
if (OpSize == 8) {
sbfx(Dst, GetReg<RA_64>(Op->Header.Args[0].ID()), Op->lsb, Op->Width);
} else {
LogMan::Msg::DFmt("Unimplemented Sbfe size");
LogMan::Msg::D("Unimplemented Sbfe size");
}
}
@@ -924,7 +913,7 @@ Condition MapSelectCC(IR::CondClassType Cond) {
case FEXCore::IR::COND_MI:
case FEXCore::IR::COND_PL:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
LogMan::Msg::A("Unsupported compare type");
return Condition::nv;
}
}
@@ -942,7 +931,7 @@ DEF_OP(Select) {
} else if (IsFPR(Op->Cmp1.ID())) {
fcmp(GRFCMP(Op->Cmp1.ID()), GRFCMP(Op->Cmp2.ID()));
} else {
LOGMAN_MSG_A_FMT("Select: Expected GPR or FPR");
LogMan::Msg::A("Select: Expected GPR or FPR");
}
auto cc = MapSelectCC(Op->Cond);
@@ -953,7 +942,7 @@ DEF_OP(Select) {
if (is_const_true || is_const_false) {
if (is_const_false != true || is_const_true != true || const_true != 1 || const_false != 0) {
LOGMAN_MSG_A_FMT("Select: Unsupported compare inline parameters");
LogMan::Msg::A("Select: Unsupported compare inline parameters");
}
cset(GRS(Node), cc);
} else {
@@ -977,53 +966,38 @@ DEF_OP(VExtractToGPR) {
case 8:
umov(GetReg<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()).V2D(), Op->Idx);
break;
default: LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
default: LogMan::Msg::A("Unhandled ExtractElementSize: %d", OpSize);
}
}
DEF_OP(Float_ToGPR_ZU) {
LogMan::Msg::D("Unimplemented");
}
DEF_OP(Float_ToGPR_ZS) {
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
aarch64::Register Dst{};
aarch64::VRegister Src{};
if (Op->SrcElementSize == 8) {
Src = GetSrc(Op->Header.Args[0].ID()).D();
if (Op->Header.ElementSize == 8) {
fcvtzs(GetReg<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()).D());
}
else {
Src = GetSrc(Op->Header.Args[0].ID()).S();
fcvtzs(GetReg<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()).S());
}
}
if (IROp->Size == 8) {
Dst = GetReg<RA_64>(Node);
}
else {
Dst = GetReg<RA_32>(Node);
}
fcvtzs(Dst, Src);
DEF_OP(Float_ToGPR_U) {
LogMan::Msg::D("Unimplemented");
}
DEF_OP(Float_ToGPR_S) {
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
aarch64::Register Dst{};
aarch64::VRegister Src{};
if (Op->SrcElementSize == 8) {
if (Op->Header.ElementSize == 8) {
frinti(VTMP1.D(), GetSrc(Op->Header.Args[0].ID()).D());
Src = VTMP1.D();
fcvtzs(GetReg<RA_64>(Node), VTMP1.D());
}
else {
frinti(VTMP1.S(), GetSrc(Op->Header.Args[0].ID()).S());
Src = VTMP1.S();
fcvtzs(GetReg<RA_32>(Node), VTMP1.S());
}
if (IROp->Size == 8) {
Dst = GetReg<RA_64>(Node);
}
else {
Dst = GetReg<RA_32>(Node);
}
fcvtzs(Dst, Src);
}
DEF_OP(FCmp) {
@@ -1040,7 +1014,7 @@ DEF_OP(FCmp) {
bool set = false;
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ)) {
LOGMAN_THROW_A_FMT(IR::FCMP_FLAG_EQ == 0, "IR::FCMP_FLAG_EQ must equal 0");
LogMan::Throw::A(IR::FCMP_FLAG_EQ == 0, "IR::FCMP_FLAG_EQ must equal 0");
// EQ or unordered
cset(Dst, Condition::eq); // Z = 1
csinc(Dst, Dst, xzr, Condition::vc); // IF !V ? Z : 1
@@ -1092,7 +1066,6 @@ void Arm64JITCore::RegisterALUHandlers() {
REGISTER_OP(UMULH, UMulH);
REGISTER_OP(OR, Or);
REGISTER_OP(AND, And);
REGISTER_OP(ANDN, Andn);
REGISTER_OP(XOR, Xor);
REGISTER_OP(LSHL, Lshl);
REGISTER_OP(LSHR, Lshr);
@@ -1115,7 +1088,9 @@ void Arm64JITCore::RegisterALUHandlers() {
REGISTER_OP(SBFE, Sbfe);
REGISTER_OP(SELECT, Select);
REGISTER_OP(VEXTRACTTOGPR, VExtractToGPR);
REGISTER_OP(FLOAT_TOGPR_ZU, Float_ToGPR_ZU);
REGISTER_OP(FLOAT_TOGPR_ZS, Float_ToGPR_ZS);
REGISTER_OP(FLOAT_TOGPR_U, Float_ToGPR_U);
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
REGISTER_OP(FCMP, FCmp);
+56 -108
View File
@@ -9,7 +9,7 @@ $end_info$
namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CASPair>();
uint8_t OpSize = IROp->Size;
@@ -34,7 +34,7 @@ DEF_OP(CASPair) {
mov(Dst.first, TMP3);
mov(Dst.second, TMP4);
break;
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
default: LogMan::Msg::A("Unsupported: %d", OpSize);
}
}
else {
@@ -44,7 +44,6 @@ DEF_OP(CASPair) {
aarch64::Label LoopNotExpected;
aarch64::Label LoopExpected;
bind(&LoopTop);
ldaxp(TMP2.W(), TMP3.W(), MemOperand(MemSrc));
cmp(TMP2.W(), Expected.first.W());
ccmp(TMP3.W(), Expected.second.W(), NoFlag, Condition::eq);
@@ -70,7 +69,6 @@ DEF_OP(CASPair) {
aarch64::Label LoopNotExpected;
aarch64::Label LoopExpected;
bind(&LoopTop);
ldaxp(TMP2.X(), TMP3.X(), MemOperand(MemSrc));
cmp(TMP2.X(), Expected.first.X());
ccmp(TMP3.X(), Expected.second.X(), NoFlag, Condition::eq);
@@ -91,7 +89,7 @@ DEF_OP(CASPair) {
bind(&LoopExpected);
break;
}
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
default: LogMan::Msg::A("Unsupported: %d", OpSize);
}
}
}
@@ -117,7 +115,7 @@ DEF_OP(CAS) {
case 2: casalh(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
case 4: casal(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
case 8: casal(TMP2.X(), Desired.X(), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
default: LogMan::Msg::A("Unsupported: %d", OpSize);
}
mov(GetReg<RA_64>(Node), TMP2);
}
@@ -208,7 +206,7 @@ DEF_OP(CAS) {
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", OpSize);
default: LogMan::Msg::A("Unhandled Atomic size: %d", OpSize);
}
}
}
@@ -219,17 +217,17 @@ DEF_OP(AtomicAdd) {
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (SupportsAtomics) {
switch (IROp->Size) {
switch (Op->Size) {
case 1: staddlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 2: staddlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 4: staddl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 8: staddl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -266,7 +264,7 @@ DEF_OP(AtomicAdd) {
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -278,17 +276,17 @@ DEF_OP(AtomicSub) {
if (SupportsAtomics) {
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (IROp->Size) {
switch (Op->Size) {
case 1: staddlb(TMP2.W(), MemOperand(MemSrc)); break;
case 2: staddlh(TMP2.W(), MemOperand(MemSrc)); break;
case 4: staddl(TMP2.W(), MemOperand(MemSrc)); break;
case 8: staddl(TMP2.X(), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -325,7 +323,7 @@ DEF_OP(AtomicSub) {
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -337,17 +335,17 @@ DEF_OP(AtomicAnd) {
if (SupportsAtomics) {
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (IROp->Size) {
switch (Op->Size) {
case 1: stclrlb(TMP2.W(), MemOperand(MemSrc)); break;
case 2: stclrlh(TMP2.W(), MemOperand(MemSrc)); break;
case 4: stclrl(TMP2.W(), MemOperand(MemSrc)); break;
case 8: stclrl(TMP2.X(), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -384,7 +382,7 @@ DEF_OP(AtomicAnd) {
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -395,17 +393,17 @@ DEF_OP(AtomicOr) {
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (SupportsAtomics) {
switch (IROp->Size) {
switch (Op->Size) {
case 1: stsetlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 2: stsetlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 4: stsetl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 8: stsetl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -442,7 +440,7 @@ DEF_OP(AtomicOr) {
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -453,17 +451,17 @@ DEF_OP(AtomicXor) {
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (SupportsAtomics) {
switch (IROp->Size) {
switch (Op->Size) {
case 1: steorlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 2: steorlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 4: steorl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 8: steorl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -500,7 +498,7 @@ DEF_OP(AtomicXor) {
cbnz(TMP2, &LoopTop);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -512,40 +510,41 @@ DEF_OP(AtomicSwap) {
if (SupportsAtomics) {
mov(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (IROp->Size) {
switch (Op->Size) {
case 1: swplb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: swplh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: swpl(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: swpl(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
mov(TMP3, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
stlxrb(TMP4.W(), GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
uxtb(GetReg<RA_32>(Node), TMP2.W());
uxtb(GetReg<RA_64>(Node), TMP2.W());
break;
}
case 2: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
stlxrh(TMP4.W(), GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
uxtw(GetReg<RA_32>(Node), TMP2.W());
uxtw(GetReg<RA_64>(Node), TMP2.W());
break;
}
case 4: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
stlxr(TMP4.W(), GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
break;
@@ -554,12 +553,12 @@ DEF_OP(AtomicSwap) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
stlxr(TMP4, GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
stlxr(TMP4, TMP3.X(), MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2.X());
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -569,17 +568,17 @@ DEF_OP(AtomicFetchAdd) {
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (SupportsAtomics) {
switch (IROp->Size) {
switch (Op->Size) {
case 1: ldaddalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldaddalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldaddal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldaddal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -620,7 +619,7 @@ DEF_OP(AtomicFetchAdd) {
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -631,17 +630,17 @@ DEF_OP(AtomicFetchSub) {
if (SupportsAtomics) {
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (IROp->Size) {
switch (Op->Size) {
case 1: ldaddalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldaddalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldaddal(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldaddal(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -682,7 +681,7 @@ DEF_OP(AtomicFetchSub) {
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -693,17 +692,17 @@ DEF_OP(AtomicFetchAnd) {
if (SupportsAtomics) {
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
switch (IROp->Size) {
switch (Op->Size) {
case 1: ldclralb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldclralh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldclral(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldclral(TMP2.X(), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -744,7 +743,7 @@ DEF_OP(AtomicFetchAnd) {
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -754,17 +753,17 @@ DEF_OP(AtomicFetchOr) {
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (SupportsAtomics) {
switch (IROp->Size) {
switch (Op->Size) {
case 1: ldsetalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldsetalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldsetal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldsetal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -805,7 +804,7 @@ DEF_OP(AtomicFetchOr) {
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
@@ -815,17 +814,17 @@ DEF_OP(AtomicFetchXor) {
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
if (SupportsAtomics) {
switch (IROp->Size) {
switch (Op->Size) {
case 1: ldeoralb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldeoralh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldeoral(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldeoral(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
else {
// TMP2-TMP3
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
@@ -866,61 +865,11 @@ DEF_OP(AtomicFetchXor) {
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled Atomic size: %d", Op->Size);
}
}
}
DEF_OP(AtomicFetchNeg) {
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
// TMP2-TMP3
switch (IROp->Size) {
case 1: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
neg(TMP3.W(), TMP2.W());
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
break;
}
case 2: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
neg(TMP3.W(), TMP2.W());
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
break;
}
case 4: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
neg(TMP3.W(), TMP2.W());
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
break;
}
case 8: {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
neg(TMP3, TMP2);
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
#undef DEF_OP
void Arm64JITCore::RegisterAtomicHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
@@ -937,7 +886,6 @@ void Arm64JITCore::RegisterAtomicHandlers() {
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
#undef REGISTER_OP
}
}
+53 -178
View File
@@ -4,30 +4,27 @@ tags: backend|arm64
$end_info$
*/
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "Interface/Core/InternalThreadState.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/MathUtils.h>
#include <Interface/HLE/Thunks/Thunks.h>
namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(GuestCallDirect) {
LogMan::Msg::DFmt("Unimplemented");
LogMan::Msg::D("Unimplemented");
}
DEF_OP(GuestCallIndirect) {
LogMan::Msg::DFmt("Unimplemented");
LogMan::Msg::D("Unimplemented");
}
DEF_OP(GuestReturn) {
LogMan::Msg::DFmt("Unimplemented");
LogMan::Msg::D("Unimplemented");
}
DEF_OP(SignalReturn) {
@@ -76,7 +73,7 @@ DEF_OP(ExitFunction) {
uint64_t NewRIP;
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
Literal l_BranchHost{ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress};
Literal l_BranchHost{Dispatcher->ExitFunctionLinkerAddress};
Literal l_BranchGuest{NewRIP};
ldr(x0, &l_BranchHost);
@@ -99,23 +96,30 @@ DEF_OP(ExitFunction) {
br(x1);
bind(&FullLookup);
LoadConstant(TMP1, ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress);
LoadConstant(TMP1, Dispatcher->AbsoluteLoopTopAddress);
str(RipReg, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
br(TMP1);
}
}
DEF_OP(Jump) {
const auto Op = IROp->C<IR::IROp_Jump>();
const auto ArgID = Op->Args(0).ID();
auto Op = IROp->C<IR::IROp_Jump>();
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
Label *TargetLabel;
auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID());
if (IsTarget == JumpTargets.end()) {
TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second;
}
else {
TargetLabel = &IsTarget->second;
}
PendingTargetLabel = TargetLabel;
}
#define GRCMP(Node) (Op->CompareSize == 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
#define GRFCMP(Node) (Op->CompareSize == 4 ? GetDst(Node).S() : GetDst(Node).D())
static Condition MapBranchCC(IR::CondClassType Cond) {
Condition MapBranchCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_EQ: return Condition::eq;
case FEXCore::IR::COND_NEQ: return Condition::ne;
@@ -130,7 +134,7 @@ static Condition MapBranchCC(IR::CondClassType Cond) {
case FEXCore::IR::COND_FLU: return Condition::lt;
case FEXCore::IR::COND_FGE: return Condition::ge;
case FEXCore::IR::COND_FLEU:return Condition::le;
case FEXCore::IR::COND_FGT: return Condition::gt;
case FEXCore::IR::COND_FGT: return Condition::hi;
case FEXCore::IR::COND_FU: return Condition::vs;
case FEXCore::IR::COND_FNU: return Condition::vc;
case FEXCore::IR::COND_VS:
@@ -138,7 +142,7 @@ static Condition MapBranchCC(IR::CondClassType Cond) {
case FEXCore::IR::COND_MI:
case FEXCore::IR::COND_PL:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
LogMan::Msg::A("Unsupported compare type");
return Condition::nv;
}
}
@@ -147,34 +151,51 @@ static Condition MapBranchCC(IR::CondClassType Cond) {
DEF_OP(CondJump) {
auto Op = IROp->C<IR::IROp_CondJump>();
Label *TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
Label *TrueTargetLabel;
Label *FalseTargetLabel;
auto TrueIter = JumpTargets.find(Op->TrueBlock.ID());
auto FalseIter = JumpTargets.find(Op->FalseBlock.ID());
if (TrueIter == JumpTargets.end()) {
TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
}
else {
TrueTargetLabel = &TrueIter->second;
}
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
bool isConst = IsInlineConstant(Op->Cmp2, &Const);
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
LogMan::Throw::A(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
cbz(GRCMP(Op->Cmp1.ID()), TrueTargetLabel);
} else if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_NEQ) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
LogMan::Throw::A(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
cbnz(GRCMP(Op->Cmp1.ID()), TrueTargetLabel);
} else {
if (IsGPR(Op->Cmp1.ID())) {
if (isConst) {
if (isConst)
cmp(GRCMP(Op->Cmp1.ID()), Const);
} else {
else
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
}
} else if (IsFPR(Op->Cmp1.ID())) {
fcmp(GRFCMP(Op->Cmp1.ID()), GRFCMP(Op->Cmp2.ID()));
} else {
LOGMAN_MSG_A_FMT("CondJump: Expected GPR or FPR");
LogMan::Msg::A("CondJump: Expected GPR or FPR");
}
b(TrueTargetLabel, MapBranchCC(Op->Cond));
}
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
if (FalseIter == JumpTargets.end()) {
FalseTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
}
else {
FalseTargetLabel = &FalseIter->second;
}
PendingTargetLabel = FalseTargetLabel;
}
DEF_OP(Syscall) {
@@ -212,152 +233,6 @@ DEF_OP(Syscall) {
mov(GetReg<RA_64>(Node), x0);
}
DEF_OP(InlineSyscall) {
auto Op = IROp->C<IR::IROp_InlineSyscall>();
// Arguments are passed as follows:
// X8: SyscallNumber - RA INTERSECT
// X0: Arg0 & Return
// X1: Arg1
// X2: Arg2
// X3: Arg3
// X4: Arg4 - RA INTERSECT
// X5: Arg5 - RA INTERSECT
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
const static std::array<vixl::aarch64::Register, FEXCore::HLE::SyscallArguments::MAX_ARGS-1> RegArgs = {{
x0, x1, x2, x3, x4, x5
}};
bool Intersects{};
// We always need to spill x8 since we can't know if it is live at this SSA location
uint32_t SpillMask = 1U << 8;
std::vector<vixl::aarch64::Register> IntersectRegs(FEXCore::HLE::SyscallArguments::MAX_ARGS);
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
if (Reg.GetCode() == x8.GetCode() ||
Reg.GetCode() == x4.GetCode() ||
Reg.GetCode() == x5.GetCode()) {
SpillMask |= (1U << Reg.GetCode());
Intersects = true;
}
}
// XXX: For some reason spilling only the x4, x5, and x8 registers was causing issues
// Come back to this once investigation reveals why it fails the gvisor ioctl test
// For now override to all GPRs
SpillMask = ~0U;
// Ordering is incredibly important here
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
// Only spill the registers that intersect with our usage
SpillStaticRegs(false, SpillMask);
// Now that we are spilled, store in the state that we are in a syscall
// Still without overwriting registers that matter
// 16bit LoadConstant to be a single instruction
// We must always spill at least one register (x8) so this value always has a bit set
// This gives the signal handler a value to check to see if we are in a syscall at all
LoadConstant(x0, SpillMask & 0xFFFF);
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
// Now that we have claimed to be a syscall we can set up the arguments
if (Intersects) {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
if (CTX->Config.Is64BitMode()) {
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
// In the case of intersection with x4, x5, or x8 then these are currently SRA
// for registers RAX, RBX, and RSI. Which have just been spilled
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
if (Reg.GetCode() == x8.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
}
else if (Reg.GetCode() == x4.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
}
else if (Reg.GetCode() == x5.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
}
}
}
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
if (CTX->Config.Is64BitMode()) {
auto Reg = GetReg<RA_64>(Op->Header.Args[i].ID());
// In the case of intersection with x4, x5, or x8 then these are currently SRA
// for registers RAX, RBX, and RSI. Which have just been spilled
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
if (Reg.GetCode() == x8.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
}
else if (Reg.GetCode() == x4.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
}
else if (Reg.GetCode() == x5.GetCode()) {
ldr(RegArgs[i], MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
}
else {
mov(RegArgs[i], Reg);
}
}
else {
auto Reg = GetReg<RA_32>(Op->Header.Args[i].ID());
if (Reg.GetCode() == w8.GetCode()) {
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSI])));
}
else if (Reg.GetCode() == w4.GetCode()) {
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX])));
}
else if (Reg.GetCode() == w5.GetCode()) {
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
}
else {
uxtw(RegArgs[i], Reg);
}
}
}
}
else {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
if (CTX->Config.Is64BitMode()) {
mov(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
}
else {
uxtw(RegArgs[i], GetReg<RA_32>(Op->Header.Args[i].ID()));
}
}
}
LoadConstant(x8, Op->HostSyscallNumber);
svc(0);
// On updated signal mask we can receive a signal RIGHT HERE
// Now that we are done in the syscall we need to carefully peel back the state
// First unspill the registers from before
FillStaticRegs(false, SpillMask);
// Now the registers we've spilled are back in their original host registers
// We can safely claim we are no longer in a syscall
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
// Result is now in x0
// Move result to its destination register
if (CTX->Config.Is64BitMode()) {
mov(GetReg<RA_64>(Node), x0);
}
else {
uxtw(GetReg<RA_64>(Node), w0);
}
}
DEF_OP(Thunk) {
auto Op = IROp->C<IR::IROp_Thunk>();
// Arguments are passed as follows:
@@ -379,20 +254,21 @@ DEF_OP(Thunk) {
FillStaticRegs(); // load from ctx after ra64 refill
}
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
const auto *OldCode = (const uint8_t *)&Op->CodeOriginalLow;
uint8_t *OldCode = (uint8_t *)&Op->CodeOriginalLow;
int len = Op->CodeLength;
int idx = 0;
LoadConstant(GetReg<RA_64>(Node), 0);
LoadConstant(x0, Entry + Op->Offset);
LoadConstant(x0, IR->GetHeader()->Entry + Op->Offset);
LoadConstant(x1, 1);
while (len >= 8)
{
ldr(x2, MemOperand(x0, idx));
LoadConstant(x3, *(const uint32_t *)(OldCode + idx));
LoadConstant(x3, *(uint32_t *)(OldCode + idx));
cmp(x2, x3);
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
len -= 8;
@@ -401,7 +277,7 @@ DEF_OP(ValidateCode) {
while (len >= 4)
{
ldr(w2, MemOperand(x0, idx));
LoadConstant(w3, *(const uint32_t *)(OldCode + idx));
LoadConstant(w3, *(uint32_t *)(OldCode + idx));
cmp(w2, w3);
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
len -= 4;
@@ -410,7 +286,7 @@ DEF_OP(ValidateCode) {
while (len >= 2)
{
ldrh(w2, MemOperand(x0, idx));
LoadConstant(w3, *(const uint16_t *)(OldCode + idx));
LoadConstant(w3, *(uint16_t *)(OldCode + idx));
cmp(w2, w3);
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
len -= 2;
@@ -419,7 +295,7 @@ DEF_OP(ValidateCode) {
while (len >= 1)
{
ldrb(w2, MemOperand(x0, idx));
LoadConstant(w3, *(const uint8_t *)(OldCode + idx));
LoadConstant(w3, *(uint8_t *)(OldCode + idx));
cmp(w2, w3);
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
len -= 1;
@@ -435,7 +311,7 @@ DEF_OP(RemoveCodeEntry) {
PushDynamicRegsAndLR();
mov(x0, STATE);
LoadConstant(x1, Entry);
LoadConstant(x1, IR->GetHeader()->Entry);
LoadConstant(x2, reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit));
SpillStaticRegs();
@@ -492,7 +368,6 @@ void Arm64JITCore::RegisterBranchHandlers() {
REGISTER_OP(JUMP, Jump);
REGISTER_OP(CONDJUMP, CondJump);
REGISTER_OP(SYSCALL, Syscall);
REGISTER_OP(INLINESYSCALL, InlineSyscall);
REGISTER_OP(THUNK, Thunk);
REGISTER_OP(VALIDATECODE, ValidateCode);
REGISTER_OP(REMOVECODEENTRY, RemoveCodeEntry);
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(VInsGPR) {
auto Op = IROp->C<IR::IROp_VInsGPR>();
mov(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
@@ -31,7 +31,7 @@ DEF_OP(VInsGPR) {
ins(GetDst(Node).V2D(), Op->Index, GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default: LogMan::Msg::A("Unknown Element Size: %d", Op->Header.ElementSize); break;
}
}
@@ -52,10 +52,14 @@ DEF_OP(VCastFromGPR) {
case 8:
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()).X());
break;
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Float_FromGPR_U) {
LogMan::Msg::A("Unimplemented");
}
DEF_OP(Float_FromGPR_S) {
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
@@ -91,7 +95,20 @@ DEF_OP(Float_FToF) {
fcvt(GetDst(Node).S(), GetSrc(Op->Header.Args[0].ID()).D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
default: LogMan::Msg::A("Unknown FCVT sizes: 0x%x", Conv);
}
}
DEF_OP(Vector_UToF) {
auto Op = IROp->C<IR::IROp_Vector_UToF>();
switch (Op->Header.ElementSize) {
case 4:
ucvtf(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
ucvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
@@ -104,7 +121,20 @@ DEF_OP(Vector_SToF) {
case 8:
scvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Vector_FToZU) {
auto Op = IROp->C<IR::IROp_Vector_FToZU>();
switch (Op->Header.ElementSize) {
case 4:
fcvtzu(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
fcvtzu(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
@@ -117,7 +147,22 @@ DEF_OP(Vector_FToZS) {
case 8:
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Vector_FToU) {
auto Op = IROp->C<IR::IROp_Vector_FToU>();
switch (Op->Header.ElementSize) {
case 4:
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
fcvtzu(GetDst(Node).V4S(), GetDst(Node).V4S());
break;
case 8:
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
fcvtzu(GetDst(Node).V2D(), GetDst(Node).V2D());
break;
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
@@ -132,7 +177,7 @@ DEF_OP(Vector_FToS) {
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
@@ -149,63 +194,7 @@ DEF_OP(Vector_FToF) {
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
}
}
DEF_OP(Vector_FToI) {
auto Op = IROp->C<IR::IROp_Vector_FToI>();
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (Op->Header.ElementSize) {
case 4:
frintn(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
frintn(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
}
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (Op->Header.ElementSize) {
case 4:
frintm(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
frintm(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
}
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (Op->Header.ElementSize) {
case 4:
frintp(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
frintp(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
}
break;
case FEXCore::IR::Round_Towards_Zero.Val:
switch (Op->Header.ElementSize) {
case 4:
frintz(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
frintz(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
}
break;
case FEXCore::IR::Round_Host.Val:
switch (Op->Header.ElementSize) {
case 4:
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
break;
case 8:
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
break;
}
break;
default: LogMan::Msg::A("Unknown Conversion Type : 0%04x", Conv); break;
}
}
@@ -214,13 +203,16 @@ void Arm64JITCore::RegisterConversionHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
REGISTER_OP(VINSGPR, VInsGPR);
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
REGISTER_OP(FLOAT_FROMGPR_U, Float_FromGPR_U);
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
REGISTER_OP(FLOAT_FTOF, Float_FToF);
REGISTER_OP(VECTOR_UTOF, Vector_UToF);
REGISTER_OP(VECTOR_STOF, Vector_SToF);
REGISTER_OP(VECTOR_FTOZU, Vector_FToZU);
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
REGISTER_OP(VECTOR_FTOU, Vector_FToU);
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
#undef REGISTER_OP
}
}
@@ -10,7 +10,7 @@ $end_info$
namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(AESImc) {
auto Op = IROp->C<IR::IROp_VAESImc>();
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(GetHostFlag) {
auto Op = IROp->C<IR::IROp_GetHostFlag>();
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), Op->Flag, 1);
+169 -113
View File
@@ -11,7 +11,6 @@ $end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/ArchHelpers/Arm64.h"
#include "Interface/Core/ArchHelpers/MContext.h"
@@ -23,8 +22,6 @@ $end_info$
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Core/UContext.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include <sys/mman.h>
@@ -42,12 +39,11 @@ void Arm64JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
using namespace vixl;
using namespace vixl::aarch64;
void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
FallbackInfo Info;
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_MSG_A_FMT("Unhandled IR Op: {}", FEXCore::IR::GetName(IROp->Op));
#endif
auto Name = FEXCore::IR::GetName(IROp->Op);
LogMan::Msg::A("Unhandled IR Op: %s", std::string(Name).c_str());
} else {
switch(Info.ABI) {
case FABI_VOID_U16:{
@@ -55,7 +51,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
LoadConstant(x1, (uintptr_t)Info.fn);
blr(x1);
@@ -112,12 +108,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
if (Info.ABI == FABI_F80_I16) {
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
}
else {
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
}
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
LoadConstant(x1, (uintptr_t)Info.fn);
blr(x1);
@@ -138,7 +129,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
LoadConstant(x2, (uintptr_t)Info.fn);
@@ -158,7 +149,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
LoadConstant(x2, (uintptr_t)Info.fn);
@@ -178,7 +169,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
LoadConstant(x2, (uintptr_t)Info.fn);
@@ -197,7 +188,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
LoadConstant(x2, (uintptr_t)Info.fn);
@@ -216,7 +207,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
LoadConstant(x2, (uintptr_t)Info.fn);
@@ -235,10 +226,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
umov(x3, GetSrc(IROp->Args[1].ID()).V2D(), 1);
LoadConstant(x4, (uintptr_t)Info.fn);
@@ -257,7 +248,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
LoadConstant(x2, (uintptr_t)Info.fn);
@@ -278,10 +269,10 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
umov(x1, GetSrc(IROp->Args[0].ID()).V2D(), 1);
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
umov(x3, GetSrc(IROp->Args[1].ID()).V2D(), 1);
LoadConstant(x4, (uintptr_t)Info.fn);
@@ -299,39 +290,118 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
case FABI_UNKNOWN:
default:
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}", FEXCore::IR::GetName(IROp->Op), Info.ABI);
#endif
break;
auto Name = FEXCore::IR::GetName(IROp->Op);
LogMan::Msg::A("Unhandled IR Fallback abi: %s %d", std::string(Name).c_str(), Info.ABI);
}
}
}
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
void Arm64JITCore::Op_NoOp(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
}
Arm64JITCore::CodeBuffer Arm64JITCore::AllocateNewCodeBuffer(size_t Size) {
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t*>(
FEXCore::Allocator::mmap(nullptr,
mmap(nullptr,
Buffer.Size,
PROT_READ | PROT_WRITE | PROT_EXEC,
MAP_PRIVATE | MAP_ANONYMOUS,
-1, 0));
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
LogMan::Throw::A(!!Buffer.Ptr, "Couldn't allocate code buffer");
Dispatcher->RegisterCodeBuffer(Buffer.Ptr, Buffer.Size);
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void Arm64JITCore::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
munmap(Buffer.Ptr, Buffer.Size);
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
}
bool Arm64JITCore::HandleSIGBUS(int Signal, void *info, void *ucontext) {
uint32_t *PC = (uint32_t*)ArchHelpers::Context::GetPc(ucontext);
uint32_t Instr = PC[0];
if (!Dispatcher->IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext))) {
// Wasn't a sigbus in JIT code
return false;
}
// 1 = 16bit
// 2 = 32bit
// 3 = 64bit
uint32_t Size = (Instr & 0xC000'0000) >> 30;
uint32_t AddrReg = (Instr >> 5) & 0x1F;
uint32_t DataReg = Instr & 0x1F;
uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
0b1011'0000'0000; // Inner shareable all
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
uint32_t LDR = 0b0011'1000'0111'1111'0110'1000'0000'0000;
LDR |= Size << 30;
LDR |= AddrReg << 5;
LDR |= DataReg;
PC[-1] = DMB;
PC[0] = LDR;
PC[1] = DMB;
// Back up one instruction and have another go
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
}
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
uint32_t STR = 0b0011'1000'0011'1111'0110'1000'0000'0000;
STR |= Size << 30;
STR |= AddrReg << 5;
STR |= DataReg;
PC[-1] = DMB;
PC[0] = STR;
PC[1] = DMB;
// Back up one instruction and have another go
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) - 4);
}
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASPAL_MASK) == FEXCore::ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
if (FEXCore::ArchHelpers::Arm64::HandleCASPAL(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
LogMan::Msg::E("Unhandled JIT SIGBUS CASPAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
return false;
}
}
else if ((Instr & FEXCore::ArchHelpers::Arm64::CASAL_MASK) == FEXCore::ArchHelpers::Arm64::CASAL_INST) { // CASAL
if (FEXCore::ArchHelpers::Arm64::HandleCASAL(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
LogMan::Msg::E("Unhandled JIT SIGBUS CASAL: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
return false;
}
}
else if ((Instr & FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_MASK) == FEXCore::ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
if (FEXCore::ArchHelpers::Arm64::HandleAtomicMemOp(ucontext, info, Instr)) {
// Skip this instruction now
ArchHelpers::Context::SetPc(ucontext, ArchHelpers::Context::GetPc(ucontext) + 4);
return true;
}
else {
uint8_t Op = (PC[0] >> 12) & 0xF;
LogMan::Msg::E("Unhandled JIT SIGBUS Atomic mem op 0x%02x: PC: %p Instruction: 0x%08x\n", Op, PC, PC[0]);
return false;
}
}
else {
LogMan::Msg::E("Unhandled JIT SIGBUS: PC: %p Instruction: 0x%08x\n", PC, PC[0]);
return false;
}
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(&PC[-1], 16);
return true;
}
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
: Arm64Emitter(0)
, CTX {ctx}
@@ -340,9 +410,9 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
DispatcherConfig config;
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
config.StaticRegisterAssignment = true;
Dispatcher = std::make_unique<Arm64Dispatcher>(CTX, ThreadState, config);
Dispatcher = new Arm64Dispatcher(CTX, ThreadState, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
}
@@ -354,7 +424,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
CurrentCodeBuffer = &InitialCodeBuffer;
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
RAPass = Thread->PassManager->GetRAPass();
#if DEBUG
Decoder.AppendVisitor(&Disasm)
@@ -396,38 +466,29 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
if (!CompileThread) {
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
ThreadSharedData.SignalReturnInstruction = Dispatcher->SignalHandlerReturnAddress;
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
ThreadSharedData.Dispatcher = Dispatcher.get();
// This will register the host signal handler per thread, which is fine
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
}, true);
});
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
if (!Core->Dispatcher->IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext))) {
// Wasn't a sigbus in JIT code
return false;
}
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Core->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
}, true);
return Core->HandleSIGBUS(Signal, info, ucontext);
});
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
}, true);
});
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
};
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
}
}
@@ -484,85 +545,77 @@ Arm64JITCore::~Arm64JITCore() {
FreeCodeBuffer(InitialCodeBuffer);
}
IR::PhysicalRegister Arm64JITCore::GetPhys(IR::NodeID Node) const {
static IR::PhysicalRegister GetPhys(IR::RegisterAllocationData *RAData, uint32_t Node) {
auto PhyReg = RAData->GetNodeRegister(Node);
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
LogMan::Throw::A(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa%d. Class: %d", Node, PhyReg.Class);
return PhyReg;
}
template<>
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(IR::NodeID Node) const {
auto Reg = GetPhys(Node);
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(uint32_t Node) {
auto Reg = GetPhys(RAData, Node);
if (Reg.Class == IR::GPRFixedClass.Val) {
return SRA64[Reg.Reg].W();
} else if (Reg.Class == IR::GPRClass.Val) {
return RA64[Reg.Reg].W();
} else {
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
}
FEX_UNREACHABLE;
__builtin_unreachable();
}
template<>
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(IR::NodeID Node) const {
auto Reg = GetPhys(Node);
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(uint32_t Node) {
auto Reg = GetPhys(RAData, Node);
if (Reg.Class == IR::GPRFixedClass.Val) {
return SRA64[Reg.Reg];
} else if (Reg.Class == IR::GPRClass.Val) {
return RA64[Reg.Reg];
} else {
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
}
FEX_UNREACHABLE;
__builtin_unreachable();
}
template<>
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_32>(IR::NodeID Node) const {
uint32_t Reg = GetPhys(Node).Reg;
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_32>(uint32_t Node) {
uint32_t Reg = GetPhys(RAData, Node).Reg;
return RA32Pair[Reg];
}
template<>
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_64>(IR::NodeID Node) const {
uint32_t Reg = GetPhys(Node).Reg;
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_64>(uint32_t Node) {
uint32_t Reg = GetPhys(RAData, Node).Reg;
return RA64Pair[Reg];
}
aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
auto Reg = GetPhys(Node);
aarch64::VRegister Arm64JITCore::GetSrc(uint32_t Node) {
auto Reg = GetPhys(RAData, Node);
if (Reg.Class == IR::FPRFixedClass.Val) {
return SRAFPR[Reg.Reg];
} else if (Reg.Class == IR::FPRClass.Val) {
return RAFPR[Reg.Reg];
} else {
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
}
FEX_UNREACHABLE;
__builtin_unreachable();
}
aarch64::VRegister Arm64JITCore::GetDst(IR::NodeID Node) const {
auto Reg = GetPhys(Node);
aarch64::VRegister Arm64JITCore::GetDst(uint32_t Node) {
auto Reg = GetPhys(RAData, Node);
if (Reg.Class == IR::FPRFixedClass.Val) {
return SRAFPR[Reg.Reg];
} else if (Reg.Class == IR::FPRClass.Val) {
return RAFPR[Reg.Reg];
} else {
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
LogMan::Throw::A(false, "Unexpected Class: %d", Reg.Class);
}
FEX_UNREACHABLE;
__builtin_unreachable();
}
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
if (OpHeader->Op == IR::IROps::OP_INLINECONSTANT) {
@@ -576,13 +629,13 @@ bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_
}
}
bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
if (Value) {
*Value = Entry + Op->Offset;
*Value = IR->GetHeader()->Entry + Op->Offset;
}
return true;
} else {
@@ -590,33 +643,34 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
}
}
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(IR::NodeID Node) const {
return FEXCore::IR::RegisterClassType {GetPhys(Node).Class};
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(uint32_t Node) {
return FEXCore::IR::RegisterClassType {GetPhys(RAData, Node).Class};
}
bool Arm64JITCore::IsFPR(IR::NodeID Node) const {
bool Arm64JITCore::IsFPR(uint32_t Node) {
auto Class = GetRegClass(Node);
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
}
bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
bool Arm64JITCore::IsGPR(uint32_t Node) {
auto Class = GetRegClass(Node);
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
}
void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
void *Arm64JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
using namespace aarch64;
JumpTargets.clear();
uint32_t SSACount = IR->GetSSACount();
this->Entry = Entry;
this->RAData = RAData;
auto HeaderOp = IR->GetHeader();
#ifndef NDEBUG
LoadConstant(x0, Entry);
LoadConstant(x0, HeaderOp->Entry);
#endif
this->IR = IR;
@@ -647,7 +701,8 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
// X1-X3 = Temp
// X4-r18 = RA
auto GuestEntry = GetCursorAddress<uint64_t>();
auto Buffer = GetBuffer();
auto Entry = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
if (CTX->GetGdbServerStatus()) {
aarch64::Label RunBlock;
@@ -664,17 +719,17 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
cbz(w0, &RunBlock);
{
// Make sure RIP is syncronized to the context
LoadConstant(x0, Entry);
LoadConstant(x0, HeaderOp->Entry);
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
// Stop the thread
LoadConstant(x0, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddressSpillSRA);
LoadConstant(x0, Dispatcher->ThreadPauseHandlerAddressSpillSRA);
br(x0);
}
bind(&RunBlock);
}
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
//LogMan::Throw::A(RAData->HasFullRA(), "Arm64 JIT only works with RA");
SpillSlots = RAData->SpillSlots();
@@ -691,14 +746,15 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
using namespace FEXCore::IR;
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
#endif
LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
{
const auto Node = IR->GetID(BlockNode);
const auto IsTarget = JumpTargets.try_emplace(Node).first;
uint32_t Node = IR->GetID(BlockNode);
auto IsTarget = JumpTargets.find(Node);
if (IsTarget == JumpTargets.end()) {
IsTarget = JumpTargets.try_emplace(Node).first;
}
// if there's a pending branch, and it is not fall-through
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
@@ -711,11 +767,11 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
}
if (DebugData) {
DebugData->Subblocks.push_back({GetCursorAddress<uintptr_t>(), 0, IR->GetID(BlockNode)});
DebugData->Subblocks.push_back({Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()), 0, IR->GetID(BlockNode)});
}
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
const auto ID = IR->GetID(CodeNode);
uint32_t ID = IR->GetID(CodeNode);
// Execute handler
OpHandler Handler = OpHandlers[IROp->Op];
@@ -723,7 +779,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
}
if (DebugData) {
DebugData->Subblocks.back().HostCodeSize = GetCursorAddress<uintptr_t>() - DebugData->Subblocks.back().HostCodeStart;
DebugData->Subblocks.back().HostCodeSize = Buffer->GetOffsetAddress<uintptr_t>(GetCursorOffset()) - DebugData->Subblocks.back().HostCodeStart;
}
}
@@ -736,16 +792,16 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
FinalizeCode();
auto CodeEnd = GetCursorAddress<uint64_t>();
CPU.EnsureIAndDCacheCoherency(reinterpret_cast<void*>(GuestEntry), CodeEnd - reinterpret_cast<uint64_t>(GuestEntry));
auto CodeEnd = Buffer->GetOffsetAddress<uint64_t>(GetCursorOffset());
CPU.EnsureIAndDCacheCoherency(reinterpret_cast<void*>(Entry), CodeEnd - reinterpret_cast<uint64_t>(Entry));
if (DebugData) {
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(CodeEnd) - reinterpret_cast<uintptr_t>(GuestEntry);
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(CodeEnd) - reinterpret_cast<uintptr_t>(Entry);
}
this->IR = nullptr;
return reinterpret_cast<void*>(GuestEntry);
return reinterpret_cast<void*>(Entry);
}
uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
@@ -755,13 +811,13 @@ uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuSt
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
if (!HostCode) {
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
//printf("ExitFunctionLink: Aborting, %lX not in cache\n", GuestRip);
Frame->State.rip = GuestRip;
return core->ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress;
return core->Dispatcher->AbsoluteLoopTopAddress;
}
uintptr_t branch = (uintptr_t)(record) - 8;
auto LinkerAddress = core->ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress;
auto LinkerAddress = core->Dispatcher->ExitFunctionLinkerAddress;
auto offset = HostCode/4 - branch/4;
if (IsInt26(offset)) {
@@ -797,7 +853,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuSt
return HostCode;
}
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
return std::make_unique<Arm64JITCore>(ctx, Thread, CompileThread);
FEXCore::CPU::CPUBackend *CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
return new Arm64JITCore(ctx, Thread, CompileThread);
}
}
+34 -60
View File
@@ -6,6 +6,7 @@ $end_info$
#pragma once
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
@@ -42,43 +43,33 @@ public:
size_t Size;
};
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread,
bool CompileThread);
explicit Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
~Arm64JITCore() override;
std::string GetName() override { return "JIT"; }
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]] std::string GetName() override { return "JIT"; }
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]] void *CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
bool NeedsOpDispatch() override { return true; }
void ClearCache() override;
bool HandleSIGBUS(int Signal, void *info, void *ucontext);
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
CodeBuffer AllocateNewCodeBuffer(size_t Size);
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
}
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
Dispatcher *Dispatcher;
Label *PendingTargetLabel;
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *ThreadState;
FEXCore::IR::IRListView const *IR;
uint64_t Entry;
std::map<IR::NodeID, aarch64::Label> JumpTargets;
std::map<IR::OrderedNodeWrapper::NodeOffsetType, aarch64::Label> JumpTargets;
/**
* @name Register Allocation
@@ -102,39 +93,33 @@ private:
constexpr static uint8_t RA_FPR = 2;
template<uint8_t RAType>
[[nodiscard]] aarch64::Register GetReg(IR::NodeID Node) const;
aarch64::Register GetReg(uint32_t Node);
template<>
[[nodiscard]] aarch64::Register GetReg<RA_32>(IR::NodeID Node) const;
aarch64::Register GetReg<RA_32>(uint32_t Node);
template<>
[[nodiscard]] aarch64::Register GetReg<RA_64>(IR::NodeID Node) const;
aarch64::Register GetReg<RA_64>(uint32_t Node);
template<uint8_t RAType>
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair(IR::NodeID Node) const;
std::pair<aarch64::Register, aarch64::Register> GetSrcPair(uint32_t Node);
template<>
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_32>(IR::NodeID Node) const;
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_32>(uint32_t Node);
template<>
[[nodiscard]] std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_64>(IR::NodeID Node) const;
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_64>(uint32_t Node);
[[nodiscard]] aarch64::VRegister GetSrc(IR::NodeID Node) const;
[[nodiscard]] aarch64::VRegister GetDst(IR::NodeID Node) const;
aarch64::VRegister GetSrc(uint32_t Node);
aarch64::VRegister GetDst(uint32_t Node);
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
FEXCore::IR::RegisterClassType GetRegClass(uint32_t Node);
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const;
bool IsFPR(uint32_t Node);
bool IsGPR(uint32_t Node);
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
MemOperand GenerateMemOperand(uint8_t AccessSize, aarch64::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
[[nodiscard]] MemOperand GenerateMemOperand(uint8_t AccessSize,
aarch64::Register Base,
IR::OrderedNodeWrapper Offset,
IR::MemOffsetType OffsetType,
uint8_t OffsetScale);
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr);
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value);
struct LiveRange {
uint32_t Begin;
@@ -175,18 +160,16 @@ private:
struct CompilerSharedData {
uint64_t SignalReturnInstruction{};
uint64_t UnimplementedInstructionAddress{};
uint32_t *SignalHandlerRefCounterPtr{};
FEXCore::CPU::Dispatcher *Dispatcher{};
};
CompilerSharedData ThreadSharedData;
IR::RegisterAllocationPass *RAPass;
IR::RegisterAllocationData *RAData;
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
using OpHandler = void (Arm64JITCore::*)(FEXCore::IR::IROp_Header *IROp, uint32_t Node);
std::array<OpHandler, FEXCore::IR::IROps::OP_LAST + 1> OpHandlers {};
void RegisterALUHandlers();
void RegisterAtomicHandlers();
void RegisterBranchHandlers();
@@ -197,7 +180,7 @@ private:
void RegisterMoveHandlers();
void RegisterVectorHandlers();
void RegisterEncryptionHandlers();
#define DEF_OP(x) void Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
///< Unhandled handler
DEF_OP(Unhandled);
@@ -225,7 +208,6 @@ private:
DEF_OP(UMulH);
DEF_OP(Or);
DEF_OP(And);
DEF_OP(Andn);
DEF_OP(Xor);
DEF_OP(Lshl);
DEF_OP(Lshr);
@@ -252,6 +234,7 @@ private:
DEF_OP(VExtractToGPR);
DEF_OP(Float_ToGPR_ZU);
DEF_OP(Float_ToGPR_ZS);
DEF_OP(Float_ToGPR_U);
DEF_OP(Float_ToGPR_S);
DEF_OP(FCmp);
@@ -269,7 +252,6 @@ private:
DEF_OP(AtomicFetchAnd);
DEF_OP(AtomicFetchOr);
DEF_OP(AtomicFetchXor);
DEF_OP(AtomicFetchNeg);
///< Branch ops
DEF_OP(GuestCallDirect);
@@ -281,7 +263,6 @@ private:
DEF_OP(Jump);
DEF_OP(CondJump);
DEF_OP(Syscall);
DEF_OP(InlineSyscall);
DEF_OP(Thunk);
DEF_OP(ValidateCode);
DEF_OP(RemoveCodeEntry);
@@ -290,13 +271,16 @@ private:
///< Conversion ops
DEF_OP(VInsGPR);
DEF_OP(VCastFromGPR);
DEF_OP(Float_FromGPR_U);
DEF_OP(Float_FromGPR_S);
DEF_OP(Float_FToF);
DEF_OP(Vector_UToF);
DEF_OP(Vector_SToF);
DEF_OP(Vector_FToZU);
DEF_OP(Vector_FToZS);
DEF_OP(Vector_FToU);
DEF_OP(Vector_FToS);
DEF_OP(Vector_FToF);
DEF_OP(Vector_FToI);
///< Flag ops
DEF_OP(GetHostFlag);
@@ -316,11 +300,8 @@ private:
DEF_OP(StoreMem);
DEF_OP(LoadMemTSO);
DEF_OP(StoreMemTSO);
DEF_OP(ParanoidLoadMemTSO);
DEF_OP(ParanoidStoreMemTSO);
DEF_OP(VLoadMemElement);
DEF_OP(VStoreMemElement);
DEF_OP(CacheLineClear);
///< Misc ops
DEF_OP(EndBlock);
@@ -346,7 +327,6 @@ private:
DEF_OP(SplatVector4);
DEF_OP(VMov);
DEF_OP(VAnd);
DEF_OP(VBic);
DEF_OP(VOr);
DEF_OP(VXor);
DEF_OP(VAdd);
@@ -357,10 +337,8 @@ private:
DEF_OP(VSQSub);
DEF_OP(VAddP);
DEF_OP(VAddV);
DEF_OP(VUMinV);
DEF_OP(VURAvg);
DEF_OP(VAbs);
DEF_OP(VPopcount);
DEF_OP(VFAdd);
DEF_OP(VFAddP);
DEF_OP(VFSub);
@@ -380,8 +358,6 @@ private:
DEF_OP(VSMax);
DEF_OP(VZip);
DEF_OP(VZip2);
DEF_OP(VUnZip);
DEF_OP(VUnZip2);
DEF_OP(VBSL);
DEF_OP(VCMPEQ);
DEF_OP(VCMPEQZ);
@@ -404,7 +380,6 @@ private:
DEF_OP(VInsElement);
DEF_OP(VInsScalarElement);
DEF_OP(VExtractElement);
DEF_OP(VDupElement);
DEF_OP(VExtr);
DEF_OP(VSLI);
DEF_OP(VSRI);
@@ -427,7 +402,6 @@ private:
DEF_OP(VSMull);
DEF_OP(VUMull2);
DEF_OP(VSMull2);
DEF_OP(VUABDL);
DEF_OP(VTBL1);
///< Encryption ops
+80 -238
View File
@@ -5,13 +5,12 @@ $end_info$
*/
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include <FEXCore/Utils/CompilerDefs.h>
namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(LoadContext) {
auto Op = IROp->C<IR::IROp_LoadContext>();
@@ -30,7 +29,7 @@ DEF_OP(LoadContext) {
case 8:
ldr(GetReg<RA_64>(Node), MemOperand(STATE, Op->Offset));
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
default: LogMan::Msg::A("Unhandled LoadContext size: %d", OpSize);
}
}
else {
@@ -51,7 +50,7 @@ DEF_OP(LoadContext) {
case 16:
ldr(Dst, MemOperand(STATE, Op->Offset));
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
default: LogMan::Msg::A("Unhandled LoadContext size: %d", OpSize);
}
}
}
@@ -73,7 +72,7 @@ DEF_OP(StoreContext) {
case 8:
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
default: LogMan::Msg::A("Unhandled StoreContext size: %d", OpSize);
}
}
else {
@@ -94,7 +93,7 @@ DEF_OP(StoreContext) {
case 16:
str(Src, MemOperand(STATE, Op->Offset));
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
default: LogMan::Msg::A("Unhandled LoadContext size: %d", OpSize);
}
}
}
@@ -107,29 +106,29 @@ DEF_OP(LoadRegister) {
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0])) / 8;
auto regOffs = Op->Offset & 7;
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
LogMan::Throw::A(regId < SRA64.size(), "out of range regId");
auto reg = SRA64[regId];
switch(Op->Header.Size) {
case 1:
LOGMAN_THROW_A_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0 || regOffs == 1, "unexpected regOffs");
ubfx(GetReg<RA_64>(Node), reg, regOffs * 8, 8);
break;
case 2:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
ubfx(GetReg<RA_64>(Node), reg, 0, 16);
break;
case 4:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
mov(GetReg<RA_32>(Node), reg.W());
break;
case 8:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
mov(GetReg<RA_64>(Node), reg);
break;
@@ -138,24 +137,24 @@ DEF_OP(LoadRegister) {
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
auto regOffs = Op->Offset & 15;
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
LogMan::Throw::A(regId < SRAFPR.size(), "out of range regId");
auto guest = SRAFPR[regId];
auto host = GetSrc(Node);
switch(Op->Header.Size) {
case 1:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
mov(host.B(), guest.B());
break;
case 2:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
fmov(host.H(), guest.H());
break;
case 4:
LOGMAN_THROW_A_FMT((regOffs & 3) == 0, "unexpected regOffs");
LogMan::Throw::A((regOffs & 3) == 0, "unexpected regOffs");
if (regOffs == 0) {
if (host.GetCode() != guest.GetCode())
fmov(host.S(), guest.S());
@@ -165,7 +164,7 @@ DEF_OP(LoadRegister) {
break;
case 8:
LOGMAN_THROW_A_FMT((regOffs & 7) == 0, "unexpected regOffs");
LogMan::Throw::A((regOffs & 7) == 0, "unexpected regOffs");
if (regOffs == 0) {
if (host.GetCode() != guest.GetCode())
mov(host.D(), guest.D());
@@ -175,13 +174,13 @@ DEF_OP(LoadRegister) {
break;
case 16:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
if (host.GetCode() != guest.GetCode())
mov(host.Q(), guest.Q());
break;
}
} else {
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
LogMan::Throw::A(false, "Unhandled Op->Class %d", Op->Class);
}
}
@@ -192,28 +191,28 @@ DEF_OP(StoreRegister) {
auto regId = Op->Offset / 8 - 1;
auto regOffs = Op->Offset & 7;
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
LogMan::Throw::A(regId < SRA64.size(), "out of range regId");
auto reg = SRA64[regId];
switch(Op->Header.Size) {
case 1:
LOGMAN_THROW_A_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0 || regOffs == 1, "unexpected regOffs");
bfi(reg, GetReg<RA_64>(Op->Value.ID()), regOffs * 8, 8);
break;
case 2:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 16);
break;
case 4:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 32);
break;
case 8:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
if (GetReg<RA_64>(Op->Value.ID()).GetCode() != reg.GetCode())
mov(reg, GetReg<RA_64>(Op->Value.ID()));
break;
@@ -222,7 +221,7 @@ DEF_OP(StoreRegister) {
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
auto regOffs = Op->Offset & 15;
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
LogMan::Throw::A(regId < SRAFPR.size(), "regId out of range");
auto guest = SRAFPR[regId];
auto host = GetSrc(Op->Value.ID());
@@ -233,35 +232,35 @@ DEF_OP(StoreRegister) {
break;
case 2:
LOGMAN_THROW_A_FMT((regOffs & 1) == 0, "unexpected regOffs");
LogMan::Throw::A((regOffs & 1) == 0, "unexpected regOffs");
ins(guest.V8H(), regOffs/2, host.V8H(), 0);
break;
case 4:
LOGMAN_THROW_A_FMT((regOffs & 3) == 0, "unexpected regOffs");
LogMan::Throw::A((regOffs & 3) == 0, "unexpected regOffs");
ins(guest.V4S(), regOffs/4, host.V4S(), 0);
break;
case 8:
LOGMAN_THROW_A_FMT((regOffs & 7) == 0, "unexpected regOffs");
LogMan::Throw::A((regOffs & 7) == 0, "unexpected regOffs");
ins(guest.V2D(), regOffs / 8, host.V2D(), 0);
break;
case 16:
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
LogMan::Throw::A(regOffs == 0, "unexpected regOffs");
if (guest.GetCode() != host.GetCode())
mov(guest.Q(), host.Q());
break;
}
} else {
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
LogMan::Throw::A(false, "Unhandled Op->Class %d", Op->Class);
}
}
DEF_OP(LoadContextIndexed) {
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
size_t size = IROp->Size;
size_t size = Op->Size;
auto index = GetReg<RA_64>(Op->Header.Args[0].ID());
if (Op->Class == FEXCore::IR::GPRClass) {
@@ -288,17 +287,15 @@ DEF_OP(LoadContextIndexed) {
ldr(GetReg<RA_64>(Node), MemOperand(TMP1, Op->BaseOffset));
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
}
break;
}
case 16:
LOGMAN_MSG_A_FMT("Invalid Class load of size 16");
LogMan::Msg::A("Invalid Class load of size 16");
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed stride: {}", Op->Stride);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
}
}
else {
@@ -335,21 +332,19 @@ DEF_OP(LoadContextIndexed) {
}
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
}
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed stride: {}", Op->Stride);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
}
}
}
DEF_OP(StoreContextIndexed) {
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
size_t size = IROp->Size;
size_t size = Op->Size;
auto index = GetReg<RA_64>(Op->Header.Args[1].ID());
if (Op->Class == FEXCore::IR::GPRClass) {
@@ -378,17 +373,15 @@ DEF_OP(StoreContextIndexed) {
str(value, MemOperand(TMP1, Op->BaseOffset));
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
}
break;
}
case 16:
LOGMAN_MSG_A_FMT("Invalid Class store of size 16");
LogMan::Msg::A("Invalid Class load of size 16");
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed stride: {}", Op->Stride);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
}
}
else {
@@ -427,14 +420,12 @@ DEF_OP(StoreContextIndexed) {
}
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
}
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed stride: {}", Op->Stride);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
}
}
}
@@ -442,7 +433,7 @@ DEF_OP(StoreContextIndexed) {
DEF_OP(SpillRegister) {
auto Op = IROp->C<IR::IROp_SpillRegister>();
uint8_t OpSize = IROp->Size;
uint32_t SlotOffset = Op->Slot * 16;
uint32_t SlotOffset = Op->Slot * 16 + 16;
if (Op->Class == FEXCore::IR::GPRClass) {
switch (OpSize) {
@@ -462,7 +453,7 @@ DEF_OP(SpillRegister) {
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
}
} else if (Op->Class == FEXCore::IR::FPRClass) {
switch (OpSize) {
@@ -478,17 +469,17 @@ DEF_OP(SpillRegister) {
str(GetSrc(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
}
} else {
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
LogMan::Msg::A("Unhandled SpillRegister class: %d", Op->Class.Val);
}
}
DEF_OP(FillRegister) {
auto Op = IROp->C<IR::IROp_FillRegister>();
uint8_t OpSize = IROp->Size;
uint32_t SlotOffset = Op->Slot * 16;
uint32_t SlotOffset = Op->Slot * 16 + 16;
if (Op->Class == FEXCore::IR::GPRClass) {
switch (OpSize) {
@@ -508,7 +499,7 @@ DEF_OP(FillRegister) {
ldr(GetReg<RA_64>(Node), MemOperand(sp, SlotOffset));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
}
} else if (Op->Class == FEXCore::IR::FPRClass) {
switch (OpSize) {
@@ -524,10 +515,10 @@ DEF_OP(FillRegister) {
ldr(GetDst(Node), MemOperand(sp, SlotOffset));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
}
} else {
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
LogMan::Msg::A("Unhandled FillRegister class: %d", Op->Class.Val);
}
}
@@ -547,7 +538,7 @@ MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Registe
return MemOperand(Base);
} else {
if (OffsetScale != 1 && OffsetScale != AccessSize) {
LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetScale: {}", OffsetScale);
LogMan::Msg::A("Unhandled GenerateMemOperand OffsetScale: %d", OffsetScale);
}
uint64_t Const;
if (IsInlineConstant(Offset, &Const)) {
@@ -559,23 +550,22 @@ MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Registe
case IR::MEM_OFFSET_UXTW.Val: return MemOperand(Base, RegOffset.W(), Extend::UXTW, (int)std::log2(OffsetScale) );
case IR::MEM_OFFSET_SXTW.Val: return MemOperand(Base, RegOffset.W(), Extend::SXTW, (int)std::log2(OffsetScale) );
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
default: LogMan::Msg::A("Unhandled GenerateMemOperand OffsetType: %d", OffsetType.Val); break;
}
}
}
FEX_UNREACHABLE;
__builtin_unreachable();
}
DEF_OP(LoadMem) {
auto Op = IROp->C<IR::IROp_LoadMem>();
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
auto MemSrc = GenerateMemOperand(Op->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class == FEXCore::IR::GPRClass) {
auto Dst = GetReg<RA_64>(Node);
switch (IROp->Size) {
switch (Op->Size) {
case 1:
ldrb(Dst, MemSrc);
break;
@@ -588,12 +578,12 @@ DEF_OP(LoadMem) {
case 8:
ldr(Dst, MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
}
}
else {
auto Dst = GetDst(Node);
switch (IROp->Size) {
switch (Op->Size) {
case 1:
ldr(Dst.B(), MemSrc);
break;
@@ -609,7 +599,7 @@ DEF_OP(LoadMem) {
case 16:
ldr(Dst, MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
}
}
}
@@ -620,11 +610,11 @@ DEF_OP(LoadMemTSO) {
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
if (!Op->Offset.IsInvalid()) {
LOGMAN_MSG_A_FMT("LoadMemTSO: No offset allowed");
LogMan::Msg::A("LoadMemTSO: No offset allowed");
}
if (SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
if (IROp->Size == 1) {
if (Op->Size == 1) {
// 8bit load is always aligned to natural alignment
auto Dst = GetReg<RA_64>(Node);
ldaprb(Dst, MemSrc);
@@ -633,7 +623,7 @@ DEF_OP(LoadMemTSO) {
// Aligned
auto Dst = GetReg<RA_64>(Node);
nop();
switch (IROp->Size) {
switch (Op->Size) {
case 2:
ldaprh(Dst, MemSrc);
break;
@@ -643,13 +633,13 @@ DEF_OP(LoadMemTSO) {
case 8:
ldapr(Dst, MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
}
nop();
}
}
else if (Op->Class == FEXCore::IR::GPRClass) {
if (IROp->Size == 1) {
if (Op->Size == 1) {
// 8bit load is always aligned to natural alignment
auto Dst = GetReg<RA_64>(Node);
ldarb(Dst, MemSrc);
@@ -658,7 +648,7 @@ DEF_OP(LoadMemTSO) {
// Aligned
auto Dst = GetReg<RA_64>(Node);
nop();
switch (IROp->Size) {
switch (Op->Size) {
case 2:
ldarh(Dst, MemSrc);
break;
@@ -668,7 +658,7 @@ DEF_OP(LoadMemTSO) {
case 8:
ldar(Dst, MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
}
nop();
}
@@ -676,7 +666,7 @@ DEF_OP(LoadMemTSO) {
else {
dmb(InnerShareable, BarrierAll);
auto Dst = GetDst(Node);
switch (IROp->Size) {
switch (Op->Size) {
case 2:
ldr(Dst.H(), MemSrc);
break;
@@ -689,7 +679,7 @@ DEF_OP(LoadMemTSO) {
case 16:
ldr(Dst, MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
}
dmb(InnerShareable, BarrierAll);
}
@@ -700,10 +690,10 @@ DEF_OP(StoreMem) {
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
auto MemSrc = GenerateMemOperand(Op->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class == FEXCore::IR::GPRClass) {
switch (IROp->Size) {
switch (Op->Size) {
case 1:
strb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
break;
@@ -716,12 +706,12 @@ DEF_OP(StoreMem) {
case 8:
str(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
}
}
else {
auto Src = GetSrc(Op->Header.Args[1].ID());
switch (IROp->Size) {
switch (Op->Size) {
case 1:
str(Src.B(), MemSrc);
break;
@@ -737,7 +727,7 @@ DEF_OP(StoreMem) {
case 16:
str(Src, MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
}
}
}
@@ -747,17 +737,17 @@ DEF_OP(StoreMemTSO) {
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
if (!Op->Offset.IsInvalid()) {
LOGMAN_MSG_A_FMT("StoreMemTSO: No offset allowed");
LogMan::Msg::A("StoreMemTSO: No offset allowed");
}
if (Op->Class == FEXCore::IR::GPRClass) {
if (IROp->Size == 1) {
if (Op->Size == 1) {
// 8bit load is always aligned to natural alignment
stlrb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
}
else {
nop();
switch (IROp->Size) {
switch (Op->Size) {
case 2:
stlrh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
break;
@@ -767,7 +757,7 @@ DEF_OP(StoreMemTSO) {
case 8:
stlr(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
}
nop();
}
@@ -775,7 +765,7 @@ DEF_OP(StoreMemTSO) {
else {
dmb(InnerShareable, BarrierAll);
auto Src = GetSrc(Op->Header.Args[1].ID());
switch (IROp->Size) {
switch (Op->Size) {
case 1:
str(Src.B(), MemSrc);
break;
@@ -791,159 +781,18 @@ DEF_OP(StoreMemTSO) {
case 16:
str(Src, MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
}
dmb(InnerShareable, BarrierAll);
}
}
DEF_OP(ParanoidLoadMemTSO) {
auto Op = IROp->C<IR::IROp_LoadMemTSO>();
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
if (!Op->Offset.IsInvalid()) {
LOGMAN_MSG_A_FMT("ParanoidLoadMemTSO: No offset allowed");
}
if (Op->Class == FEXCore::IR::GPRClass) {
if (IROp->Size == 1) {
// 8bit load is always aligned to natural alignment
auto Dst = GetReg<RA_64>(Node);
ldarb(Dst, MemSrc);
}
else {
auto Dst = GetReg<RA_64>(Node);
switch (IROp->Size) {
case 2:
ldarh(Dst, MemSrc);
break;
case 4:
ldar(Dst.W(), MemSrc);
break;
case 8:
ldar(Dst, MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", IROp->Size);
}
}
}
else {
auto Dst = GetDst(Node);
switch (IROp->Size) {
case 2:
ldarh(TMP1.W(), MemSrc);
fmov(Dst.H(), TMP1.W());
break;
case 4:
ldar(TMP1.W(), MemSrc);
fmov(Dst.S(), TMP1.W());
break;
case 8:
ldar(TMP1, MemSrc);
fmov(Dst.D(), TMP1);
break;
case 16:
nop();
ldaxp(TMP1, TMP2, MemSrc);
clrex();
mov(Dst.V2D(), 0, TMP1);
mov(Dst.V2D(), 1, TMP2);
break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", IROp->Size);
}
}
}
DEF_OP(ParanoidStoreMemTSO) {
auto Op = IROp->C<IR::IROp_StoreMemTSO>();
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
if (!Op->Offset.IsInvalid()) {
LOGMAN_MSG_A_FMT("ParanoidStoreMemTSO: No offset allowed");
}
if (Op->Class == FEXCore::IR::GPRClass) {
if (IROp->Size == 1) {
// 8bit load is always aligned to natural alignment
stlrb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
}
else {
switch (IROp->Size) {
case 2:
stlrh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
break;
case 4:
stlr(GetReg<RA_32>(Op->Header.Args[1].ID()), MemSrc);
break;
case 8:
stlr(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", IROp->Size);
}
}
}
else {
auto Src = GetSrc(Op->Header.Args[1].ID());
if (IROp->Size == 1) {
// 8bit load is always aligned to natural alignment
mov(TMP1.W(), Src.V16B(), 0);
stlrb(TMP1, MemSrc);
}
else {
switch (IROp->Size) {
case 2:
mov(TMP1.W(), Src.V8H(), 0);
stlrh(TMP1, MemSrc);
break;
case 4:
mov(TMP1.W(), Src.V4S(), 0);
stlr(TMP1.W(), MemSrc);
break;
case 8:
mov(TMP1, Src.V2D(), 0);
stlr(TMP1, MemSrc);
break;
case 16: {
// Move vector to GPRs
mov(TMP1, Src.V2D(), 0);
mov(TMP2, Src.V2D(), 1);
Label B;
bind(&B);
// ldaxp must not have both the destination registers be the same
ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS. Overwritten with DMB
stlxp(TMP3, TMP1, TMP2, MemSrc); // <- Can also hit SIGBUS
cbnz(TMP3, &B); // < Overwritten with DMB
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", IROp->Size);
}
}
}
}
DEF_OP(VLoadMemElement) {
LOGMAN_MSG_A_FMT("Unimplemented");
LogMan::Msg::A("Unimplemented");
}
DEF_OP(VStoreMemElement) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(CacheLineClear) {
auto Op = IROp->C<IR::IROp_CacheLineClear>();
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
// Clear dcache only
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
mov(TMP1, MemReg);
for (size_t i = 0; i < std::max(1U, DCacheLineSize / 64U); ++i) {
dc(DataCacheOp::CVAU, TMP1);
add(TMP1, TMP1, DCacheLineSize);
}
dsb(InnerShareable, BarrierAll);
LogMan::Msg::A("Unimplemented");
}
#undef DEF_OP
@@ -961,17 +810,10 @@ void Arm64JITCore::RegisterMemoryHandlers() {
REGISTER_OP(STOREFLAG, StoreFlag);
REGISTER_OP(LOADMEM, LoadMem);
REGISTER_OP(STOREMEM, StoreMem);
if (ParanoidTSO()) {
REGISTER_OP(LOADMEMTSO, ParanoidLoadMemTSO);
REGISTER_OP(STOREMEMTSO, ParanoidStoreMemTSO);
}
else {
REGISTER_OP(LOADMEMTSO, LoadMemTSO);
REGISTER_OP(STOREMEMTSO, StoreMemTSO);
}
REGISTER_OP(LOADMEMTSO, LoadMemTSO);
REGISTER_OP(STOREMEMTSO, StoreMemTSO);
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
#undef REGISTER_OP
}
}
+10 -49
View File
@@ -7,17 +7,10 @@ $end_info$
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
static void PrintValue(uint64_t Value) {
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
}
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
}
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(Fence) {
auto Op = IROp->C<IR::IROp_Fence>();
@@ -31,46 +24,36 @@ DEF_OP(Fence) {
case IR::Fence_Store.Val:
dmb(FullSystem, BarrierWrites);
break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
default: LogMan::Msg::A("Unknown Fence: %d", Op->Fence); break;
}
}
DEF_OP(Break) {
auto Op = IROp->C<IR::IROp_Break>();
switch (Op->Reason) {
case FEXCore::IR::Break_Unimplemented: // Hard fault
case FEXCore::IR::Break_Interrupt: // Guest ud2
case FEXCore::IR::Break_Overflow: // overflow
case 0: // Hard fault
case 5: // Guest ud2
hlt(4);
break;
case FEXCore::IR::Break_Halt: { // HLT
case 4: { // HLT
// Time to quit
// Set our stack to the starting stack location
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
add(sp, TMP1, 0);
// Now we need to jump to the thread stop handler
LoadConstant(TMP1, ThreadSharedData.Dispatcher->ThreadStopHandlerAddressSpillSRA);
LoadConstant(TMP1, Dispatcher->ThreadStopHandlerAddressSpillSRA);
br(TMP1);
break;
}
case FEXCore::IR::Break_Interrupt3: { // INT3
case 6: { // INT3
ResetStack();
LoadConstant(TMP1, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddressSpillSRA);
LoadConstant(TMP1, Dispatcher->ThreadPauseHandlerAddressSpillSRA);
br(TMP1);
break;
}
case FEXCore::IR::Break_InvalidInstruction:
{
ResetStack();
LoadConstant(TMP1, ThreadSharedData.Dispatcher->UnimplementedInstructionAddress);
br(TMP1);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
default: LogMan::Msg::A("Unknown Break reason: %d", Op->Reason);
}
}
@@ -132,28 +115,6 @@ DEF_OP(SetRoundingMode) {
msr(FPCR, TMP1);
}
DEF_OP(Print) {
auto Op = IROp->C<IR::IROp_Print>();
PushDynamicRegsAndLR();
if (IsGPR(Op->Header.Args[0].ID())) {
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
LoadConstant(x3, reinterpret_cast<uint64_t>(PrintValue));
}
else {
fmov(x0, GetSrc(Op->Header.Args[0].ID()).V1D());
// Bug in vixl that source vector needs to b V1D rather than V2D?
fmov(x1, GetSrc(Op->Header.Args[0].ID()).V1D(), 1);
LoadConstant(x3, reinterpret_cast<uint64_t>(PrintVectorValue));
}
SpillStaticRegs();
blr(x3);
FillStaticRegs();
PopDynamicRegsAndLR();
}
#undef DEF_OP
void Arm64JITCore::RegisterMiscHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
@@ -166,7 +127,7 @@ void Arm64JITCore::RegisterMiscHandlers() {
REGISTER_OP(BREAK, Break);
REGISTER_OP(PHI, NoOp);
REGISTER_OP(PHIVALUE, NoOp);
REGISTER_OP(PRINT, Print);
REGISTER_OP(PRINT, Unhandled);
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
REGISTER_OP(INVALIDATEFLAGS, NoOp);
@@ -10,7 +10,7 @@ namespace FEXCore::CPU {
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(ExtractElementPair) {
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
switch (Op->Header.Size) {
@@ -26,7 +26,7 @@ DEF_OP(ExtractElementPair) {
mov (GetReg<RA_64>(Node), Regs[Op->Element]);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Size"); break;
default: LogMan::Msg::A("Unknown Size"); break;
}
}
@@ -52,7 +52,7 @@ DEF_OP(CreateElementPair) {
RegTmp = x0;
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Size"); break;
default: LogMan::Msg::A("Unknown Size"); break;
}
if (Dst.first.GetCode() != RegSecond.GetCode()) {
File diff suppressed because it is too large. Load diff
+3 -10
View File
@@ -1,7 +1,5 @@
#pragma once
#include <memory>
namespace FEXCore::Context {
struct Context;
}
@@ -13,11 +11,6 @@ struct InternalThreadState;
namespace FEXCore::CPU {
class CPUBackend;
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread,
bool CompileThread);
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread,
bool CompileThread);
} // namespace FEXCore::CPU
FEXCore::CPU::CPUBackend *CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
FEXCore::CPU::CPUBackend *CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
}
+66 -111
View File
@@ -5,22 +5,10 @@ $end_info$
*/
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
#include <stdint.h>
#include <utility>
#include <xbyak/xbyak.h>
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
#define GRS(Node) (IROp->Size <= 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
#define GRCMP(Node) (Op->CompareSize == 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(TruncElementPair) {
auto Op = IROp->C<IR::IROp_TruncElementPair>();
@@ -32,7 +20,7 @@ DEF_OP(TruncElementPair) {
mov(Dst.second, Src.second);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", Op->Size); break;
default: LogMan::Msg::A("Unhandled Truncation size: %d", Op->Size); break;
}
}
@@ -44,7 +32,7 @@ DEF_OP(Constant) {
DEF_OP(EntrypointOffset) {
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
auto Constant = Entry + Op->Offset;
auto Constant = IR->GetHeader()->Entry + Op->Offset;
mov(GetDst<RA_64>(Node), Constant);
}
@@ -82,7 +70,7 @@ DEF_OP(Add) {
case 8:
add(rax, Const);
break;
default: LOGMAN_MSG_A_FMT("Unhandled Add size: {}", OpSize);
default: LogMan::Msg::A("Unhandled Add size: %d", OpSize);
break;
}
} else {
@@ -93,7 +81,7 @@ DEF_OP(Add) {
case 8:
add(rax, GetSrc<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled Add size: {}", OpSize);
default: LogMan::Msg::A("Unhandled Add size: %d", OpSize);
break;
}
}
@@ -115,7 +103,7 @@ DEF_OP(Sub) {
case 8:
sub(rax, Const);
break;
default: LOGMAN_MSG_A_FMT("Unhandled Sub size: {}", OpSize);
default: LogMan::Msg::A("Unhandled Sub size: %d", OpSize);
break;
}
} else {
@@ -126,7 +114,7 @@ DEF_OP(Sub) {
case 8:
sub(rax, GetSrc<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled Sub size: {}", OpSize);
default: LogMan::Msg::A("Unhandled Sub size: %d", OpSize);
break;
}
}
@@ -148,7 +136,7 @@ DEF_OP(Neg) {
Src = GetSrc<RA_64>(Op->Header.Args[0].ID());
Dst = GetDst<RA_64>(Node);
break;
default: LOGMAN_MSG_A_FMT("Unhandled Neg size: {}", OpSize);
default: LogMan::Msg::A("Unhandled Neg size: %d", OpSize);
break;
}
mov(Dst, Src);
@@ -172,7 +160,7 @@ DEF_OP(Mul) {
imul(rax, GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov(Dst, rax);
break;
default: LOGMAN_MSG_A_FMT("Unknown Mul size: {}", OpSize);
default: LogMan::Msg::A("Unknown Sext size: %d", OpSize);
}
}
@@ -191,7 +179,7 @@ DEF_OP(UMul) {
mul(GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov(GetDst<RA_64>(Node), rax);
break;
default: LOGMAN_MSG_A_FMT("Unknown UMul size: {}", OpSize);
default: LogMan::Msg::A("Unknown Sext size: %d", OpSize);
}
}
@@ -230,7 +218,7 @@ DEF_OP(Div) {
mov(GetDst<RA_64>(Node), rax);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown DIV Size: {}", Size); break;
default: LogMan::Msg::A("Unknown UDIV Size: %d", Size); break;
}
}
@@ -273,7 +261,7 @@ DEF_OP(UDiv) {
mov(GetDst<RA_64>(Node), rax);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown UDIV OpSize: {}", OpSize); break;
default: LogMan::Msg::A("Unknown UDIV OpSize: %d", OpSize); break;
}
}
@@ -310,7 +298,7 @@ DEF_OP(Rem) {
mov(GetDst<RA_64>(Node), rdx);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Rem Size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown UDIV Size: %d", OpSize); break;
}
}
@@ -353,7 +341,7 @@ DEF_OP(URem) {
mov(GetDst<RA_64>(Node), rdx);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown URem OpSize: {}", OpSize); break;
default: LogMan::Msg::A("Unknown UDIV OpSize: %d", OpSize); break;
}
}
@@ -372,7 +360,7 @@ DEF_OP(MulH) {
imul(GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov(GetDst<RA_64>(Node), rdx);
break;
default: LOGMAN_MSG_A_FMT("Unknown MulH size: {}", OpSize);
default: LogMan::Msg::A("Unknown Sext size: %d", OpSize);
}
}
@@ -391,7 +379,7 @@ DEF_OP(UMulH) {
mul(GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov(GetDst<RA_64>(Node), rdx);
break;
default: LOGMAN_MSG_A_FMT("Unknown UMulH size: {}", OpSize);
default: LogMan::Msg::A("Unknown Sext size: %d", OpSize);
}
}
@@ -422,25 +410,6 @@ DEF_OP(And) {
mov(Dst, rax);
}
DEF_OP(Andn) {
auto Op = IROp->C<IR::IROp_Andn>();
const auto& Lhs = Op->Header.Args[0];
const auto& Rhs = Op->Header.Args[1];
auto Dst = GRD(Node);
uint64_t Const{};
if (IsInlineConstant(Rhs, &Const)) {
mov(Dst, GRS(Lhs.ID()));
and_(Dst, ~Const);
} else {
const auto Temp = IROp->Size <= 4 ? Xbyak::Reg{rax.cvt32()} : Xbyak::Reg{rax};
mov(Temp, GRS(Rhs.ID()));
not_(Temp);
and_(Temp, GRS(Lhs.ID()));
mov(Dst, Temp);
}
}
DEF_OP(Xor) {
auto Op = IROp->C<IR::IROp_Xor>();
auto Dst = GetDst<RA_64>(Node);
@@ -472,7 +441,7 @@ DEF_OP(Lshl) {
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
shl(GetDst<RA_64>(Node), Const);
break;
default: LOGMAN_MSG_A_FMT("Unknown LSHL Size: {}\n", OpSize); break;
default: LogMan::Msg::A("Unknown LSHL Size: %d\n", OpSize); break;
};
} else {
mov(rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
@@ -487,7 +456,7 @@ DEF_OP(Lshl) {
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
shl(GetDst<RA_64>(Node), cl);
break;
default: LOGMAN_MSG_A_FMT("Unknown LSHL Size: {}\n", OpSize); break;
default: LogMan::Msg::A("Unknown LSHL Size: %d\n", OpSize); break;
};
}
}
@@ -519,7 +488,7 @@ DEF_OP(Lshr) {
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
shr(GetDst<RA_64>(Node), Const);
break;
default: LOGMAN_MSG_A_FMT("Unknown Size: {}\n", OpSize); break;
default: LogMan::Msg::A("Unknown Size: %d\n", OpSize); break;
};
} else {
@@ -543,7 +512,7 @@ DEF_OP(Lshr) {
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
shr(GetDst<RA_64>(Node), cl);
break;
default: LOGMAN_MSG_A_FMT("Unknown Size: {}\n", OpSize); break;
default: LogMan::Msg::A("Unknown Size: %d\n", OpSize); break;
};
}
}
@@ -577,7 +546,7 @@ DEF_OP(Ashr) {
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
sar(GetDst<RA_64>(Node), Const);
break;
default: LOGMAN_MSG_A_FMT("Unknown ASHR Size: {}\n", OpSize); break;
default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break;
};
} else {
@@ -602,7 +571,7 @@ DEF_OP(Ashr) {
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
sar(GetDst<RA_64>(Node), cl);
break;
default: LOGMAN_MSG_A_FMT("Unknown ASHR Size: {}\n", OpSize); break;
default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break;
};
}
}
@@ -627,7 +596,7 @@ DEF_OP(Ror) {
ror(rax, Const);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown ROR Size: {}\n", OpSize); break;
default: LogMan::Msg::A("Unknown ROR Size: %d\n", OpSize); break;
}
} else {
mov (rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
@@ -643,7 +612,7 @@ DEF_OP(Ror) {
ror(rax, cl);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown ROR Size: {}\n", OpSize); break;
default: LogMan::Msg::A("Unknown ROR Size: %d\n", OpSize); break;
}
}
mov(GetDst<RA_64>(Node), rax);
@@ -699,7 +668,7 @@ DEF_OP(LDiv) {
mov(GetDst<RA_64>(Node), rax);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LDIV OpSize: {}", OpSize); break;
default: LogMan::Msg::A("Unknown LDIV OpSize: %d", OpSize); break;
}
}
@@ -731,7 +700,7 @@ DEF_OP(LUDiv) {
mov(GetDst<RA_64>(Node), rax);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUDIV OpSize: {}", OpSize); break;
default: LogMan::Msg::A("Unknown LUDIV OpSize: %d", OpSize); break;
}
}
@@ -763,7 +732,7 @@ DEF_OP(LRem) {
mov(GetDst<RA_64>(Node), rdx);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LREM OpSize: {}", OpSize); break;
default: LogMan::Msg::A("Unknown LREM OpSize: %d", OpSize); break;
}
}
@@ -795,7 +764,7 @@ DEF_OP(LURem) {
mov(GetDst<RA_64>(Node), rdx);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUREM OpSize: {}", OpSize); break;
default: LogMan::Msg::A("Unknown LUDIV OpSize: %d", OpSize); break;
}
}
@@ -860,7 +829,7 @@ DEF_OP(FindMSB) {
case 8:
bsr(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown FindMSB OpSize: {}", OpSize);
default: LogMan::Msg::A("Unknown OpSize: %d", OpSize);
}
}
@@ -884,7 +853,7 @@ DEF_OP(FindTrailingZeros) {
mov(rax, 0x40);
cmovz(GetDst<RA_64>(Node), rax);
break;
default: LOGMAN_MSG_A_FMT("Unknown FindTrailingZeros size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown size: %d", OpSize); break;
}
}
@@ -907,7 +876,7 @@ DEF_OP(CountLeadingZeroes) {
lzcnt(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unknown CountLeadingZeros size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown size: %d", OpSize); break;
}
}
else {
@@ -946,7 +915,7 @@ DEF_OP(CountLeadingZeroes) {
mov(GetDst<RA_64>(Node), rax);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown CountLeadingZeros size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown size: %d", OpSize); break;
}
}
}
@@ -968,7 +937,7 @@ DEF_OP(Rev) {
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
bswap(GetDst<RA_64>(Node).cvt64());
break;
default: LOGMAN_MSG_A_FMT("Unknown REV size: {}", OpSize); break;
default: LogMan::Msg::A("Unknown REV size: %d", OpSize); break;
}
}
@@ -1001,7 +970,9 @@ DEF_OP(Bfi) {
DEF_OP(Bfe) {
auto Op = IROp->C<IR::IROp_Bfe>();
LOGMAN_THROW_A_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
uint8_t OpSize = IROp->Size;
LogMan::Throw::A(OpSize <= 8, "OpSize is too large for BFE: %d", OpSize);
auto Dst = GetDst<RA_64>(Node);
@@ -1072,6 +1043,10 @@ DEF_OP(Sbfe) {
}
}
#define GRS(Node) (IROp->Size <= 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
#define GRCMP(Node) (Op->CompareSize == 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
DEF_OP(Select) {
auto Op = IROp->C<IR::IROp_Select>();
auto Dst = GRD(Node);
@@ -1098,7 +1073,7 @@ DEF_OP(Select) {
if (is_const_true || is_const_false) {
if (is_const_false != true || is_const_true != true || const_true != 1 || const_false != 0) {
LOGMAN_MSG_A_FMT("Select: Unsupported compare inline parameters");
LogMan::Msg::A("Select: Unsupported compare inline parameters");
}
(this->*SetCC)(al);
movzx(Dst, al);
@@ -1129,67 +1104,46 @@ DEF_OP(VExtractToGPR) {
pextrq(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default: LogMan::Msg::A("Unknown Element Size: %d", Op->Header.ElementSize); break;
}
}
DEF_OP(Float_ToGPR_ZU) {
LogMan::Msg::D("Unimplemented");
}
DEF_OP(Float_ToGPR_ZS) {
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0804: // int64_t <- float
cvttss2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
break;
case 0x0808: // int64_t <- double
cvttsd2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
break;
case 0x0404: // int32_t <- float
cvttss2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
break;
case 0x0408: // int32_t <- double
cvttsd2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
break;
if (Op->Header.ElementSize == 8) {
cvttsd2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
}
else {
cvttss2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
}
}
DEF_OP(Float_ToGPR_U) {
LogMan::Msg::D("Unimplemented");
}
DEF_OP(Float_ToGPR_S) {
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0804: // int64_t <- float
cvtss2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
break;
case 0x0808: // int64_t <- double
cvtsd2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
break;
case 0x0404: // int32_t <- float
cvtss2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
break;
case 0x0408: // int32_t <- double
cvtsd2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
break;
if (Op->Header.ElementSize == 8) {
cvtsd2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
}
else {
cvtss2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
}
}
DEF_OP(FCmp) {
auto Op = IROp->C<IR::IROp_FCmp>();
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED)) {
if (Op->ElementSize == 4) {
ucomiss(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
}
else {
ucomisd(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
}
if (Op->ElementSize == 4) {
ucomiss(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
}
else {
if (Op->ElementSize == 4) {
comiss(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
}
else {
comisd(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
}
ucomisd(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
}
mov (rdx, 0);
@@ -1241,7 +1195,6 @@ void X86JITCore::RegisterALUHandlers() {
REGISTER_OP(UMULH, UMulH);
REGISTER_OP(OR, Or);
REGISTER_OP(AND, And);
REGISTER_OP(ANDN, Andn);
REGISTER_OP(XOR, Xor);
REGISTER_OP(LSHL, Lshl);
REGISTER_OP(LSHR, Lshr);
@@ -1264,7 +1217,9 @@ void X86JITCore::RegisterALUHandlers() {
REGISTER_OP(SBFE, Sbfe);
REGISTER_OP(SELECT, Select);
REGISTER_OP(VEXTRACTTOGPR, VExtractToGPR);
REGISTER_OP(FLOAT_TOGPR_ZU, Float_ToGPR_ZU);
REGISTER_OP(FLOAT_TOGPR_ZS, Float_ToGPR_ZS);
REGISTER_OP(FLOAT_TOGPR_U, Float_ToGPR_U);
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
REGISTER_OP(FCMP, FCmp);
#undef REGISTER_OP
@@ -5,17 +5,10 @@ $end_info$
*/
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
#include <stdint.h>
#include <utility>
#include <xbyak/xbyak.h>
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CAS>();
uint8_t OpSize = IROp->Size;
@@ -62,7 +55,7 @@ DEF_OP(CASPair) {
mov(Dst.second, rdx);
break;
}
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
default: LogMan::Msg::A("Unsupported: %d", OpSize);
}
}
@@ -81,6 +74,7 @@ DEF_OP(CAS) {
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[2].ID());
mov(rdx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
// RCX now contains pointer
@@ -88,31 +82,31 @@ DEF_OP(CAS) {
// RDX contains our desired
lock();
switch (OpSize) {
case 1: {
cmpxchg(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
movzx(GetDst<RA_64>(Node), al);
cmpxchg(byte [MemReg], dl);
movzx(rax, al);
break;
}
case 2: {
cmpxchg(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
movzx(GetDst<RA_64>(Node), ax);
cmpxchg(word [MemReg], dx);
movzx(rax, ax);
break;
}
case 4: {
cmpxchg(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
// RAX now contains the result
mov (GetDst<RA_64>(Node), eax);
cmpxchg(dword [MemReg], edx);
break;
}
case 8: {
cmpxchg(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
// RAX now contains the result
mov (GetDst<RA_64>(Node), rax);
cmpxchg(qword [MemReg], rdx);
break;
}
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
default: LogMan::Msg::A("Unsupported: %d", OpSize);
}
// RAX now contains the result
mov (GetDst<RA_64>(Node), rax);
}
DEF_OP(AtomicAdd) {
@@ -121,7 +115,7 @@ DEF_OP(AtomicAdd) {
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
lock();
switch (IROp->Size) {
switch (Op->Size) {
case 1:
add(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
break;
@@ -134,7 +128,7 @@ DEF_OP(AtomicAdd) {
case 8:
add(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicAdd size: %d", Op->Size);
}
}
@@ -143,7 +137,7 @@ DEF_OP(AtomicSub) {
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
lock();
switch (IROp->Size) {
switch (Op->Size) {
case 1:
sub(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
break;
@@ -156,7 +150,7 @@ DEF_OP(AtomicSub) {
case 8:
sub(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicAdd size: %d", Op->Size);
}
}
@@ -165,7 +159,7 @@ DEF_OP(AtomicAnd) {
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
lock();
switch (IROp->Size) {
switch (Op->Size) {
case 1:
and_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
break;
@@ -178,7 +172,7 @@ DEF_OP(AtomicAnd) {
case 8:
and_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicAdd size: %d", Op->Size);
}
}
@@ -187,7 +181,7 @@ DEF_OP(AtomicOr) {
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
lock();
switch (IROp->Size) {
switch (Op->Size) {
case 1:
or_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
break;
@@ -200,7 +194,7 @@ DEF_OP(AtomicOr) {
case 8:
or_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicAdd size: %d", Op->Size);
}
}
@@ -209,7 +203,7 @@ DEF_OP(AtomicXor) {
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
lock();
switch (IROp->Size) {
switch (Op->Size) {
case 1:
xor_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
break;
@@ -222,7 +216,7 @@ DEF_OP(AtomicXor) {
case 8:
xor_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicAdd size: %d", Op->Size);
}
}
@@ -232,19 +226,19 @@ DEF_OP(AtomicSwap) {
Xbyak::Reg MemReg = rax;
mov(MemReg, GetSrc<RA_64>(Op->Header.Args[0].ID()));
switch (IROp->Size) {
switch (Op->Size) {
case 1:
movzx(GetDst<RA_64>(Node), GetSrc<RA_8>(Op->Header.Args[1].ID()));
mov(GetDst<RA_8>(Node), GetSrc<RA_8>(Op->Header.Args[1].ID()));
lock();
xchg(byte [MemReg], GetDst<RA_8>(Node));
break;
case 2:
movzx(GetDst<RA_64>(Node), GetSrc<RA_16>(Op->Header.Args[1].ID()));
mov(GetDst<RA_16>(Node), GetSrc<RA_16>(Op->Header.Args[1].ID()));
lock();
xchg(word [MemReg], GetDst<RA_16>(Node));
break;
case 4:
mov(GetDst<RA_64>(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()));
mov(GetDst<RA_32>(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()));
lock();
xchg(dword [MemReg], GetDst<RA_32>(Node));
break;
@@ -253,7 +247,7 @@ DEF_OP(AtomicSwap) {
lock();
xchg(qword [MemReg], GetDst<RA_64>(Node));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicSwap size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicAdd size: %d", Op->Size);
}
}
@@ -261,15 +255,15 @@ DEF_OP(AtomicFetchAdd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
switch (IROp->Size) {
switch (Op->Size) {
case 1:
movzx(rcx, GetSrc<RA_8>(Op->Header.Args[1].ID()));
mov(cl, GetSrc<RA_8>(Op->Header.Args[1].ID()));
lock();
xadd(byte [MemReg], cl);
movzx(GetDst<RA_32>(Node), cl);
break;
case 2:
movzx(rcx, GetSrc<RA_16>(Op->Header.Args[1].ID()));
mov(cx, GetSrc<RA_16>(Op->Header.Args[1].ID()));
lock();
xadd(word [MemReg], cx);
movzx(GetDst<RA_32>(Node), cx);
@@ -278,7 +272,7 @@ DEF_OP(AtomicFetchAdd) {
mov(ecx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
lock();
xadd(dword [MemReg], ecx);
mov(GetDst<RA_64>(Node), ecx);
mov(GetDst<RA_32>(Node), ecx);
break;
case 8:
mov(rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
@@ -286,7 +280,7 @@ DEF_OP(AtomicFetchAdd) {
xadd(qword [MemReg], rcx);
mov(GetDst<RA_64>(Node), rcx);
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchAdd size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicFetchAdd size: %d", Op->Size);
}
}
@@ -294,7 +288,7 @@ DEF_OP(AtomicFetchSub) {
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
switch (IROp->Size) {
switch (Op->Size) {
case 1:
mov(cl, GetSrc<RA_8>(Op->Header.Args[1].ID()));
neg(cl);
@@ -323,7 +317,7 @@ DEF_OP(AtomicFetchSub) {
xadd(qword [MemReg], rcx);
mov(GetDst<RA_64>(Node), rcx);
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchSub size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicFetchAdd size: %d", Op->Size);
}
}
@@ -333,7 +327,7 @@ DEF_OP(AtomicFetchAnd) {
// TMP1 = rax
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
mov(TMP1.cvt8(), byte [MemReg]);
@@ -401,7 +395,7 @@ DEF_OP(AtomicFetchAnd) {
mov(GetDst<RA_64>(Node), TMP3.cvt64());
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchAnd size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicFetchAdd size: %d", Op->Size);
}
}
@@ -410,7 +404,7 @@ DEF_OP(AtomicFetchOr) {
// TMP1 = rax
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
mov(TMP1.cvt8(), byte [MemReg]);
@@ -478,7 +472,7 @@ DEF_OP(AtomicFetchOr) {
mov(GetDst<RA_64>(Node), TMP3.cvt64());
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchOr size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicFetchAdd size: %d", Op->Size);
}
}
@@ -487,7 +481,7 @@ DEF_OP(AtomicFetchXor) {
// TMP1 = rax
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
mov(TMP1.cvt8(), byte [MemReg]);
@@ -555,83 +549,7 @@ DEF_OP(AtomicFetchXor) {
mov(GetDst<RA_64>(Node), TMP3.cvt64());
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchXor size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchNeg) {
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
switch (IROp->Size) {
case 1: {
mov(TMP1.cvt8(), byte [MemReg]);
Label Loop;
L(Loop);
mov(TMP2.cvt8(), TMP1.cvt8());
mov(TMP3.cvt8(), TMP1.cvt8());
neg(TMP2.cvt8());
// Updates RAX with the value from memory
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
jne(Loop);
// Result is the previous value from memory, which is currently in TMP3
movzx(GetDst<RA_64>(Node), TMP3.cvt8());
break;
}
case 2: {
mov(TMP1.cvt16(), word [MemReg]);
Label Loop;
L(Loop);
mov(TMP2.cvt16(), TMP1.cvt16());
mov(TMP3.cvt16(), TMP1.cvt16());
neg(TMP2.cvt16());
// Updates RAX with the value from memory
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
jne(Loop);
// Result is the previous value from memory, which is currently in TMP3
movzx(GetDst<RA_64>(Node), TMP3.cvt16());
break;
}
case 4: {
mov(TMP1.cvt32(), dword [MemReg]);
Label Loop;
L(Loop);
mov(TMP2.cvt32(), TMP1.cvt32());
mov(TMP3.cvt32(), TMP1.cvt32());
neg(TMP2.cvt32());
// Updates RAX with the value from memory
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
jne(Loop);
// Result is the previous value from memory, which is currently in TMP3
mov(GetDst<RA_32>(Node), TMP3.cvt32());
break;
}
case 8: {
mov(TMP1.cvt64(), qword [MemReg]);
Label Loop;
L(Loop);
mov(TMP2.cvt64(), TMP1.cvt64());
mov(TMP3.cvt64(), TMP1.cvt64());
neg(TMP2.cvt64());
// Updates RAX with the value from memory
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
jne(Loop);
// Result is the previous value from memory, which is currently in TMP3
mov(GetDst<RA_64>(Node), TMP3.cvt64());
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchNeg size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled AtomicFetchAdd size: %d", Op->Size);
}
}
@@ -651,7 +569,6 @@ void X86JITCore::RegisterAtomicHandlers() {
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
#undef REGISTER_OP
}
}
@@ -4,41 +4,25 @@ tags: backend|x86-64
$end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include "Interface/HLE/Thunks/Thunks.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
#include <memory>
#include <stddef.h>
#include <stdint.h>
#include <unordered_map>
#include <utility>
#include <xbyak/xbyak.h>
#include <Interface/HLE/Thunks/Thunks.h>
namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(GuestCallDirect) {
LogMan::Msg::DFmt("Unimplemented");
LogMan::Msg::D("Unimplemented");
}
DEF_OP(GuestCallIndirect) {
LogMan::Msg::DFmt("Unimplemented");
LogMan::Msg::D("Unimplemented");
}
DEF_OP(GuestReturn) {
LogMan::Msg::DFmt("Unimplemented");
LogMan::Msg::D("Unimplemented");
}
DEF_OP(SignalReturn) {
@@ -97,7 +81,7 @@ DEF_OP(ExitFunction) {
jmp(qword[rax]);
L(l_BranchHost);
dq(ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress);
dq(Dispatcher->ExitFunctionLinkerAddress);
L(l_BranchGuest);
dq(NewRIP);
} else {
@@ -117,7 +101,7 @@ DEF_OP(ExitFunction) {
jmp(qword[LookupBase + 0]);
L(FullLookup);
mov(rax, ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress);
mov(rax, Dispatcher->AbsoluteLoopTopAddress);
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], RipReg);
jmp(rax);
}
@@ -128,10 +112,18 @@ DEF_OP(ExitFunction) {
}
DEF_OP(Jump) {
const auto Op = IROp->C<IR::IROp_Jump>();
const auto ArgID = Op->Args(0).ID();
auto Op = IROp->C<IR::IROp_Jump>();
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
Label *TargetLabel;
auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID());
if (IsTarget == JumpTargets.end()) {
TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second;
}
else {
TargetLabel = &IsTarget->second;
}
PendingTargetLabel = TargetLabel;
}
#define GRCMP(Node) (Op->CompareSize == 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
@@ -139,7 +131,18 @@ DEF_OP(Jump) {
DEF_OP(CondJump) {
auto Op = IROp->C<IR::IROp_CondJump>();
Label *TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
Label *TrueTargetLabel;
Label *FalseTargetLabel;
auto TrueIter = JumpTargets.find(Op->TrueBlock.ID());
auto FalseIter = JumpTargets.find(Op->FalseBlock.ID());
if (TrueIter == JumpTargets.end()) {
TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
}
else {
TrueTargetLabel = &TrueIter->second;
}
if (IsGPR(Op->Cmp1.ID())) {
uint64_t Const;
@@ -149,18 +152,24 @@ DEF_OP(CondJump) {
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
}
} else if (IsFPR(Op->Cmp1.ID())) {
if (Op->CompareSize == 4) {
if (Op->CompareSize == 4)
ucomiss(GetSrc(Op->Cmp1.ID()), GetSrc(Op->Cmp2.ID()));
} else {
else
ucomisd(GetSrc(Op->Cmp1.ID()), GetSrc(Op->Cmp2.ID()));
}
}
auto [_, __, JCC] = GetCC(Op->Cond);
(this->*JCC)(*TrueTargetLabel, T_NEAR);
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
if (FalseIter == JumpTargets.end()) {
FalseTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
}
else {
FalseTargetLabel = &FalseIter->second;
}
PendingTargetLabel = FalseTargetLabel;
}
DEF_OP(Syscall) {
@@ -239,28 +248,28 @@ DEF_OP(Thunk) {
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
const auto* OldCode = (const uint8_t*)&Op->CodeOriginalLow;
uint8_t* OldCode = (uint8_t*)&Op->CodeOriginalLow;
int len = Op->CodeLength;
int idx = 0;
xor_(GetDst<RA_64>(Node), GetDst<RA_64>(Node));
mov(rax, Entry + Op->Offset);
mov(rax, IR->GetHeader()->Entry + Op->Offset);
mov(rbx, 1);
while (len >= 4) {
cmp(dword[rax + idx], *(const uint32_t*)(OldCode + idx));
cmp(dword[rax + idx], *(uint32_t*)(OldCode + idx));
cmovne(GetDst<RA_64>(Node), rbx);
len-=4;
idx+=4;
}
while (len >= 2) {
mov(rcx, *(const uint16_t*)(OldCode + idx));
mov(rcx, *(uint16_t*)(OldCode + idx));
cmp(word[rax + idx], cx);
cmovne(GetDst<RA_64>(Node), rbx);
len-=2;
idx+=2;
}
while (len >= 1) {
cmp(byte[rax + idx], *(const uint8_t*)(OldCode + idx));
cmp(byte[rax + idx], *(uint8_t*)(OldCode + idx));
cmovne(GetDst<RA_64>(Node), rbx);
len-=1;
idx+=1;
@@ -277,7 +286,7 @@ DEF_OP(RemoveCodeEntry) {
sub(rsp, 8); // Align
mov(rdi, STATE);
mov(rax, Entry); // imm64 move
mov(rax, IR->GetHeader()->Entry); // imm64 move
mov(rsi, rax);
@@ -310,9 +319,8 @@ DEF_OP(CPUID) {
//
// Result: RAX, RDX. 4xi32
// rsi can be in the source registers, so copy argument to edx first
mov (edx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
mov (esi, GetSrc<RA_32>(Op->Header.Args[0].ID()));
mov (rsi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
mov (rdx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov (rdi, reinterpret_cast<uint64_t>(&CTX->CPUID));
auto NumPush = RA64.size();
@@ -5,17 +5,11 @@ $end_info$
*/
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
#include <stdint.h>
#include <xbyak/xbyak.h>
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(VInsGPR) {
auto Op = IROp->C<IR::IROp_VInsGPR>();
movapd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
@@ -37,7 +31,7 @@ DEF_OP(VInsGPR) {
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[1].ID()), Op->Index);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default: LogMan::Msg::A("Unknown Element Size: %d", Op->Header.ElementSize); break;
}
}
@@ -58,10 +52,14 @@ DEF_OP(VCastFromGPR) {
case 8:
vmovq(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()).cvt64());
break;
default: LOGMAN_MSG_A_FMT("Unknown VCastFromGPR element size: {}", Op->Header.ElementSize);
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Float_FromGPR_U) {
LogMan::Msg::A("Unimplemented");
}
DEF_OP(Float_FromGPR_S) {
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
@@ -97,10 +95,14 @@ DEF_OP(Float_FToF) {
cvtsd2ss(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Float_FToF sizes: 0x{:x}", Conv);
default: LogMan::Msg::A("Unknown FCVT sizes: 0x%x", Conv);
}
}
DEF_OP(Vector_UToF) {
LogMan::Msg::A("Unimplemented");
}
DEF_OP(Vector_SToF) {
auto Op = IROp->C<IR::IROp_Vector_SToF>();
switch (Op->Header.ElementSize) {
@@ -119,10 +121,14 @@ DEF_OP(Vector_SToF) {
cvtsi2sd(xmm15, rax);
movlhps(GetDst(Node), xmm15);
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Vector_FToZU) {
LogMan::Msg::A("Unimplemented");
}
DEF_OP(Vector_FToZS) {
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
switch (Op->Header.ElementSize) {
@@ -132,10 +138,14 @@ DEF_OP(Vector_FToZS) {
case 8:
cvttpd2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
DEF_OP(Vector_FToU) {
LogMan::Msg::A("Unimplemented");
}
DEF_OP(Vector_FToS) {
auto Op = IROp->C<IR::IROp_Vector_FToS>();
switch (Op->Header.ElementSize) {
@@ -145,7 +155,7 @@ DEF_OP(Vector_FToS) {
case 8:
cvtpd2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
default: LogMan::Msg::A("Unknown castGPR element size: %d", Op->Header.ElementSize);
}
}
@@ -162,39 +172,7 @@ DEF_OP(Vector_FToF) {
cvtpd2ps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv); break;
}
}
DEF_OP(Vector_FToI) {
auto Op = IROp->C<IR::IROp_Vector_FToI>();
uint8_t RoundMode{};
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
RoundMode = 0b0000'0'0'00;
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
RoundMode = 0b0000'0'0'01;
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
RoundMode = 0b0000'0'0'10;
break;
case FEXCore::IR::Round_Towards_Zero.Val:
RoundMode = 0b0000'0'0'11;
break;
case FEXCore::IR::Round_Host.Val:
RoundMode = 0b0000'0'1'00;
break;
}
switch (Op->Header.ElementSize) {
case 4:
roundps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
break;
case 8:
roundpd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
break;
default: LogMan::Msg::A("Unknown Conversion Type : 0%04x", Conv); break;
}
}
@@ -203,13 +181,16 @@ void X86JITCore::RegisterConversionHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
REGISTER_OP(VINSGPR, VInsGPR);
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
REGISTER_OP(FLOAT_FROMGPR_U, Float_FromGPR_U);
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
REGISTER_OP(FLOAT_FTOF, Float_FToF);
REGISTER_OP(VECTOR_UTOF, Vector_UToF);
REGISTER_OP(VECTOR_STOF, Vector_SToF);
REGISTER_OP(VECTOR_FTOZU, Vector_FToZU);
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
REGISTER_OP(VECTOR_FTOU, Vector_FToU);
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
#undef REGISTER_OP
}
}
@@ -5,15 +5,10 @@ $end_info$
*/
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include <FEXCore/IR/IR.h>
#include <array>
#include <stdint.h>
#include <xbyak/xbyak.h>
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(AESImc) {
auto Op = IROp->C<IR::IROp_VAESImc>();
@@ -5,16 +5,11 @@ $end_info$
*/
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include <FEXCore/IR/IR.h>
#include <array>
#include <stdint.h>
#include <xbyak/xbyak.h>
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(GetHostFlag) {
auto Op = IROp->C<IR::IROp_GetHostFlag>();
+94 -116
View File
@@ -6,73 +6,56 @@ $end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include "Interface/IR/PassManager.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Core/UContext.h>
#include <algorithm>
#include <array>
#include <bits/types/stack_t.h>
#include <memory>
#include <stddef.h>
#include <stdint.h>
#include <cmath>
#include <signal.h>
#include <sys/mman.h>
#include <tuple>
#include <unordered_map>
#include <utility>
#include <vector>
#include <xbyak/xbyak.h>
#include "Interface/Core/Interpreter/InterpreterOps.h"
// #define DEBUG_RA 1
// #define DEBUG_CYCLES
namespace FEXCore::CPU {
CodeBuffer AllocateNewCodeBuffer(FEXCore::Context::Context *CTX, size_t Size) {
CodeBuffer AllocateNewCodeBuffer(size_t Size) {
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t*>(
FEXCore::Allocator::mmap(nullptr,
mmap(nullptr,
Buffer.Size,
PROT_READ | PROT_WRITE | PROT_EXEC,
MAP_PRIVATE | MAP_ANONYMOUS,
-1, 0));
LOGMAN_THROW_A_FMT(Buffer.Ptr != reinterpret_cast<uint8_t*>(~0ULL), "Couldn't allocate code buffer");
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
LogMan::Throw::A(Buffer.Ptr != reinterpret_cast<uint8_t*>(~0ULL), "Couldn't allocate code buffer");
return Buffer;
}
void FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
munmap(Buffer.Ptr, Buffer.Size);
}
}
namespace FEXCore::CPU {
void X86JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Original);
ThreadSharedData = Core->ThreadSharedData;
}
void X86JITCore::PushRegs() {
sub(rsp, 16 * RAXMM_x.size());
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
movaps(ptr[rsp + i * 16], RAXMM_x[i]);
for (auto &Xmm : RAXMM_x) {
sub(rsp, 16);
movaps(ptr[rsp], Xmm);
}
for (auto &Reg : RA64)
@@ -91,19 +74,17 @@ void X86JITCore::PopRegs() {
for (uint32_t i = RA64.size(); i > 0; --i)
pop(RA64[i - 1]);
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
movaps(RAXMM_x[i], ptr[rsp + i * 16]);
for (uint32_t i = RAXMM_x.size(); i > 0; --i) {
movaps(RAXMM_x[i - 1], ptr[rsp]);
add(rsp, 16);
}
add(rsp, 16 * RAXMM_x.size());
}
void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
void X86JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
FallbackInfo Info;
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_MSG_A_FMT("Unhandled IR Op: {}", FEXCore::IR::GetName(IROp->Op));
#endif
auto Name = FEXCore::IR::GetName(IROp->Op);
LogMan::Msg::A("Unhandled IR Op: %s", std::string(Name).c_str());
} else {
switch(Info.ABI) {
case FABI_VOID_U16: {
@@ -300,15 +281,13 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
case FABI_UNKNOWN:
default:
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}", FEXCore::IR::GetName(IROp->Op), Info.ABI);
#endif
break;
auto Name = FEXCore::IR::GetName(IROp->Op);
LogMan::Msg::A("Unhandled IR Fallback abi: %s %d", std::string(Name).c_str(), Info.ABI);
}
}
}
void X86JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
void X86JITCore::Op_NoOp(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
}
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer, bool CompileThread)
@@ -319,7 +298,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
{
CurrentCodeBuffer = &InitialCodeBuffer;
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
RAPass = Thread->PassManager->GetRAPass();
RAPass->AllocateRegisterSet(RegisterCount, RegisterClasses);
RAPass->AddRegisters(FEXCore::IR::GPRClass, NumGPRs);
@@ -351,33 +330,31 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
Dispatcher = std::make_unique<X86Dispatcher>(CTX, ThreadState, config);
Dispatcher = new X86Dispatcher(CTX, ThreadState, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
ThreadSharedData.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress;
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
ThreadSharedData.Dispatcher = Dispatcher.get();
// This will register the host signal handler per thread, which is fine
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
}, true);
});
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
}, true);
});
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
};
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
for (uint32_t Signal = 0; Signal < SignalDelegator::MAX_SIGNALS; ++Signal) {
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
}
}
@@ -419,7 +396,7 @@ void X86JITCore::ClearCache() {
CurrentCodeBuffer->Size *= 1.5;
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MAX_CODE_SIZE);
InitialCodeBuffer = AllocateNewCodeBuffer(CTX, CurrentCodeBuffer->Size);
InitialCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
}
}
@@ -427,112 +404,112 @@ void X86JITCore::ClearCache() {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto NewCodeBuffer = AllocateNewCodeBuffer(CTX, X86JITCore::INITIAL_CODE_SIZE);
auto NewCodeBuffer = AllocateNewCodeBuffer(X86JITCore::INITIAL_CODE_SIZE);
EmplaceNewCodeBuffer(NewCodeBuffer);
setNewBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
}
}
IR::PhysicalRegister X86JITCore::GetPhys(IR::NodeID Node) const {
IR::PhysicalRegister X86JITCore::GetPhys(uint32_t Node) {
auto PhyReg = RAData->GetNodeRegister(Node);
LOGMAN_THROW_A_FMT(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
LogMan::Throw::A(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa%d. Class: %d", Node, PhyReg.Class);
return PhyReg;
}
bool X86JITCore::IsFPR(IR::NodeID Node) const {
bool X86JITCore::IsFPR(uint32_t Node) {
return RAData->GetNodeRegister(Node).Class == IR::FPRClass.Val;
}
bool X86JITCore::IsGPR(IR::NodeID Node) const {
bool X86JITCore::IsGPR(uint32_t Node) {
return RAData->GetNodeRegister(Node).Class == IR::GPRClass.Val;
}
template<uint8_t RAType>
Xbyak::Reg X86JITCore::GetSrc(IR::NodeID Node) const {
Xbyak::Reg X86JITCore::GetSrc(uint32_t Node) {
// rax, rcx, rdx, rsi, r8, r9,
// r10
// Callee Saved
// rbx, rbp, r12, r13, r14, r15
auto PhyReg = GetPhys(Node);
if constexpr (RAType == RA_64)
if (RAType == RA_64)
return RA64[PhyReg.Reg].cvt64();
else if constexpr (RAType == RA_XMM)
else if (RAType == RA_XMM)
return RAXMM[PhyReg.Reg];
else if constexpr (RAType == RA_32)
else if (RAType == RA_32)
return RA64[PhyReg.Reg].cvt32();
else if constexpr (RAType == RA_16)
else if (RAType == RA_16)
return RA64[PhyReg.Reg].cvt16();
else if constexpr (RAType == RA_8)
else if (RAType == RA_8)
return RA64[PhyReg.Reg].cvt8();
}
template
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_64>(IR::NodeID Node) const;
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_64>(uint32_t Node);
template
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_32>(IR::NodeID Node) const;
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_32>(uint32_t Node);
template
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_16>(IR::NodeID Node) const;
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_16>(uint32_t Node);
template
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_8>(IR::NodeID Node) const;
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_8>(uint32_t Node);
Xbyak::Xmm X86JITCore::GetSrc(IR::NodeID Node) const {
Xbyak::Xmm X86JITCore::GetSrc(uint32_t Node) {
auto PhyReg = GetPhys(Node);
return RAXMM_x[PhyReg.Reg];
}
template<uint8_t RAType>
Xbyak::Reg X86JITCore::GetDst(IR::NodeID Node) const {
Xbyak::Reg X86JITCore::GetDst(uint32_t Node) {
auto PhyReg = GetPhys(Node);
if constexpr (RAType == RA_64)
if (RAType == RA_64)
return RA64[PhyReg.Reg].cvt64();
else if constexpr (RAType == RA_XMM)
else if (RAType == RA_XMM)
return RAXMM[PhyReg.Reg];
else if constexpr (RAType == RA_32)
else if (RAType == RA_32)
return RA64[PhyReg.Reg].cvt32();
else if constexpr (RAType == RA_16)
else if (RAType == RA_16)
return RA64[PhyReg.Reg].cvt16();
else if constexpr (RAType == RA_8)
else if (RAType == RA_8)
return RA64[PhyReg.Reg].cvt8();
}
template
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_64>(IR::NodeID Node) const;
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_64>(uint32_t Node);
template
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_32>(IR::NodeID Node) const;
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_32>(uint32_t Node);
template
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_16>(IR::NodeID Node) const;
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_16>(uint32_t Node);
template
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_8>(IR::NodeID Node) const;
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_8>(uint32_t Node);
template<uint8_t RAType>
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair(IR::NodeID Node) const {
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair(uint32_t Node) {
auto PhyReg = GetPhys(Node);
if constexpr (RAType == RA_64)
if (RAType == RA_64)
return RA64Pair[PhyReg.Reg];
else if constexpr (RAType == RA_32)
else if (RAType == RA_32)
return {RA64Pair[PhyReg.Reg].first.cvt32(), RA64Pair[PhyReg.Reg].second.cvt32()};
}
template
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_64>(IR::NodeID Node) const;
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_64>(uint32_t Node);
template
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_32>(IR::NodeID Node) const;
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_32>(uint32_t Node);
Xbyak::Xmm X86JITCore::GetDst(IR::NodeID Node) const {
Xbyak::Xmm X86JITCore::GetDst(uint32_t Node) {
auto PhyReg = GetPhys(Node);
return RAXMM_x[PhyReg.Reg];
}
bool X86JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
bool X86JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
if (OpHeader->Op == IR::IROps::OP_INLINECONSTANT) {
@@ -546,13 +523,13 @@ bool X86JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t*
}
}
bool X86JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
bool X86JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
if (Value) {
*Value = Entry + Op->Offset;
*Value = IR->GetHeader()->Entry + Op->Offset;
}
return true;
} else {
@@ -585,7 +562,7 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
LogMan::Msg::A("Unsupported compare type");
break;
}
@@ -593,11 +570,10 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
return { &CodeGenerator::sete , &CodeGenerator::cmove , &CodeGenerator::je };
}
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
void *X86JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
JumpTargets.clear();
uint32_t SSACount = IR->GetSSACount();
this->Entry = Entry;
this->RAData = RAData;
// Fairly excessive buffer range to make sure we don't overflow
@@ -606,7 +582,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
ThreadState->CTX->ClearCodeCache(ThreadState, false);
}
void *GuestEntry = getCurr<void*>();
void *Entry = getCurr<void*>();
this->IR = IR;
if (CTX->GetGdbServerStatus()) {
@@ -621,14 +597,14 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
je(RunBlock);
// Else we need to pause now
mov(rax, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddress);
mov(rax, Dispatcher->ThreadPauseHandlerAddress);
jmp(rax);
ud2();
L(RunBlock);
}
LOGMAN_THROW_A_FMT(RAData != nullptr, "Needs RA");
LogMan::Throw::A(RAData != nullptr, "Needs RA");
SpillSlots = RAData->SpillSlots();
@@ -637,7 +613,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
}
#ifdef BLOCKSTATS
BlockSamplingData::BlockData *SamplingData = CTX->BlockData->GetBlockData(Entry);
BlockSamplingData::BlockData *SamplingData = CTX->BlockData->GetBlockData(HeaderOp->Entry);
if (GetSamplingData) {
mov(rcx, reinterpret_cast<uintptr_t>(SamplingData));
rdtsc();
@@ -684,16 +660,18 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
using namespace FEXCore::IR;
{
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
#endif
LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
const auto Node = IR->GetID(BlockNode);
const auto IsTarget = JumpTargets.try_emplace(Node).first;
uint32_t Node = IR->GetID(BlockNode);
auto IsTarget = JumpTargets.find(Node);
if (IsTarget == JumpTargets.end()) {
IsTarget = JumpTargets.try_emplace(Node).first;
}
// if there is a pending branch, and it is not fall-through
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second) {
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
{
jmp(*PendingTargetLabel, T_NEAR);
}
PendingTargetLabel = nullptr;
@@ -722,10 +700,10 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
Inst << "\t" << Name << " ";
}
const uint8_t NumArgs = IR::GetArgs(IROp->Op);
uint8_t NumArgs = IR::GetArgs(IROp->Op);
for (uint8_t i = 0; i < NumArgs; ++i) {
const auto ArgNode = IROp->Args[i].ID();
const uint64_t PhysReg = RAPass->GetNodeRegister(ArgNode);
uint32_t ArgNode = IROp->Args[i].ID();
uint64_t PhysReg = RAPass->GetNodeRegister(ArgNode);
if (PhysReg >= GPRPairBase)
Inst << "Pair" << GetPhys(ArgNode) << (i + 1 == NumArgs ? "" : ", ");
else if (PhysReg >= XMMBase)
@@ -734,10 +712,10 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
Inst << "Reg" << GetPhys(ArgNode) << (i + 1 == NumArgs ? "" : ", ");
}
LogMan::Msg::DFmt("{}", Inst.str());
LogMan::Msg::D("%s", Inst.str().c_str());
}
#endif
const auto ID = IR->GetID(CodeNode);
uint32_t ID = IR->GetID(CodeNode);
// Execute handler
OpHandler Handler = OpHandlers[IROp->Op];
@@ -752,15 +730,15 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
}
PendingTargetLabel = nullptr;
void *GuestExit = getCurr<void*>();
void *Exit = getCurr<void*>();
this->IR = nullptr;
ready();
if (DebugData) {
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(GuestExit) - reinterpret_cast<uintptr_t>(GuestEntry);
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(Exit) - reinterpret_cast<uintptr_t>(Entry);
}
return GuestEntry;
return Entry;
}
uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
@@ -771,10 +749,10 @@ uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateF
if (!HostCode) {
Thread->CurrentFrame->State.rip = GuestRip;
return core->ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress;
return core->Dispatcher->AbsoluteLoopTopAddress;
}
auto LinkerAddress = core->ThreadSharedData.Dispatcher->ExitFunctionLinkerAddress;
auto LinkerAddress = core->Dispatcher->ExitFunctionLinkerAddress;
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
// undo the link
record[0] = LinkerAddress;
@@ -784,7 +762,7 @@ uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateF
return HostCode;
}
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(ctx, CompileThread ? X86JITCore::MAX_CODE_SIZE : X86JITCore::INITIAL_CODE_SIZE), CompileThread);
FEXCore::CPU::CPUBackend *CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
return new X86JITCore(ctx, Thread, AllocateNewCodeBuffer(CompileThread ? X86JITCore::MAX_CODE_SIZE : X86JITCore::INITIAL_CODE_SIZE), CompileThread);
}
}
+35 -51
View File
@@ -6,9 +6,12 @@ $end_info$
#pragma once
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/BlockSamplingData.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Common/MathUtils.h"
#define XBYAK64
#include <xbyak/xbyak.h>
#include <xbyak/xbyak_util.h>
@@ -18,7 +21,6 @@ using namespace Xbyak;
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/MathUtils.h>
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <tuple>
@@ -29,9 +31,14 @@ struct CodeBuffer {
size_t Size;
};
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
CodeBuffer AllocateNewCodeBuffer(size_t Size);
void FreeCodeBuffer(CodeBuffer Buffer);
}
namespace FEXCore::CPU {
// Temp registers
// rax, rcx, rdx, rsi, r8, r9,
// r10, r11
@@ -56,22 +63,14 @@ const std::array<Xbyak::Xmm, 11> RAXMM_x = { xmm1, xmm2, xmm3, xmm4, xmm5, xmm6
class X86JITCore final : public CPUBackend, public Xbyak::CodeGenerator {
public:
explicit X86JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread,
CodeBuffer Buffer,
bool CompileThread);
explicit X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer, bool CompileThread);
~X86JITCore() override;
std::string GetName() override { return "JIT"; }
void *CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]] std::string GetName() override { return "JIT"; }
void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]] void *CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
bool NeedsOpDispatch() override { return true; }
void ClearCache() override;
@@ -79,19 +78,14 @@ public:
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 256;
void CopyNecessaryDataForCompileThread(CPUBackend *Original) override;
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true, bool IncludeCompileService = true) const override {
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher, IncludeCompileService);
}
private:
Label* PendingTargetLabel{};
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *ThreadState;
FEXCore::IR::IRListView const *IR;
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
uint64_t Entry;
FEXCore::CPU::Dispatcher *Dispatcher;
std::unordered_map<IR::NodeID, Label> JumpTargets;
std::unordered_map<IR::OrderedNodeWrapper::NodeOffsetType, Label> JumpTargets;
Xbyak::util::Cpu Features{};
bool MemoryDebug = false;
@@ -117,27 +111,26 @@ private:
constexpr static uint8_t RA_64 = 3;
constexpr static uint8_t RA_XMM = 4;
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const;
IR::PhysicalRegister GetPhys(uint32_t Node);
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
bool IsFPR(uint32_t Node);
bool IsGPR(uint32_t Node);
template<uint8_t RAType>
[[nodiscard]] Xbyak::Reg GetSrc(IR::NodeID Node) const;
Xbyak::Reg GetSrc(uint32_t Node);
template<uint8_t RAType>
[[nodiscard]] std::pair<Xbyak::Reg, Xbyak::Reg> GetSrcPair(IR::NodeID Node) const;
std::pair<Xbyak::Reg, Xbyak::Reg> GetSrcPair(uint32_t Node);
template<uint8_t RAType>
[[nodiscard]] Xbyak::Reg GetDst(IR::NodeID Node) const;
Xbyak::Reg GetDst(uint32_t Node);
[[nodiscard]] Xbyak::Xmm GetSrc(IR::NodeID Node) const;
[[nodiscard]] Xbyak::Xmm GetDst(IR::NodeID Node) const;
Xbyak::Xmm GetSrc(uint32_t Node);
Xbyak::Xmm GetDst(uint32_t Node);
[[nodiscard]] Xbyak::RegExp GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset,
IR::MemOffsetType OffsetType, uint8_t OffsetScale) const;
Xbyak::RegExp GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr);
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value);
IR::RegisterAllocationPass *RAPass;
FEXCore::IR::RegisterAllocationData *RAData;
@@ -166,11 +159,8 @@ private:
struct CompilerSharedData {
uint64_t SignalHandlerReturnAddress{};
uint64_t UnimplementedInstructionAddress{};
uint32_t *SignalHandlerRefCounterPtr{};
FEXCore::CPU::Dispatcher *Dispatcher{};
};
CompilerSharedData ThreadSharedData;
@@ -182,8 +172,8 @@ private:
std::tuple<SetCC, CMovCC, JCC> GetCC(IR::CondClassType cond);
using OpHandler = void (X86JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
using OpHandler = void (X86JITCore::*)(FEXCore::IR::IROp_Header *IROp, uint32_t Node);
std::array<OpHandler, FEXCore::IR::IROps::OP_LAST + 1> OpHandlers {};
void RegisterALUHandlers();
void RegisterAtomicHandlers();
void RegisterBranchHandlers();
@@ -197,7 +187,7 @@ private:
void PushRegs();
void PopRegs();
#define DEF_OP(x) void Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
///< Unhandled handler
DEF_OP(Unhandled);
@@ -225,7 +215,6 @@ private:
DEF_OP(UMulH);
DEF_OP(Or);
DEF_OP(And);
DEF_OP(Andn);
DEF_OP(Xor);
DEF_OP(Lshl);
DEF_OP(Lshr);
@@ -250,7 +239,9 @@ private:
DEF_OP(Sbfe);
DEF_OP(Select);
DEF_OP(VExtractToGPR);
DEF_OP(Float_ToGPR_ZU);
DEF_OP(Float_ToGPR_ZS);
DEF_OP(Float_ToGPR_U);
DEF_OP(Float_ToGPR_S);
DEF_OP(FCmp);
DEF_OP(F80Cmp);
@@ -269,7 +260,6 @@ private:
DEF_OP(AtomicFetchAnd);
DEF_OP(AtomicFetchOr);
DEF_OP(AtomicFetchXor);
DEF_OP(AtomicFetchNeg);
///< Branch ops
DEF_OP(GuestCallDirect);
@@ -289,14 +279,16 @@ private:
///< Conversion ops
DEF_OP(VInsGPR);
DEF_OP(VCastFromGPR);
DEF_OP(Float_FromGPR_U);
DEF_OP(Float_FromGPR_S);
DEF_OP(Float_FToF);
DEF_OP(Vector_UToF);
DEF_OP(Vector_SToF);
DEF_OP(Vector_FToZU);
DEF_OP(Vector_FToZS);
DEF_OP(Vector_FToU);
DEF_OP(Vector_FToS);
DEF_OP(Vector_FToF);
DEF_OP(Vector_FToI);
///< Flag ops
DEF_OP(GetHostFlag);
@@ -314,7 +306,6 @@ private:
DEF_OP(StoreMem);
DEF_OP(VLoadMemElement);
DEF_OP(VStoreMemElement);
DEF_OP(CacheLineClear);
///< Misc ops
DEF_OP(EndBlock);
@@ -339,7 +330,6 @@ private:
DEF_OP(SplatVector);
DEF_OP(VMov);
DEF_OP(VAnd);
DEF_OP(VBic);
DEF_OP(VOr);
DEF_OP(VXor);
DEF_OP(VAdd);
@@ -350,10 +340,8 @@ private:
DEF_OP(VSQSub);
DEF_OP(VAddP);
DEF_OP(VAddV);
DEF_OP(VUMinV);
DEF_OP(VURAvg);
DEF_OP(VAbs);
DEF_OP(VPopcount);
DEF_OP(VFAdd);
DEF_OP(VFAddP);
DEF_OP(VFSub);
@@ -373,8 +361,6 @@ private:
DEF_OP(VSMax);
DEF_OP(VZip);
DEF_OP(VZip2);
DEF_OP(VUnZip);
DEF_OP(VUnZip2);
DEF_OP(VBSL);
DEF_OP(VCMPEQ);
DEF_OP(VCMPEQZ);
@@ -397,7 +383,6 @@ private:
DEF_OP(VInsElement);
DEF_OP(VInsScalarElement);
DEF_OP(VExtractElement);
DEF_OP(VDupElement);
DEF_OP(VExtr);
DEF_OP(VSLI);
DEF_OP(VSRI);
@@ -420,7 +405,6 @@ private:
DEF_OP(VSMull);
DEF_OP(VUMull2);
DEF_OP(VSMull2);
DEF_OP(VUABDL);
DEF_OP(VTBL1);
///< Encryption ops
@@ -5,19 +5,13 @@ $end_info$
*/
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <array>
#include <stddef.h>
#include <stdint.h>
#include <xbyak/xbyak.h>
#include <cmath>
namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(LoadContext) {
auto Op = IROp->C<IR::IROp_LoadContext>();
@@ -42,10 +36,10 @@ DEF_OP(LoadContext) {
}
break;
case 16: {
LOGMAN_MSG_A_FMT("Invalid GPR load of size 16");
LogMan::Msg::A("Invalid GPR load of size 16");
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
default: LogMan::Msg::A("Unhandled LoadContext size: %d", OpSize);
}
}
else {
@@ -75,7 +69,7 @@ DEF_OP(LoadContext) {
movups(GetDst(Node), xword [STATE + Op->Offset]);
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
default: LogMan::Msg::A("Unhandled LoadContext size: %d", OpSize);
}
}
}
@@ -104,9 +98,9 @@ DEF_OP(StoreContext) {
}
break;
case 16:
LogMan::Msg::DFmt("Invalid store size of 16");
LogMan::Msg::D("Invalid store size of 16");
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
default: LogMan::Msg::A("Unhandled StoreContext size: %d", OpSize);
}
}
else {
@@ -135,14 +129,14 @@ DEF_OP(StoreContext) {
movups(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()));
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
default: LogMan::Msg::A("Unhandled StoreContext size: %d", OpSize);
}
}
}
DEF_OP(LoadContextIndexed) {
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
size_t size = IROp->Size;
size_t size = Op->Size;
Reg index = GetSrc<RA_64>(Op->Header.Args[0].ID());
if (Op->Class.Val == 0) {
@@ -166,18 +160,17 @@ DEF_OP(LoadContextIndexed) {
mov(GetDst<RA_64>(Node), qword [rax + index * Op->Stride]);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
}
break;
}
case 16:
LOGMAN_MSG_A_FMT("Invalid Class load of size 16");
LogMan::Msg::A("Invalid Class load of size 16");
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed stride: {}", Op->Stride);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
}
}
else {
switch (Op->Stride) {
@@ -202,8 +195,7 @@ DEF_OP(LoadContextIndexed) {
vmovq(GetDst(Node), qword [rax + index * Op->Stride]);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
}
break;
}
@@ -231,14 +223,12 @@ DEF_OP(LoadContextIndexed) {
movups(GetDst(Node), xword [STATE + rax]);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed size: %d", Op->Size);
}
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed stride: {}", Op->Stride);
break;
LogMan::Msg::A("Unhandled LoadContextIndexed stride: %d", Op->Stride);
}
}
}
@@ -246,7 +236,7 @@ DEF_OP(LoadContextIndexed) {
DEF_OP(StoreContextIndexed) {
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
Reg index = GetSrc<RA_64>(Op->Header.Args[1].ID());
size_t size = IROp->Size;
size_t size = Op->Size;
if (Op->Class.Val == 0) {
auto value = GetSrc<RA_64>(Op->Header.Args[0].ID());
@@ -258,14 +248,13 @@ DEF_OP(StoreContextIndexed) {
case 4:
case 8: {
if (!(size == 1 || size == 2 || size == 4 || size == 8)) {
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
LogMan::Msg::A("Unhandled StoreContextIndexed size: %d", Op->Size);
}
mov(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
mov(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed stride: {}", Op->Stride);
break;
LogMan::Msg::A("Unhandled StoreContextIndexed stride: %d", Op->Stride);
}
}
else {
@@ -278,20 +267,19 @@ DEF_OP(StoreContextIndexed) {
lea(rax, dword [STATE + Op->BaseOffset]);
switch (size) {
case 1:
pextrb(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
pextrb(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value, 0);
break;
case 2:
pextrw(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
pextrw(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value, 0);
break;
case 4:
vmovd(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
vmovd(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value);
break;
case 8:
vmovq(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
vmovq(AddressFrame(Op->Size * 8) [rax + index * Op->Stride], value);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
break;
LogMan::Msg::A("Unhandled StoreContextIndexed size: %d", size);
}
break;
}
@@ -301,16 +289,16 @@ DEF_OP(StoreContextIndexed) {
lea(rax, dword [rax + Op->BaseOffset]);
switch (size) {
case 1:
pextrb(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
pextrb(AddressFrame(Op->Size * 8) [STATE + rax], value, 0);
break;
case 2:
pextrw(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
pextrw(AddressFrame(Op->Size * 8) [STATE + rax], value, 0);
break;
case 4:
vmovd(AddressFrame(IROp->Size * 8) [STATE + rax], value);
vmovd(AddressFrame(Op->Size * 8) [STATE + rax], value);
break;
case 8:
vmovq(AddressFrame(IROp->Size * 8) [STATE + rax], value);
vmovq(AddressFrame(Op->Size * 8) [STATE + rax], value);
break;
case 16:
if (Op->BaseOffset % 16 == 0)
@@ -319,14 +307,12 @@ DEF_OP(StoreContextIndexed) {
movups(xword [STATE + rax], value);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
break;
LogMan::Msg::A("Unhandled StoreContextIndexed size: %d", size);
}
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed stride: {}", Op->Stride);
break;
LogMan::Msg::A("Unhandled StoreContextIndexed stride: %d", Op->Stride);
}
}
}
@@ -354,7 +340,7 @@ DEF_OP(SpillRegister) {
mov(qword [rsp + SlotOffset], GetSrc<RA_64>(Op->Header.Args[0].ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
}
} else if (Op->Class == FEXCore::IR::FPRClass) {
switch (OpSize) {
@@ -370,10 +356,10 @@ DEF_OP(SpillRegister) {
movaps(xword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize);
}
} else {
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
LogMan::Msg::A("Unhandled SpillRegister class: %d", Op->Class.Val);
}
@@ -402,7 +388,7 @@ DEF_OP(FillRegister) {
mov(GetDst<RA_64>(Node), qword [rsp + SlotOffset]);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
default: LogMan::Msg::A("Unhandled FillRegister size: %d", OpSize);
}
} else if (Op->Class == FEXCore::IR::FPRClass) {
switch (OpSize) {
@@ -418,10 +404,10 @@ DEF_OP(FillRegister) {
movaps(GetDst(Node), xword [rsp + SlotOffset]);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
default: LogMan::Msg::A("Unhandled FillRegister size: %d", OpSize);
}
} else {
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
LogMan::Msg::A("Unhandled FillRegister class: %d", Op->Class.Val);
}
}
@@ -439,16 +425,16 @@ DEF_OP(StoreFlag) {
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
}
Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) const {
Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
if (Offset.IsInvalid()) {
return Base;
} else {
if (OffsetScale != 1 && OffsetScale != 2 && OffsetScale != 4 && OffsetScale != 8) {
LOGMAN_MSG_A_FMT("Unhandled GenerateModRM OffsetScale: {}", OffsetScale);
LogMan::Msg::A("Unhandled GenerateModRM OffsetScale: %d", OffsetScale);
}
if (OffsetType != IR::MEM_OFFSET_SXTX) {
LOGMAN_MSG_A_FMT("Unhandled GenerateModRM OffsetType: {}", OffsetType.Val);
LogMan::Msg::A("Unhandled GenerateModRM OffsetType: %d", OffsetType.Val);
}
uint64_t Const;
@@ -472,7 +458,7 @@ DEF_OP(LoadMem) {
if (Op->Class.Val == 0) {
auto Dst = GetDst<RA_64>(Node);
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
movzx (Dst, byte [MemPtr]);
}
@@ -489,14 +475,14 @@ DEF_OP(LoadMem) {
mov(Dst, qword [MemPtr]);
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
}
}
else
{
auto Dst = GetDst(Node);
switch (IROp->Size) {
switch (Op->Size) {
case 1: {
movzx(eax, byte [MemPtr]);
vmovd(Dst, eax);
@@ -516,7 +502,7 @@ DEF_OP(LoadMem) {
}
break;
case 16: {
if (IROp->Size == Op->Align)
if (Op->Size == Op->Align)
movups(GetDst(Node), xword [MemPtr]);
else
movups(GetDst(Node), xword [MemPtr]);
@@ -525,7 +511,7 @@ DEF_OP(LoadMem) {
}
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size);
}
}
}
@@ -538,7 +524,7 @@ DEF_OP(StoreMem) {
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class.Val == 0) {
switch (IROp->Size) {
switch (Op->Size) {
case 1:
mov(byte [MemPtr], GetSrc<RA_8>(Op->Header.Args[1].ID()));
break;
@@ -551,11 +537,11 @@ DEF_OP(StoreMem) {
case 8:
mov(qword [MemPtr], GetSrc<RA_64>(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
}
}
else {
switch (IROp->Size) {
switch (Op->Size) {
case 1:
pextrb(byte [MemPtr], GetSrc(Op->Header.Args[1].ID()), 0);
break;
@@ -569,30 +555,22 @@ DEF_OP(StoreMem) {
vmovq(qword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
break;
case 16:
if (IROp->Size == Op->Align)
if (Op->Size == Op->Align)
movups(xword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
else
movups(xword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size);
}
}
}
DEF_OP(VLoadMemElement) {
LOGMAN_MSG_A_FMT("Unimplemented");
LogMan::Msg::A("Unimplemented");
}
DEF_OP(VStoreMemElement) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(CacheLineClear) {
auto Op = IROp->C<IR::IROp_CacheLineClear>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
clflush(ptr [MemReg]);
LogMan::Msg::A("Unimplemented");
}
#undef DEF_OP
@@ -614,7 +592,6 @@ void X86JITCore::RegisterMemoryHandlers() {
REGISTER_OP(STOREMEMTSO, StoreMem);
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
#undef REGISTER_OP
}
}
+26 -51
View File
@@ -4,29 +4,15 @@ tags: backend|x86-64
$end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/IR/IR.h>
#include <array>
#include <stddef.h>
#include <stdint.h>
#include <xbyak/xbyak.h>
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
static void PrintValue(uint64_t Value) {
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
LogMan::Msg::D("Value: 0x%lx", Value);
}
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
}
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(Fence) {
auto Op = IROp->C<IR::IROp_Fence>();
@@ -40,29 +26,28 @@ DEF_OP(Fence) {
case IR::Fence_Store.Val:
sfence();
break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
default: LogMan::Msg::A("Unknown Fence: %d", Op->Fence); break;
}
}
DEF_OP(Break) {
auto Op = IROp->C<IR::IROp_Break>();
switch (Op->Reason) {
case FEXCore::IR::Break_Unimplemented: // Hard fault
case FEXCore::IR::Break_Interrupt: // Guest ud2
case FEXCore::IR::Break_Overflow: // overflow
case 0: // Hard fault
case 5: // Guest ud2
ud2();
break;
case FEXCore::IR::Break_Halt: { // HLT
break;
case 4: { // HLT
// Time to quit
// Set our stack to the starting stack location
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
// Now we need to jump to the thread stop handler
mov(TMP1, ThreadSharedData.Dispatcher->ThreadStopHandlerAddress);
mov(TMP1, Dispatcher->ThreadStopHandlerAddress);
jmp(TMP1);
break;
}
case FEXCore::IR::Break_Interrupt3: // INT3
case 6: // INT3
{
if (CTX->GetGdbServerStatus()) {
// Adjust the stack first for a regular return
@@ -71,7 +56,7 @@ DEF_OP(Break) {
}
// This jump target needs to be a constant offset here
mov(TMP1, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddress);
mov(TMP1, Dispatcher->ThreadPauseHandlerAddress);
jmp(TMP1);
}
else {
@@ -80,24 +65,12 @@ DEF_OP(Break) {
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
// Now we need to jump to the thread stop handler
mov(TMP1, ThreadSharedData.Dispatcher->ThreadStopHandlerAddress);
mov(TMP1, Dispatcher->ThreadStopHandlerAddress);
jmp(TMP1);
}
break;
}
case FEXCore::IR::Break_InvalidInstruction:
{
if (SpillSlots) {
add(rsp, SpillSlots * 16);
}
// Need to be outside of JIT cache space to ensure cache clearing correctness
mov(TMP1, ThreadSharedData.Dispatcher->UnimplementedInstructionAddress);
jmp(TMP1);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
default: LogMan::Msg::A("Unknown Break reason: %d", Op->Reason);
}
}
@@ -137,22 +110,24 @@ DEF_OP(SetRoundingMode) {
DEF_OP(Print) {
auto Op = IROp->C<IR::IROp_Print>();
PushRegs();
if (IsGPR(Op->Header.Args[0].ID())) {
mov (rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
for (auto &Reg : RA64)
push(Reg);
mov(rax, reinterpret_cast<uintptr_t>(PrintValue));
}
else {
pextrq(rdi, GetSrc(Op->Header.Args[0].ID()), 0);
pextrq(rsi, GetSrc(Op->Header.Args[0].ID()), 1);
auto NumPush = RA64.size();
if (NumPush & 1)
sub(rsp, 8); // Align
mov(rax, reinterpret_cast<uintptr_t>(PrintVectorValue));
}
mov (rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
mov(rax, reinterpret_cast<uintptr_t>(PrintValue));
call(rax);
PopRegs();
if (NumPush & 1)
add(rsp, 8); // Align
for (uint32_t i = RA64.size(); i > 0; --i)
pop(RA64[i - 1]);
}
#undef DEF_OP
@@ -5,17 +5,11 @@ $end_info$
*/
#include "Interface/Core/JIT/x86_64/JITClass.h"
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/IR/IR.h>
#include <array>
#include <stdint.h>
#include <utility>
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
DEF_OP(ExtractElementPair) {
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
switch (Op->Header.Size) {
@@ -31,7 +25,7 @@ DEF_OP(ExtractElementPair) {
mov (GetDst<RA_64>(Node), Regs[Op->Element]);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Size"); break;
default: LogMan::Msg::A("Unknown Size"); break;
}
}
@@ -57,7 +51,7 @@ DEF_OP(CreateElementPair) {
RegTmp = rax;
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Size"); break;
default: LogMan::Msg::A("Unknown Size"); break;
}
if (Dst.first != RegSecond) {
File diff suppressed because it is too large. Load diff
+9 -12
View File
@@ -5,12 +5,9 @@ desc: Stores information about blocks, and provides C++ implementations to looku
$end_info$
*/
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Context/Context.h"
#include "Interface/Core/Core.h"
#include "Interface/Core/LookupCache.h"
#include <sys/mman.h>
namespace FEXCore {
@@ -29,27 +26,27 @@ LookupCache::LookupCache(FEXCore::Context::Context *CTX)
// Allocate a region of memory that we can use to back our block pointers
// We need one pointer per page of virtual memory
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, ctx->Config.VirtualMemSize / 4096 * 8, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
PagePointer = reinterpret_cast<uintptr_t>(mmap(nullptr, ctx->Config.VirtualMemSize / 4096 * 8, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
// Allocate our memory backing our pages
// We need 32KB per guest page (One pointer per byte)
// XXX: We can drop down to 16KB if we store 4byte offsets from the code base
// We currently limit to 128MB of real memory for caching for the total cache size.
// Can end up being inefficient if we compile a small number of blocks per page
PageMemory = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, CODE_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
PageMemory = reinterpret_cast<uintptr_t>(mmap(nullptr, CODE_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
LogMan::Throw::A(PageMemory != -1ULL, "Failed to allocate page memory");
// L1 Cache
L1Pointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, L1_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
L1Pointer = reinterpret_cast<uintptr_t>(mmap(nullptr, L1_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
LogMan::Throw::A(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
VirtualMemSize = ctx->Config.VirtualMemSize;
}
LookupCache::~LookupCache() {
FEXCore::Allocator::munmap(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8);
FEXCore::Allocator::munmap(reinterpret_cast<void*>(PageMemory), CODE_SIZE);
FEXCore::Allocator::munmap(reinterpret_cast<void*>(L1Pointer), L1_SIZE);
munmap(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8);
munmap(reinterpret_cast<void*>(PageMemory), CODE_SIZE);
munmap(reinterpret_cast<void*>(L1Pointer), L1_SIZE);
}
void LookupCache::HintUsedRange(uint64_t Address, uint64_t Size) {
+9 -23
View File
@@ -1,18 +1,10 @@
#pragma once
#include "Interface/Context/Context.h"
#include <FEXCore/Utils/LogManager.h>
#include <cstdint>
#include <functional>
#include <map>
#include <stddef.h>
#include <utility>
#include <vector>
namespace FEXCore {
namespace Context {
struct Context;
}
class LookupCache {
public:
@@ -46,21 +38,14 @@ public:
std::map<uint64_t, std::vector<uint64_t>> CodePages;
void AddBlockMapping(uint64_t Address, void *HostCode, uint64_t Start, uint64_t Length) {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
auto InsertPoint =
#endif
BlockList.emplace(Address, (uintptr_t)HostCode);
LOGMAN_THROW_A_FMT(InsertPoint.second == true, "Dupplicate block mapping added");
auto InsertPoint = BlockList.emplace(Address, (uintptr_t)HostCode);
LogMan::Throw::A(InsertPoint.second == true, "Dupplicate block mapping added");
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length) >> 12; CurrentPage <= EndPage; CurrentPage++) {
CodePages[CurrentPage].push_back(Address);
}
// There is no need to update L1 or L2, they will get updated on first lookup
// However, adding to L1 here increases performance
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
L1Entry.GuestCode = Address;
L1Entry.HostCode = (uintptr_t)HostCode;
// no need to update L1 or L2, they will get updated on first lookup
}
void Erase(uint64_t Address) {
@@ -109,8 +94,8 @@ public:
void HintUsedRange(uint64_t Address, uint64_t Size);
uintptr_t GetL1Pointer() const { return L1Pointer; }
uintptr_t GetPagePointer() const { return PagePointer; }
uintptr_t GetL1Pointer() { return L1Pointer; }
uintptr_t GetPagePointer() { return PagePointer; }
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
@@ -120,8 +105,9 @@ private:
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
// Do L1
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
L1Entry.GuestCode = Address;
L1Entry.HostCode = HostCode;
if (L1Entry.GuestCode == Address) {
L1Entry.GuestCode = L1Entry.HostCode = 0;
}
// Do ful map
auto FullAddress = Address;
File diff suppressed because it is too large. Load diff
+59 -119
View File
@@ -3,9 +3,7 @@
#include "Interface/Core/Frontend.h"
#include "Interface/Context/Context.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Debug/X86Tables.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/IR.h>
@@ -14,10 +12,9 @@
#include <FEXCore/Utils/LogManager.h>
#include <cstdint>
#include <functional>
#include <map>
#include <stddef.h>
#include <utility>
#include <vector>
#include <set>
namespace FEXCore::IR {
class Pass;
@@ -27,20 +24,18 @@ class OpDispatchBuilder final : public IREmitter {
friend class FEXCore::IR::Pass;
friend class FEXCore::IR::PassManager;
enum class SelectionFlag {
Nothing, // must rely on x86 flags
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
FCMP, // flags were set by a ucomis* / comis*
enum {
FLAGS_OP_NONE, // must rely on x86 flags
FLAGS_OP_CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
FLAGS_OP_AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
FLAGS_OP_FCMP, // flags were set by a ucomis* / comis*
};
public:
SelectionFlag flagsOp{};
uint8_t flagsOpSize{};
OrderedNode* flagsOpDest{};
OrderedNode* flagsOpSrc{};
OrderedNode* flagsOpDestSigned{};
OrderedNode* flagsOpSrcSigned{};
int flagsOp;
uint8_t flagsOpSize;
OrderedNode* flagsOpDest, *flagsOpSrc;
OrderedNode* flagsOpDestSigned, *flagsOpSrcSigned;
FEXCore::Context::Context *CTX{};
bool ShouldDump {false};
@@ -54,7 +49,7 @@ public:
OrderedNode* GetNewJumpBlock(uint64_t RIP) {
auto it = JumpTargets.find(RIP);
LOGMAN_THROW_A_FMT(it != JumpTargets.end(), "Couldn't find block generated for 0x{:x}", RIP);
LogMan::Throw::A(it != JumpTargets.end(), "Couldn't find block generated for 0x%lx", RIP);
return it->second.BlockEntry;
}
@@ -64,7 +59,7 @@ public:
it->second.HaveEmitted = true;
if (CurrentCodeBlock->Wrapped(DualListData.ListBegin()).ID() == it->second.BlockEntry->Wrapped(DualListData.ListBegin()).ID()) return;
if (CurrentCodeBlock->Wrapped(ListData.Begin()).ID() == it->second.BlockEntry->Wrapped(ListData.Begin()).ID()) return;
// We have hit a RIP that is a jump target
// Thus we need to end up in a new block
@@ -72,7 +67,7 @@ public:
}
void StartNewBlock() {
flagsOp = SelectionFlag::Nothing;
flagsOp = FLAGS_OP_NONE;
}
bool FinishOp(uint64_t NextRIP, bool LastOp) {
@@ -86,14 +81,14 @@ public:
// rdi, 0x8
// cmp qword [rdi-8], 0
// jne .label
if (LastOp && !BlockSetRIP) {
if (!BlockSetRIP) {
auto it = JumpTargets.find(NextRIP);
if (it == JumpTargets.end()) {
if (it == JumpTargets.end() && LastOp) {
const uint8_t GPRSize = CTX->GetGPRSize();
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
// If we don't have a jump target to a new block then we have to leave
// Set the RIP to the next instruction and leave
auto RelocatedNextRIP = _EntrypointOffset(NextRIP - Entry, GPRSize);
auto RelocatedNextRIP = _EntrypointOffset(NextRIP - Current_Header->Entry, GPRSize);
_ExitFunction(RelocatedNextRIP);
}
else if (it != JumpTargets.end()) {
@@ -109,8 +104,7 @@ public:
OpDispatchBuilder(FEXCore::Context::Context *ctx);
void ResetWorkingList();
void ResetDecodeFailure() { DecodeFailure = false; }
bool HadDecodeFailure() const { return DecodeFailure; }
bool HadDecodeFailure() { return DecodeFailure; }
void BeginFunction(uint64_t RIP, std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
void Finalize();
@@ -234,9 +228,9 @@ public:
void PopcountOp(OpcodeArgs);
void XLATOp(OpcodeArgs);
enum class Segment {
FS,
GS,
enum Segment {
Segment_FS,
Segment_GS,
};
template<Segment Seg>
void ReadSegmentReg(OpcodeArgs);
@@ -265,6 +259,12 @@ public:
template<size_t ElementSize>
void PSUBQOp(OpcodeArgs);
template<size_t ElementSize>
void PMINUOp(OpcodeArgs);
template<size_t ElementSize>
void PMAXUOp(OpcodeArgs);
void PMINSWOp(OpcodeArgs);
void PMAXSWOp(OpcodeArgs);
template<size_t ElementSize>
void MOVMSKOp(OpcodeArgs);
void MOVMSKOpOne(OpcodeArgs);
template<size_t ElementSize>
@@ -274,16 +274,20 @@ public:
void PSHUFBOp(OpcodeArgs);
template<size_t ElementSize, bool HalfSize, bool Low>
void PSHUFDOp(OpcodeArgs);
void MOVDOp(OpcodeArgs);
template<size_t ElementSize>
void PCMPEQOp(OpcodeArgs);
template<size_t ElementSize>
void PCMPGTOp(OpcodeArgs);
void MOVDOp(OpcodeArgs);
template<size_t ElementSize, bool Scalar, uint32_t SrcIndex>
void PSRLDOp(OpcodeArgs);
template<size_t ElementSize>
void PSRLI(OpcodeArgs);
template<size_t ElementSize>
void PSLLI(OpcodeArgs);
template<size_t ElementSize>
template<size_t ElementSize, bool Scalar, uint32_t SrcIndex>
void PSLL(OpcodeArgs);
template<size_t ElementSize>
template<size_t ElementSize, bool Scalar, uint32_t SrcIndex>
void PSRAOp(OpcodeArgs);
void PSRLDQ(OpcodeArgs);
void PSLLDQ(OpcodeArgs);
@@ -292,21 +296,21 @@ public:
template<size_t ElementSize>
void PAVGOp(OpcodeArgs);
void MOVDDUPOp(OpcodeArgs);
template<size_t DstElementSize>
template<size_t DstElementSize, bool Signed>
void CVTGPR_To_FPR(OpcodeArgs);
template<size_t SrcElementSize, bool HostRoundingMode>
template<size_t SrcElementSize, bool Signed, bool HostRoundingMode>
void CVTFPR_To_GPR(OpcodeArgs);
template<size_t SrcElementSize, bool Widen>
template<size_t SrcElementSize, bool Signed, bool Widen>
void Vector_CVT_Int_To_Float(OpcodeArgs);
template<size_t DstElementSize, size_t SrcElementSize>
void Scalar_CVT_Float_To_Float(OpcodeArgs);
template<size_t DstElementSize, size_t SrcElementSize>
void Vector_CVT_Float_To_Float(OpcodeArgs);
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
template<size_t SrcElementSize, bool Signed, bool Narrow, bool HostRoundingMode>
void Vector_CVT_Float_To_Int(OpcodeArgs);
template<size_t SrcElementSize, bool Widen>
template<size_t SrcElementSize, bool Signed, bool Widen>
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
template<size_t SrcElementSize, bool HostRoundingMode>
template<size_t SrcElementSize, bool Signed, bool Narrow, bool HostRoundingMode>
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
void MASKMOVOp(OpcodeArgs);
void MOVBetweenGPR_FPR(OpcodeArgs);
@@ -320,28 +324,16 @@ public:
void ANDNOp(OpcodeArgs);
template<size_t ElementSize>
void PINSROp(OpcodeArgs);
void InsertPSOp(OpcodeArgs);
template<size_t ElementSize>
void PExtrOp(OpcodeArgs);
template<size_t ElementSize, bool Signed>
void PMULOp(OpcodeArgs);
template<size_t ElementSize>
void PSIGN(OpcodeArgs);
// BMI1 Ops
void ANDNBMIOp(OpcodeArgs);
void BEXTRBMIOp(OpcodeArgs);
void BLSIBMIOp(OpcodeArgs);
void BLSMSKBMIOp(OpcodeArgs);
void BLSRBMIOp(OpcodeArgs);
// BMI2 Ops
void BMI2Shift(OpcodeArgs);
void BZHI(OpcodeArgs);
void MULX(OpcodeArgs);
void RORX(OpcodeArgs);
// ADX Ops
void ADXOp(OpcodeArgs);
template<size_t ElementSize>
void PABS(OpcodeArgs);
// X87 Ops
template<size_t width>
@@ -401,8 +393,6 @@ public:
void X87FRSTOR(OpcodeArgs);
void X87FXAM(OpcodeArgs);
void X87FCMOV(OpcodeArgs);
void X87EMMS(OpcodeArgs);
void X87FFREE(OpcodeArgs);
void FXCH(OpcodeArgs);
@@ -468,8 +458,6 @@ public:
template<uint8_t FenceType>
void FenceOp(OpcodeArgs);
void StoreFenceOrCLFlush(OpcodeArgs);
void PSADBW(OpcodeArgs);
void AESImcOp(OpcodeArgs);
@@ -479,27 +467,8 @@ public:
void AESDecLastOp(OpcodeArgs);
void AESKeyGenAssist(OpcodeArgs);
template<size_t ElementSize, size_t DstElementSize, bool Signed>
void ExtendVectorElements(OpcodeArgs);
template<size_t ElementSize, bool Scalar>
void VectorRound(OpcodeArgs);
template<size_t ElementSize>
void VectorBlend(OpcodeArgs);
template<size_t ElementSize>
void VectorVariableBlend(OpcodeArgs);
void PTestOp(OpcodeArgs);
void PHMINPOSUWOp(OpcodeArgs);
template<size_t ElementSize>
void DPPOp(OpcodeArgs);
void MPSADBWOp(OpcodeArgs);
void UnimplementedOp(OpcodeArgs);
void InvalidOp(OpcodeArgs);
#undef OpcodeArgs
void SetPackedRFLAG(bool Lower8, OrderedNode *Src);
@@ -522,34 +491,13 @@ private:
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align);
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align);
[[nodiscard]] static uint32_t GPROffset(X86State::X86Reg reg) {
LOGMAN_THROW_A_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
return static_cast<uint32_t>(offsetof(Core::CPUState, gregs[static_cast<size_t>(reg)]));
}
[[nodiscard]] static uint32_t MMBaseOffset() {
return static_cast<uint32_t>(offsetof(Core::CPUState, mm[0][0]));
}
[[nodiscard]] uint8_t GetDstSize(X86Tables::DecodedOp Op) const;
[[nodiscard]] uint8_t GetSrcSize(X86Tables::DecodedOp Op) const;
[[nodiscard]] uint32_t GetDstBitSize(X86Tables::DecodedOp Op) const;
[[nodiscard]] uint32_t GetSrcBitSize(X86Tables::DecodedOp Op) const;
uint8_t GetDstSize(FEXCore::X86Tables::DecodedOp Op);
uint8_t GetSrcSize(FEXCore::X86Tables::DecodedOp Op);
template<unsigned BitOffset>
void SetRFLAG(OrderedNode *Value) {
flagsOp = SelectionFlag::Nothing;
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
}
void SetRFLAG(OrderedNode *Value, unsigned BitOffset) {
flagsOp = SelectionFlag::Nothing;
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
}
OrderedNode *GetRFLAG(unsigned BitOffset) {
return _LoadFlag(BitOffset);
}
void SetRFLAG(OrderedNode *Value);
void SetRFLAG(OrderedNode *Value, unsigned BitOffset);
OrderedNode *GetRFLAG(unsigned BitOffset);
OrderedNode *SelectCC(uint8_t OP, OrderedNode *TrueValue, OrderedNode *FalseValue);
@@ -572,22 +520,14 @@ private:
void GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
OrderedNode * GetX87Top();
enum class X87Tag {
Valid = 0b00,
Zero = 0b01,
Special = 0b10,
Empty = 0b11
};
void SetX87TopTag(OrderedNode *Value, X87Tag Tag);
OrderedNode *GetX87FTW(OrderedNode *Value);
void SetX87Top(OrderedNode *Value);
bool DestIsLockedMem(FEXCore::X86Tables::DecodedOp Op) const {
return DestIsMem(Op) && (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0;
bool DestIsLockedMem(FEXCore::X86Tables::DecodedOp Op) {
return Op->Dest.TypeNone.Type !=FEXCore::X86Tables::DecodedOperand::TYPE_GPR && (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK);
}
bool DestIsMem(FEXCore::X86Tables::DecodedOp Op) const {
return !Op->Dest.IsGPR();
bool DestIsMem(FEXCore::X86Tables::DecodedOp Op) {
return Op->Dest.TypeNone.Type !=FEXCore::X86Tables::DecodedOperand::TYPE_GPR;
}
void CreateJumpBlocks(std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
@@ -598,16 +538,16 @@ private:
OrderedNode* _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Align = 1) {
if (CTX->Config.TSOEnabled)
return _StoreMemTSO(ssa0, ssa1, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
return _StoreMemTSO(ssa0, ssa1, Invalid(), Size, Align, Class, MEM_OFFSET_SXTX, 1);
else
return _StoreMem(ssa0, ssa1, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
return _StoreMem(ssa0, ssa1, Invalid(), Size, Align, Class, MEM_OFFSET_SXTX, 1);
}
OrderedNode* _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
if (CTX->Config.TSOEnabled)
return _LoadMemTSO(ssa0, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
return _LoadMemTSO(ssa0, Invalid(), Size, Align, Class, MEM_OFFSET_SXTX, 1);
else
return _LoadMem(ssa0, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
return _LoadMem(ssa0, Invalid(), Size, Align, Class, MEM_OFFSET_SXTX, 1);
}
@@ -1,63 +0,0 @@
/*
$info$
tags: frontend|x86-to-ir, opcodes|dispatcher-implementations
desc: Handles x86/64 Crypto instructions to IR
$end_info$
*/
#include <FEXCore/Debug/X86Tables.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Core/OpcodeDispatcher.h"
#include <stdint.h>
namespace FEXCore::IR {
class OrderedNode;
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
auto Res = _VAESImc(Src);
StoreResult(FPRClass, Op, Res, -1);
}
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
auto Res = _VAESEnc(Dest, Src);
StoreResult(FPRClass, Op, Res, -1);
}
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
auto Res = _VAESEncLast(Dest, Src);
StoreResult(FPRClass, Op, Res, -1);
}
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
auto Res = _VAESDec(Dest, Src);
StoreResult(FPRClass, Op, Res, -1);
}
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
auto Res = _VAESDecLast(Dest, Src);
StoreResult(FPRClass, Op, Res, -1);
}
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
uint64_t RCON = Op->Src[1].Data.Literal.Value;
auto Res = _VAESKeyGenAssist(Src, RCON);
StoreResult(FPRClass, Op, Res, -1);
}
}
Loaded 100 of 569 files, more files were not shown because too many files have changed in this diff. Show more