Compare commits

..
Author SHA1 Message Date
Ryan Houdek ea20429351 Docs: Update for release FEX-2507.1 2025-07-11 11:37:44 -07:00
Alyssa Rosenzweig 91828efa7a JIT: fix divisor masking
oversight. should fix Steam.

Fixes: de4becc26 ("OpcodeDispatcher: mask certain divisors")
Closes: #4652
Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
2025-07-11 11:34:45 -07:00
Billy Laws cce605d5e0 PoolBufferWithTimedRetirement: Unclaim in dtor
Buffers are tied to the lifetime of their owned flag, and as that
is a member of PoolBufferWithTimedRetirement we must always unclaim here.

Avoids the need to manually remember this quirk (which was forgot for the
temporary compilation buffer in JIT.cpp) at every use-site.
2025-07-11 11:34:07 -07:00
666 changed files with 62804 additions and 81637 deletions

No files matched your search

+2 -2
View File
@@ -32,7 +32,7 @@ AttributeMacros:
BinPackArguments: true
BinPackParameters: true
BitFieldColonSpacing: Both
BreakAfterAttributes: Leave
BreakAfterAttributes: Always # clang 16 required
BreakBeforeBraces: Attach
BreakBeforeBinaryOperators: None
BreakBeforeInlineASMColon: OnlyMultiline # clang 16 required
@@ -60,7 +60,7 @@ IndentRequires: false
IndentWidth: 2
InsertBraces: true
KeepEmptyLinesAtTheStartOfBlocks: true
LambdaBodyIndentation: Signature
LambdaBodyIndentation: OuterScope
LineEnding: LF # clang 16 required
MaxEmptyLinesToKeep: 2
NamespaceIndentation: Inner
+4
View File
@@ -1,4 +1,8 @@
# This file is used to ignore files and directories from clang-format
# Ignore all files in the External directory
External/*
Source/Common/cpp-optparse/*
# Files with human-indented tables for readability - don't mess with these
-6
View File
@@ -16,9 +16,3 @@
# Reformat of CodeEmitter inl files
8760c593ece92d7e9fa94c40da0368fd367c9cad
# Whole-tree reformat with clang-format-19
5267cde60e7642852d18f20ae8568643bb5293d5
# Minor reformat with clang-format-19
9fdd96af61c969cb5732471223f00eda64b7a069
+1 -4
View File
@@ -13,7 +13,6 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_PORTABLE: 1
jobs:
build_plus_test:
@@ -34,6 +33,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
@@ -136,9 +136,6 @@ jobs:
- name: FEXLinuxTests
working-directory: ${{runner.workspace}}/build
shell: bash
env:
# These tests require non-portable install due to thunks.
FEX_PORTABLE: 0
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
- name: FEXLinuxTests Results move
+1 -1
View File
@@ -20,7 +20,6 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_PORTABLE: 1
jobs:
glibc_fault_test:
@@ -41,6 +40,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1 -1
View File
@@ -13,7 +13,6 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_PORTABLE: 1
jobs:
hostrunner_tests:
@@ -34,6 +33,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -33,6 +33,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+2 -1
View File
@@ -48,6 +48,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
@@ -77,7 +78,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
+11 -4
View File
@@ -40,8 +40,11 @@ jobs:
echo "Formatting files:"
echo "$CHANGED_FILES"
- name: Check git-clang-format-19 exists
run: which git-clang-format-19
- name: Check for correct clang-format version
run: clang-format --version | grep -qF '16.0.6'
- name: Check git-clang-format-16 exists
run: which git-clang-format-16
- name: Setup Python env
uses: actions/setup-python@v4
@@ -55,15 +58,19 @@ jobs:
- name: Run code formatter
env:
CLANG_FORMAT_PATH: 'git-clang-format-19'
CLANG_FORMAT_PATH: 'git-clang-format-16'
GITHUB_PR_NUMBER: ${{ github.event.pull_request.number }}
START_REV: ${{ github.event.pull_request.base.sha }}
END_REV: ${{ github.event.pull_request.head.sha }}
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
# TODO(pmatos): Once we adopt v18, we should be able
# to take advantage of the new --diff_from_common_commit option
# explicitly in code-format-helper.py and not have to diff starting at
# the merge base.
run: |
python ./External/code-format-helper/code-format-helper.py \
--repo "FEX-emu/FEX" \
--issue-number $GITHUB_PR_NUMBER \
--start-rev $START_REV \
--start-rev $(git merge-base $START_REV $END_REV) \
--end-rev $END_REV \
--changed-files "$CHANGED_FILES"
+1 -1
View File
@@ -13,7 +13,6 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_PORTABLE: 1
jobs:
vixl_simulator:
@@ -35,6 +34,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
-88
View File
@@ -1,88 +0,0 @@
name: Wine DLL artifacts
on:
push:
branches:
- main
env:
BUILD_TYPE: Release
jobs:
wine_dll_artifacts:
runs-on: ${{ matrix.arch }}
strategy:
matrix:
arch: [[self-hosted, ARM64, mingw]]
fail-fast: false
steps:
- uses: actions/checkout@v3
- name: Add MingGW to PATH
run: echo "$HOME/llvm-mingw/build/bin/" >> $GITHUB_PATH
- name : submodule checkout
# Need to update submodules
run: |
git submodule sync --recursive
git submodule update --init --depth 1
- name: Clean install directory
run: |
rm -Rf ${{runner.workspace}}/build_install
mkdir ${{runner.workspace}}/build_install
- name: Clean Build Environment
run: |
rm -Rf ${{runner.workspace}}/build_arm64ec
rm -Rf ${{runner.workspace}}/build_wow64
- name: Create Build Environment arm64ec
run: |
cmake -E make_directory ${{runner.workspace}}/build_arm64ec
cmake -E make_directory ${{runner.workspace}}/build_wow64
- name: Configure CMake arm64ec
shell: bash
working-directory: ${{runner.workspace}}/build_arm64ec
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
- name: Configure CMake wow64
shell: bash
working-directory: ${{runner.workspace}}/build_wow64
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
- name: Build arm64ec
working-directory: ${{runner.workspace}}/build_arm64ec
shell: bash
run: cmake --build . --config $BUILD_TYPE
- name: install arm64ec
working-directory: ${{runner.workspace}}/build_arm64ec
shell: bash
env:
DESTDIR: ${{runner.workspace}}/build_install
run: cmake --build . --config $BUILD_TYPE -t install
- name: Build wow64
working-directory: ${{runner.workspace}}/build_wow64
shell: bash
run: cmake --build . --config $BUILD_TYPE
- name: install wow64
working-directory: ${{runner.workspace}}/build_wow64
shell: bash
env:
DESTDIR: ${{runner.workspace}}/build_install
run: cmake --build . --config $BUILD_TYPE -t install
- name: Upload libraries
uses: 'actions/upload-artifact@v4'
timeout-minutes: 1
with:
overwrite: true
name: wine_dll_artifacts
path: ${{runner.workspace}}/build_install/usr/lib/wine/aarch64-windows/lib*.dll
retention-days: 60
compression-level: 9
-3
View File
@@ -46,6 +46,3 @@
[submodule "External/tracy"]
path = External/tracy
url = https://github.com/wolfpld/tracy
[submodule "External/range-v3"]
path = External/range-v3
url = https://github.com/ericniebler/range-v3.git
+54 -31
View File
@@ -4,6 +4,7 @@ project(FEX C CXX ASM)
INCLUDE (CheckIncludeFiles)
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
option(BUILD_THUNKS "Build thunks" FALSE)
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
@@ -38,16 +39,10 @@ set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
set (X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
set (DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
set (HOSTLIBS_DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
if (NOT DATA_DIRECTORY)
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
endif()
include(GNUInstallDirs)
if (NOT HOSTLIBS_DATA_DIRECTORY)
set(HOSTLIBS_DATA_DIRECTORY "${CMAKE_INSTALL_FULL_LIBDIR}/fex-emu")
endif()
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
if (NOT CONTAINS_MINGW EQUAL -1)
message (STATUS "Mingw build")
@@ -303,8 +298,7 @@ set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-poin
include_directories(External/robin-map/include/)
include(CTest)
if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
add_subdirectory(External/vixl/)
include_directories(SYSTEM External/vixl/src/)
endif()
@@ -335,7 +329,7 @@ endif()
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
if (BUILD_TESTING)
if (BUILD_TESTS)
find_package(Catch2 3 QUIET)
if (NOT Catch2_FOUND)
add_subdirectory(External/Catch2/)
@@ -345,9 +339,6 @@ if (BUILD_TESTING)
endif()
include(Catch)
else ()
# Override any previously generated test list to avoid running stale test binaries
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
endif()
find_package(fmt QUIET)
@@ -357,12 +348,6 @@ if (NOT fmt_FOUND)
add_subdirectory(External/fmt/)
endif()
find_package(range-v3 QUIET)
if (NOT range-v3_FOUND)
add_subdirectory(External/range-v3/)
target_compile_definitions(range-v3 INTERFACE RANGES_DISABLE_DEPRECATED_WARNINGS)
endif()
add_subdirectory(External/tiny-json/)
include_directories(External/tiny-json/)
@@ -421,13 +406,6 @@ if (TUNE_CPU STREQUAL "native")
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/NeedDisabledSVE.py"
RESULT_VARIABLE NEEDS_SVE_DISABLED)
if (NEEDS_SVE_DISABLED)
message(STATUS "Platform has bugged SVE. Disabling")
set(AARCH64_CPU "cortex-a78")
endif()
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
if(COMPILER_SUPPORTS_CPU_TYPE)
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${AARCH64_CPU}")
@@ -458,8 +436,13 @@ endif()
add_compile_options(-Wall)
if (BUILD_TESTING)
include(CTest)
if (BUILD_TESTS)
message(STATUS "Unit tests are enabled")
if (NOT BUILD_TESTING)
# CMake checks this variable before generating CTestTestfile.cmake
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
endif()
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
if (TEST_JOB_COUNT)
@@ -490,11 +473,10 @@ file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.js
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/)
endforeach()
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
@@ -555,7 +537,6 @@ if (BUILD_THUNKS)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)"
DEPENDS guest-libs
COMPONENT Runtime
)
install(
@@ -565,7 +546,6 @@ if (BUILD_THUNKS)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)"
DEPENDS guest-libs-32
COMPONENT Runtime
)
add_custom_target(uninstall_guest-libs
@@ -607,3 +587,46 @@ if (OVERRIDE_VERSION STREQUAL "detect")
else()
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
endif()
# Parse the version here
# Change something like `FEX-2106.1-76-<hash>` in to a list
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
# Extract the `2106.1` element
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
# Change `2106.1` in to a list
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
# Calculate list size
list(LENGTH DESCRIBE_LIST LIST_SIZE)
# Pull out the major version
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
# Minor version only exists if there is a .1 at the end
# eg: 2106 versus 2106.1
if (LIST_SIZE GREATER 1)
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
endif()
# Package creation
set (CPACK_GENERATOR "DEB")
set (CPACK_PACKAGE_NAME fex-emu)
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/Description.txt")
# Debian defines
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/triggers")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
# binfmt_misc conflicts with qemu-user-static
# We also only install binfmt_misc on aarch64 hosts
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
endif()
include (CPack)
+35 -53
View File
@@ -36,33 +36,24 @@ public:
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
void adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
if (IsADRRange(Imm)) [[likely]] {
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, ForwardLabel* Label) {
void adr(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADR});
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return adr(rd, &Label->Backward);
adr(rd, &Label->Backward);
} else {
return adr(rd, &Label->Forward);
adr(rd, &Label->Forward);
}
}
@@ -71,42 +62,32 @@ public:
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
void adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) [[likely]] {
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
void adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADRP});
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return adrp(rd, &Label->Backward);
adrp(rd, &Label->Backward);
} else {
return adrp(rd, &Label->Forward);
adrp(rd, &Label->Forward);
}
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
if (IsADRRange(Imm)) {
// If the range is in ADR range then we can just use ADR.
return adr(rd, Label);
adr(rd, Label);
} else if (IsADRPRange(Imm)) {
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
@@ -121,28 +102,23 @@ public:
// Now even an add
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
}
return BranchEncodeSucceeded::Success;
} else {
LOGMAN_MSG_A_FMT("Unscaled offset too large");
FEX_UNREACHABLE;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
// Emit a register index and a nop. These will be backpatched.
dc32(rd.Idx());
nop();
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return LongAddressGen(rd, &Label->Backward);
LongAddressGen(rd, &Label->Backward);
} else {
return LongAddressGen(rd, &Label->Forward);
LongAddressGen(rd, &Label->Forward);
}
}
@@ -198,7 +174,7 @@ public:
// Logical immediate
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
and_(s, rd, rn, n, immr, imms);
}
@@ -209,7 +185,7 @@ public:
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
ands(s, rd, rn, n, immr, imms);
}
@@ -220,14 +196,14 @@ public:
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
orr(s, rd, rn, n, immr, imms);
}
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
uint32_t n, immr, imms;
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
eor(s, rd, rn, n, immr, imms);
}
@@ -357,7 +333,7 @@ public:
bfi(s, rd, Reg::zr, lsb, width);
}
void bfxil(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
const auto reg_size_bits = RegSizeInBits(s);
[[maybe_unused]] const auto reg_size_bits = RegSizeInBits(s);
const auto lsb_p_width = lsb + width;
LOGMAN_THROW_A_FMT(width >= 1, "bfxil needs width >= 1");
@@ -886,6 +862,12 @@ public:
}
private:
static constexpr Condition InvertCondition(Condition cond) {
// These behave as always, so it makes no sense to allow inverting these.
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
}
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b001'0010'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
@@ -995,7 +977,7 @@ private:
}
void xbfiz_helper(bool is_signed, ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
const auto lsb_p_width = lsb + width;
[[maybe_unused]] const auto lsb_p_width = lsb + width;
const auto reg_size_bits = RegSizeInBits(s);
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}", lsb_p_width, reg_size_bits, lsb, width);
File diff suppressed because it is too large. Load diff
+63 -123
View File
@@ -20,31 +20,23 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm);
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
void b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
void b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
void b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return b(Cond, &Label->Backward);
b(Cond, &Label->Backward);
} else {
return b(Cond, &Label->Forward);
b(Cond, &Label->Forward);
}
}
@@ -53,32 +45,24 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm);
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
void bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
void bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return bc(Cond, &Label->Backward);
bc(Cond, &Label->Backward);
} else {
return bc(Cond, &Label->Forward);
bc(Cond, &Label->Forward);
}
}
@@ -114,32 +98,25 @@ public:
UnconditionalBranch(Op, Imm);
}
[[nodiscard]] BranchEncodeSucceeded b(const BackwardLabel* Label) {
void b(const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0001'01 << 26;
// Can't encode.
return BranchEncodeSucceeded::Failure;
UnconditionalBranch(Op, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded b(ForwardLabel* Label) {
void b(ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded b(BiDirectionalLabel* Label) {
void b(BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return b(&Label->Backward);
b(&Label->Backward);
} else {
return b(&Label->Forward);
b(&Label->Forward);
}
}
@@ -149,33 +126,25 @@ public:
UnconditionalBranch(Op, Imm);
}
[[nodiscard]] BranchEncodeSucceeded bl(const BackwardLabel* Label) {
void bl(const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b1001'01 << 26;
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
UnconditionalBranch(Op, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded bl(ForwardLabel* Label) {
void bl(ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded bl(BiDirectionalLabel* Label) {
void bl(BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return bl(&Label->Backward);
bl(&Label->Backward);
} else {
return bl(&Label->Forward);
bl(&Label->Forward);
}
}
@@ -186,35 +155,28 @@ public:
CompareAndBranch(Op, s, rt, Imm);
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0100 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return cbz(s, rt, &Label->Backward);
cbz(s, rt, &Label->Backward);
} else {
return cbz(s, rt, &Label->Forward);
cbz(s, rt, &Label->Forward);
}
}
@@ -224,35 +186,28 @@ public:
CompareAndBranch(Op, s, rt, Imm);
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0101 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return cbnz(s, rt, &Label->Backward);
cbnz(s, rt, &Label->Backward);
} else {
return cbnz(s, rt, &Label->Forward);
cbnz(s, rt, &Label->Forward);
}
}
@@ -262,35 +217,28 @@ public:
TestAndBranch(Op, rt, Bit, Imm);
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0110 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return tbz(rt, Bit, &Label->Backward);
tbz(rt, Bit, &Label->Backward);
} else {
return tbz(rt, Bit, &Label->Forward);
tbz(rt, Bit, &Label->Forward);
}
}
@@ -299,35 +247,27 @@ public:
TestAndBranch(Op, rt, Bit, Imm);
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0111 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return tbnz(rt, Bit, &Label->Backward);
tbnz(rt, Bit, &Label->Backward);
} else {
return tbnz(rt, Bit, &Label->Forward);
tbnz(rt, Bit, &Label->Forward);
}
}
+15 -61
View File
@@ -12,7 +12,6 @@
#include <CodeEmitter/Registers.h>
#include <array>
#include <bit>
#include <cstdint>
#include <utility>
#include <type_traits>
@@ -87,14 +86,6 @@ constexpr size_t SubRegSizeInBits(SubRegSize size) {
return size_t {8} << FEXCore::ToUnderlying(size);
}
// Many floating point operations constrain their element sizes to the
// main three float sizes half, single, and double precision. This just
// combines all the checks together for brevity.
[[nodiscard]]
constexpr bool IsStandardFloatSize(SubRegSize size) {
return size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit;
}
/* This `ScalarRegSize` enum is used for most scalar float
* operations.
*
@@ -586,11 +577,6 @@ concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegi
template<typename T>
concept IsQOrDRegister = std::is_same_v<T, QRegister> || std::is_same_v<T, DRegister>;
enum class BranchEncodeSucceeded {
Success,
Failure,
};
// Whether or not a given set of vector registers are sequential
// in increasing order as far as the register file is concerned (modulo its size)
//
@@ -643,25 +629,19 @@ public:
// Bind a backward label to an address.
// Address that is bound is the current emitter location.
[[nodiscard]] bool Bind(BackwardLabel* Label) {
void Bind(BackwardLabel* Label) {
LOGMAN_THROW_A_FMT(Label->Location == nullptr, "Trying to bind a label twice");
Label->Location = GetCursorAddress<uint8_t*>();
// Always binds because it is only storing a location.
return true;
}
[[nodiscard]] bool Bind(const ForwardLabel::Reference* Label) {
void Bind(const ForwardLabel::Reference* Label) {
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
// Patch up the instructions
switch (Label->Type) {
case ForwardLabel::InstType::ADR: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!IsADRRange(Imm)) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
@@ -673,12 +653,7 @@ public:
case ForwardLabel::InstType::ADRP: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
@@ -688,13 +663,11 @@ public:
*Instruction = Inst;
break;
}
case ForwardLabel::InstType::B: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -704,13 +677,11 @@ public:
break;
}
case ForwardLabel::InstType::TEST_BRANCH: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -724,10 +695,7 @@ public:
case ForwardLabel::InstType::RELATIVE_LOAD: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -776,41 +744,27 @@ public:
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
return true;
}
// Bind a forward label to a location.
// This walks all the instructions in the label's vector.
// Then backpatching all instructions that have used the label.
[[nodiscard]] bool Bind(ForwardLabel* Label) {
bool Bound = true;
void Bind(ForwardLabel* Label) {
if (Label->FirstInst.Location) {
Bound &= Bind(&Label->FirstInst);
Bind(&Label->FirstInst);
}
for (auto& Inst : Label->Insts) {
Bound &= Bind(&Inst);
Bind(&Inst);
}
return Bound;
}
// Bind a bidirectional location to a location.
// Binds both forwards and backwards depending on how the label was used.
[[nodiscard]] bool Bind(BiDirectionalLabel* Label) {
bool Bound = true;
void Bind(BiDirectionalLabel* Label) {
if (!Label->Backward.Location) {
Bound &= Bind(&Label->Backward);
Bind(&Label->Backward);
}
Bound &= Bind(&Label->Forward);
return Bound;
}
static constexpr Condition InvertCondition(Condition cond) {
// These behave as always, so it makes no sense to allow inverting these.
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
Bind(&Label->Forward);
}
#include <CodeEmitter/VixlUtils.inl>
+48 -41
View File
@@ -60,7 +60,8 @@ public:
}
void fcmla(SubRegSize size, ZRegister zda, PRegisterMerge pv, ZRegister zn, ZRegister zm, Rotation rot) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pv <= PReg::p7.Merging(), "fcmla can only use p0 to p7");
uint32_t Op = 0b0110'0100'0000'0000'0000'0000'0000'0000;
@@ -75,7 +76,8 @@ public:
}
void fcadd(SubRegSize size, ZRegister zd, PRegisterMerge pv, ZRegister zn, ZRegister zm, Rotation rot) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pv <= PReg::p7.Merging(), "fcadd can only use p0 to p7");
LOGMAN_THROW_A_FMT(rot == Rotation::ROTATE_90 || rot == Rotation::ROTATE_270, "fcadd rotation may only be 90 or 270 degrees");
LOGMAN_THROW_A_FMT(zd == zn, "fcadd zd and zn must be the same register");
@@ -813,12 +815,16 @@ public:
// SVE Integer Misc - Unpredicated
// SVE floating-point trig select coefficient
void ftssel(SubRegSize size, ZRegister zd, ZRegister zn, ZRegister zm) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "ftssel may only use 16/32/64-bit element sizes");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "ftssel may only have "
"16-bit, 32-bit, or 64-bit "
"element sizes");
SVEIntegerMiscUnpredicated(0b00, zm.Idx(), FEXCore::ToUnderlying(size), zd, zn);
}
// SVE floating-point exponential accelerator
void fexpa(SubRegSize size, ZRegister zd, ZRegister zn) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "fexpa may only use 16/32/64-bit element sizes");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "fexpa may only have "
"16-bit, 32-bit, or 64-bit "
"element sizes");
SVEIntegerMiscUnpredicated(0b10, 0b00000, FEXCore::ToUnderlying(size), zd, zn);
}
// SVE constructive prefix (unpredicated)
@@ -1497,9 +1503,9 @@ public:
}
// SVE broadcast floating-point immediate (unpredicated)
void fdup(SubRegSize size, ZRegister zd, float Value) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Unsupported fmov size");
void fdup(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, float Value) {
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i16Bit || size == ARMEmitter::SubRegSize::i32Bit || size == ARMEmitter::SubRegSize::i64Bit,
"Unsupported fmov size");
uint32_t Imm {};
if (size == SubRegSize::i16Bit) {
LOGMAN_MSG_A_FMT("Unsupported");
@@ -1512,7 +1518,7 @@ public:
SVEBroadcastFloatImmUnpredicated(0b00, 0, Imm, size, zd);
}
void fmov(SubRegSize size, ZRegister zd, float Value) {
void fmov(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, float Value) {
fdup(size, zd, Value);
}
@@ -1541,7 +1547,7 @@ public:
void sqincp(SubRegSize size, XRegister rdn, PRegister pm) {
SVEIncDecPredicateCountScalar(0, 1, 0b10, 0b00, size, rdn, pm);
}
void sqincp(SubRegSize size, XRegister rdn, PRegister pm, WRegister wn) {
void sqincp(SubRegSize size, XRegister rdn, PRegister pm, [[maybe_unused]] WRegister wn) {
LOGMAN_THROW_A_FMT(rdn.Idx() == wn.Idx(), "rdn and wn must be the same");
SVEIncDecPredicateCountScalar(0, 1, 0b00, 0b00, size, rdn, pm);
}
@@ -1554,7 +1560,7 @@ public:
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm) {
SVEIncDecPredicateCountScalar(0, 1, 0b10, 0b10, size, rdn, pm);
}
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm, WRegister wn) {
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm, [[maybe_unused]] WRegister wn) {
LOGMAN_THROW_A_FMT(rdn.Idx() == wn.Idx(), "rdn and wn must be the same");
SVEIncDecPredicateCountScalar(0, 1, 0b00, 0b10, size, rdn, pm);
}
@@ -3296,7 +3302,7 @@ private:
const auto log2_size_bytes = FEXCore::ilog2(size_bytes);
// We can index up to 512-bit registers with dup
const auto max_index = (64U >> log2_size_bytes) - 1;
[[maybe_unused]] const auto max_index = (64U >> log2_size_bytes) - 1;
LOGMAN_THROW_A_FMT(Index <= max_index, "dup index ({}) too large. Must be within [0, {}].", Index, max_index);
// imm2:tsz make up a 7 bit wide field, with each increasing element size
@@ -3326,7 +3332,7 @@ private:
uint32_t shift = 0;
if (!is_uint8_imm) {
const bool is_uint16_imm = (imm >> 16) == 0;
[[maybe_unused]] const bool is_uint16_imm = (imm >> 16) == 0;
LOGMAN_THROW_A_FMT(is_uint16_imm, "Immediate ({}) must be a 16-bit value within [256, 65280]", imm);
LOGMAN_THROW_A_FMT((imm % 256) == 0, "Immediate ({}) must be a multiple of 256", imm);
@@ -3395,8 +3401,8 @@ private:
}
void SVEBroadcastFloatImmPredicated(SubRegSize size, ZRegister zd, PRegister pg, float value) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Unsupported fcpy/fmov size");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Unsupported fcpy/fmov "
"size");
uint32_t imm {};
if (size == SubRegSize::i16Bit) {
LOGMAN_MSG_A_FMT("Unsupported");
@@ -3572,7 +3578,7 @@ private:
// SVE2 floating-point pairwise operations
void SVEFloatPairwiseArithmetic(uint32_t opc, SubRegSize size, PRegister pg, ZRegister zd, ZRegister zn, ZRegister zm) {
LOGMAN_THROW_A_FMT(zd == zn, "zd needs to equal zn");
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Invalid float size");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Invalid float size");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0100'0001'0000'1000'0000'0000'0000;
@@ -3586,7 +3592,7 @@ private:
// SVE floating-point arithmetic (unpredicated)
void SVEFloatArithmeticUnpredicated(uint32_t opc, SubRegSize size, ZRegister zm, ZRegister zn, ZRegister zd) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Invalid float size");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Invalid float size");
uint32_t Instr = 0b0110'0101'0000'0000'0000'0000'0000'0000;
Instr |= FEXCore::ToUnderlying(size) << 22;
@@ -3694,7 +3700,7 @@ private:
// SVE floating-point arithmetic (predicated)
void SVEFloatArithmeticPredicated(uint32_t opc, SubRegSize size, PRegister pg, ZRegister zd, ZRegister zn, ZRegister zm) {
LOGMAN_THROW_A_FMT(zd == zn, "zn needs to equal zd");
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Invalid float size");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Invalid float size");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0101'0000'0000'1000'0000'0000'0000;
@@ -3722,7 +3728,9 @@ private:
}
void SVEFPRecursiveReduction(uint32_t opc, SubRegSize size, VRegister vd, PRegister pg, ZRegister zn) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "FP reduction operation can only use 16/32/64-bit element sizes");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "FP reduction operation can "
"only use 16-bit, 32-bit, "
"or 64-bit element sizes");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "FP reduction operation can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0101'0000'0000'0010'0000'0000'0000;
@@ -4104,7 +4112,7 @@ private:
// 0b111 - I - Current
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Unsupported size in {}", __func__);
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Unsupported size in {}", __func__);
uint32_t Instr = 0b0110'0101'0000'0000'1010'0000'0000'0000;
Instr |= FEXCore::ToUnderlying(size) << 22;
@@ -4152,7 +4160,7 @@ private:
const auto& op_data = mem_op.MetaType.ScalarVectorType;
const bool is_scaled = op_data.scale != 0;
const auto msize_value = FEXCore::ToUnderlying(msize);
[[maybe_unused]] const auto msize_value = FEXCore::ToUnderlying(msize);
LOGMAN_THROW_A_FMT(op_data.scale == 0 || op_data.scale == msize_value, "scale may only be 0 or {}", msize_value);
@@ -4266,7 +4274,7 @@ private:
const auto msize_value = FEXCore::ToUnderlying(msize);
const auto msize_bytes = 1U << msize_value;
const auto imm_limit = (32U << msize_value) - msize_bytes;
[[maybe_unused]] const auto imm_limit = (32U << msize_value) - msize_bytes;
const auto imm = mem_op.MetaType.VectorImmType.Imm;
const auto imm_to_encode = imm >> msize_value;
@@ -4332,8 +4340,8 @@ private:
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_A_FMT((imm % num_regs) == 0, "Offset must be a multiple of {}", num_regs);
const auto min_offset = -8 * num_regs;
const auto max_offset = 7 * num_regs;
[[maybe_unused]] const auto min_offset = -8 * num_regs;
[[maybe_unused]] const auto max_offset = 7 * num_regs;
LOGMAN_THROW_A_FMT(imm >= min_offset && imm <= max_offset,
"Invalid load/store offset ({}). Offset must be a multiple of {} and be within [{}, {}]", imm, num_regs, min_offset,
max_offset);
@@ -4440,8 +4448,8 @@ private:
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
const auto esize = static_cast<int>(16 << ssz);
const auto max_imm = (esize << 3) - esize;
const auto min_imm = -(max_imm + esize);
[[maybe_unused]] const auto max_imm = (esize << 3) - esize;
[[maybe_unused]] const auto min_imm = -(max_imm + esize);
LOGMAN_THROW_A_FMT((imm % esize) == 0, "imm ({}) must be a multiple of {}", imm, esize);
LOGMAN_THROW_A_FMT(imm >= min_imm && imm <= max_imm, "imm ({}) must be within [{}, {}]", imm, min_imm, max_imm);
@@ -4485,7 +4493,7 @@ private:
const auto msize_value = FEXCore::ToUnderlying(msize);
const auto data_size_bytes = 1U << msize_value;
const auto max_imm = (64U << msize_value) - data_size_bytes;
[[maybe_unused]] const auto max_imm = (64U << msize_value) - data_size_bytes;
LOGMAN_THROW_A_FMT((imm % data_size_bytes) == 0 && imm <= max_imm, "imm must be a multiple of {} and be within [0, {}]",
data_size_bytes, max_imm);
@@ -4713,7 +4721,7 @@ private:
void SVEFloatUnary(uint32_t opc, SubRegSize size, PRegister pg, ZRegister zn, ZRegister zd) {
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "Unsupported size in {}", __func__);
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "Unsupported size in {}", __func__);
uint32_t Instr = 0b0110'0101'0000'1100'1010'0000'0000'0000;
Instr |= FEXCore::ToUnderlying(size) << 22;
@@ -4801,7 +4809,8 @@ private:
}
void SVEFPUnaryOpsUnpredicated(uint32_t opc, SubRegSize size, ZRegister zd, ZRegister zn) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
uint32_t Instr = 0b0110'0101'0000'1000'0011'0000'0000'0000;
Instr |= FEXCore::ToUnderlying(size) << 22;
@@ -4812,7 +4821,8 @@ private:
}
void SVEFPSerialReductionPredicated(uint32_t opc, SubRegSize size, VRegister vd, PRegister pg, VRegister vn, ZRegister zm) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_A_FMT(vd == vn, "vn must be the same as vd");
@@ -4826,7 +4836,8 @@ private:
}
void SVEFPCompareWithZero(uint32_t eqlt, uint32_t ne, SubRegSize size, PRegister pd, PRegister pg, ZRegister zn) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0101'0001'0000'0010'0000'0000'0000;
@@ -4841,7 +4852,8 @@ private:
void SVEFPMultiplyAdd(uint32_t opc, SubRegSize size, ZRegister zd, PRegister pg, ZRegister zn, ZRegister zm) {
// NOTE: opc also includes the op0 bit (bit 15) like op0:opc, since the fields are adjacent
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
uint32_t Instr = 0b0110'0101'0010'0000'0000'0000'0000'0000;
@@ -4855,13 +4867,14 @@ private:
}
void SVEFPMultiplyAddIndexed(uint32_t op, SubRegSize size, ZRegister zda, ZRegister zn, ZRegister zm, uint32_t index) {
LOGMAN_THROW_A_FMT(IsStandardFloatSize(size), "SubRegSize must be 16-bit, 32-bit, or 64-bit");
LOGMAN_THROW_A_FMT(size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit, "SubRegSize must be 16-bit, "
"32-bit, or 64-bit");
LOGMAN_THROW_A_FMT((size <= SubRegSize::i32Bit && zm <= ZReg::z7) || (size == SubRegSize::i64Bit && zm <= ZReg::z15),
"16-bit and 32-bit indexed variants may only use Zm between z0-z7\n"
"64-bit variants may only use Zm between z0-z15");
const auto Underlying = FEXCore::ToUnderlying(size);
const uint32_t IndexMax = (16 / (1U << Underlying)) - 1;
[[maybe_unused]] const uint32_t IndexMax = (16 / (1U << Underlying)) - 1;
LOGMAN_THROW_A_FMT(index <= IndexMax, "Index must be within 0-{}", IndexMax);
// Can be bit 20 or 19 depending on whether or not the element size is 64-bit.
@@ -5117,13 +5130,12 @@ private:
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
using FloatToEquivalentUInt = std::conditional_t<std::is_same_v<T, float>, uint32_t, uint64_t>;
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
// Determines if a floating-point value is capable of being converted
// into an 8-bit immediate. See pseudocode definition of VFPExpandImm
// in ARM A-profile reference manual for a general overview of how this was derived.
template<typename T>
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
[[nodiscard]]
[[nodiscard, maybe_unused]]
static bool IsValidFPValueForImm8(T value) {
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
@@ -5163,13 +5175,10 @@ private:
return true;
}
#endif
protected:
static uint32_t FP32ToImm8(float value) {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
#endif
const auto bits = FEXCore::BitCast<uint32_t>(value);
const auto sign = (bits & 0x80000000) >> 24;
@@ -5180,9 +5189,7 @@ protected:
}
static uint32_t FP64ToImm8(double value) {
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
#endif
const auto bits = FEXCore::BitCast<uint64_t>(value);
const auto sign = (bits & 0x80000000'00000000) >> 56;
@@ -5208,7 +5215,7 @@ private:
uint32_t shift = 0;
if (!is_int8_imm) {
const int32_t imm16_limit = 32768;
const bool is_int16_imm = -imm16_limit <= imm && imm < imm16_limit;
[[maybe_unused]] const bool is_int16_imm = -imm16_limit <= imm && imm < imm16_limit;
LOGMAN_THROW_A_FMT(is_int16_imm, "Immediate ({}) must be a 16-bit value within [-32768, 32512]", imm);
LOGMAN_THROW_A_FMT((imm % 256) == 0, "Immediate ({}) must be a multiple of 256", imm);
+9 -7
View File
@@ -27,19 +27,21 @@ struct EmitterOps : Emitter {
public:
// Advanced SIMD scalar copy
void dup(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Index) {
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'01 << 10;
const uint32_t SizeImm = FEXCore::ToUnderlying(size);
const uint32_t IndexShift = SizeImm + 1;
const uint32_t ElementSize = 1U << SizeImm;
const uint32_t MaxIndex = 128U / (ElementSize * 8);
[[maybe_unused]] const uint32_t MaxIndex = 128U / (ElementSize * 8);
LOGMAN_THROW_A_FMT(Index < MaxIndex, "Index too large. Index={}, Max Index: {}", Index, MaxIndex);
const uint32_t imm5 = (Index << IndexShift) | ElementSize;
ASIMDScalarCopy(1, 1, imm5, 0b0000, rd, rn);
ASIMDScalarCopy(Op, 1, imm5, 0b0000, rd, rn);
}
void mov(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Index) {
void mov(ARMEmitter::ScalarRegSize size, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn, uint32_t Index) {
dup(size, rd, rn, Index);
}
@@ -1280,10 +1282,10 @@ public:
private:
// Advanced SIMD scalar copy
void ASIMDScalarCopy(uint32_t Q, uint32_t b28, uint32_t imm5, uint32_t imm4, VRegister rd, VRegister rn) {
uint32_t Instr = 0b0000'1110'0000'0000'0000'01U << 10;
void ASIMDScalarCopy(uint32_t Op, uint32_t Q, uint32_t imm5, uint32_t imm4, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn) {
uint32_t Instr = Op;
Instr |= Q << 30;
Instr |= b28 << 28;
Instr |= imm5 << 16;
Instr |= imm4 << 11;
Instr |= Encode_rn(rn);
@@ -1381,7 +1383,7 @@ private:
void ASIMDScalarXIndexedElement(uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rm, VRegister rn, VRegister rd, uint32_t index) {
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i8Bit, "Scalar size must not be 8-bit");
const auto invalid_bound = 16U >> FEXCore::ToUnderlying(size);
[[maybe_unused]] const auto invalid_bound = 16U >> FEXCore::ToUnderlying(size);
LOGMAN_THROW_A_FMT(index < invalid_bound, "Index ({}) must be within [0-{}]", index, invalid_bound - 1);
uint32_t Instr = 0b0101'1111'0000'0000'0000'0000'0000'0000;
+59 -11
View File
@@ -41,6 +41,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
[[maybe_unused]] constexpr auto kDRegSize = 64;
constexpr auto kWRegSize = 32;
constexpr auto kXRegSize = 64;
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) || (width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
@@ -128,8 +129,8 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
// Compute the repeat distance d, and set up a bitmask covering the basic
// unit of repetition (i.e. a word with the bottom d bits set). Also, in all
// of these cases the N bit of the output will be zero.
clz_a = std::countl_zero(a);
int clz_c = std::countl_zero(c);
clz_a = CountLeadingZeros(a, kXRegSize);
int clz_c = CountLeadingZeros(c, kXRegSize);
d = clz_a - clz_c;
mask = ((UINT64_C(1) << d) - 1);
out_n = 0;
@@ -150,7 +151,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
// of set bits in our word, meaning that we have the trivial case of
// d == 64 and only one 'repetition'. Set up all the same variables as in
// the general case above, and set the N bit in the output.
clz_a = std::countl_zero(a);
clz_a = CountLeadingZeros(a, kXRegSize);
d = 64;
mask = ~UINT64_C(0);
out_n = 1;
@@ -158,7 +159,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
}
// If the repeat period d is not a power of two, it can't be encoded.
if (!std::has_single_bit(uint32_t(d))) {
if (!IsPowerOf2(d)) {
return false;
}
@@ -178,7 +179,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
static const uint64_t multipliers[] = {
0x0000000000000001UL, 0x0000000100000001UL, 0x0001000100010001UL, 0x0101010101010101UL, 0x1111111111111111UL, 0x5555555555555555UL,
};
uint64_t multiplier = multipliers[std::countl_zero(uint64_t(d)) - 57];
uint64_t multiplier = multipliers[CountLeadingZeros(d, kXRegSize) - 57];
uint64_t candidate = (b - a) * multiplier;
if (value != candidate) {
@@ -193,7 +194,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
// Count the set bits in our basic stretch. The special case of clz(0) == -1
// makes the answer come out right for stretches that reach the very top of
// the word (e.g. numbers like 0xffffc00000000000).
int clz_b = (b == 0) ? -1 : std::countl_zero(b);
int clz_b = (b == 0) ? -1 : CountLeadingZeros(b, kXRegSize);
int s = clz_a - clz_b;
// Decide how many bits to rotate right by, to put the low bit of that basic
@@ -223,13 +224,9 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
// 11110s 2 UInt(s)
//
// So we 'or' (2 * -d) with our computed s to form imms.
if (n != nullptr) {
if ((n != NULL) || (imm_s != NULL) || (imm_r != NULL)) {
*n = out_n;
}
if (imm_s != nullptr) {
*imm_s = ((2 * -d) | (s - 1)) & 0x3f;
}
if (imm_r != nullptr) {
*imm_r = r;
}
@@ -284,6 +281,11 @@ INT_1_TO_63_LIST(DECLARE_IS_UINT_N)
private:
template<typename V>
static inline bool IsPowerOf2(V value) {
return (value != 0) && ((value & (value - 1)) == 0);
}
// Some compilers dislike negating unsigned integers,
// so we provide an equivalent.
template<typename T>
@@ -296,4 +298,50 @@ static inline uint64_t LowestSetBit(uint64_t value) {
return value & UnsignedNegate(value);
}
template<typename V>
static inline int CountLeadingZeros(V value, int width = (sizeof(V) * 8)) {
#if COMPILER_HAS_BUILTIN_CLZ
if (width == 32) {
return (value == 0) ? 32 : __builtin_clz(static_cast<unsigned>(value));
} else if (width == 64) {
return (value == 0) ? 64 : __builtin_clzll(value);
}
#endif
return CountLeadingZerosFallBack(value, width);
}
static inline int CountLeadingZerosFallBack(uint64_t value, int width) {
LOGMAN_THROW_A_FMT(IsPowerOf2(width) && (width <= 64), "Invalid width");
if (value == 0) {
return width;
}
int count = 0;
value = value << (64 - width);
if ((value & UINT64_C(0xffffffff00000000)) == 0) {
count += 32;
value = value << 32;
}
if ((value & UINT64_C(0xffff000000000000)) == 0) {
count += 16;
value = value << 16;
}
if ((value & UINT64_C(0xff00000000000000)) == 0) {
count += 8;
value = value << 8;
}
if ((value & UINT64_C(0xf000000000000000)) == 0) {
count += 4;
value = value << 4;
}
if ((value & UINT64_C(0xc000000000000000)) == 0) {
count += 2;
value = value << 2;
}
if ((value & UINT64_C(0x8000000000000000)) == 0) {
count += 1;
}
count += (value == 0);
return count;
}
public:
+2 -4
View File
@@ -4,8 +4,7 @@ file(GLOB GEN_CONFIG_SOURCES CONFIGURE_DEPENDS *.json.in)
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
# Any configuration file json file that needs to be generated
@@ -22,6 +21,5 @@ foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
# Then install the configured json
install(
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
+3
View File
@@ -0,0 +1,3 @@
x86 and x86-64 Linux emulator
FEX allows you to run x86 applications on ARM64 Linux devices. It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
+18
View File
@@ -0,0 +1,18 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Setup binfmt_misc
update-binfmts --import FEX-x86
update-binfmts --import FEX-x86_64
}
# Install FEXInterpreter hardlink
# Needs to be done before setting up binfmt_misc
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
+17
View File
@@ -0,0 +1,17 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Uninstall
update-binfmts --unimport FEX-x86
update-binfmts --unimport FEX-x86_64
}
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
# Remove FEXInterpreter hardlink
unlink /usr/bin/FEXInterpreter
+1
View File
@@ -0,0 +1 @@
activate-noawait ldconfig
-1
View File
@@ -4,7 +4,6 @@ set(CMAKE_RC_COMPILER ${MINGW_TRIPLE}-windres)
set(CMAKE_C_COMPILER ${MINGW_TRIPLE}-clang)
set(CMAKE_CXX_COMPILER ${MINGW_TRIPLE}-clang++)
set(CMAKE_DLLTOOL ${MINGW_TRIPLE}-dlltool)
set(CMAKE_AR ${MINGW_TRIPLE}-ar)
# Compile everything as static to avoid requiring the MinGW runtime libraries, force page aligned sections so that
# debug symbols work correctly, and disable loop alignment to workaround an LLVM bug
+1 -1
View File
@@ -14,7 +14,7 @@ RUN mkdir build
ARG CC=clang-13
ARG CXX=clang++-13
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTING=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTS=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN ninja
WORKDIR /FEX/build
+2 -4
View File
@@ -10,8 +10,7 @@ function(GenBinFmt Name)
# Then install the configured binfmt
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/
COMPONENT Runtime)
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
endfunction()
if (NOT USE_LEGACY_BINFMTMISC)
@@ -20,8 +19,7 @@ if (NOT USE_LEGACY_BINFMTMISC)
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/
COMPONENT Runtime)
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
else()
GenBinFmt(FEX-x86.in)
GenBinFmt(FEX-x86_64.in)
+1 -1
View File
@@ -1 +1 @@
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+1 -1
View File
@@ -1 +1 @@
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+1 -1
View File
@@ -45,7 +45,7 @@ pkgs.mkShell {
fi
'';
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False
FEX_CMAKE_TOOLCHAIN_ARM64EC = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_CMAKE_TOOLCHAIN_WOW64 = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_MESON_CROSSFILE = "--cross-file ${mesonCrossFile}";
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -14,4 +14,4 @@ fi
rm -rf unittests/FEXLinuxTests
set -o xtrace
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTING=ON -DBUILD_FEX_LINUX_TESTS=ON
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTS=ON -DBUILD_FEX_LINUX_TESTS=ON
-1
View File
@@ -1 +0,0 @@
DisableFormat: true
+8 -3
View File
@@ -169,9 +169,14 @@ View the diff from {self.name} here.
class ClangFormatHelper(FormatHelper):
name = "git-clang-format"
name = "clang-format"
friendly_name = "C/C++ code formatter"
@property
def cformat_wrapper_path(self) -> str:
relpath = "../../Scripts/clang-format.py"
curpath = os.path.dirname(os.path.abspath(__file__))
return os.path.abspath(os.path.normpath(os.path.join(curpath, relpath)))
@property
def instructions(self) -> str:
@@ -194,7 +199,7 @@ class ClangFormatHelper(FormatHelper):
def clang_fmt_path(self) -> str:
if "CLANG_FORMAT_PATH" in os.environ:
return os.environ["CLANG_FORMAT_PATH"]
return "git-clang-format-19"
return "git-clang-format"
def has_tool(self) -> bool:
cmd = [self.clang_fmt_path, "-h"]
@@ -212,7 +217,7 @@ class ClangFormatHelper(FormatHelper):
cf_cmd = [
self.clang_fmt_path,
"--binary=clang-format-19",
f"--binary={self.cformat_wrapper_path}",
"--diff",
]
+31 -371
View File
@@ -1,392 +1,52 @@
#
# This file is autogenerated by pip-compile with Python 3.13
# This file is autogenerated by pip-compile with Python 3.11
# by the following command:
#
# pip-compile --generate-hashes --output-file=requirements_formatting.txt --strip-extras requirements_formatting.txt.in
# pip-compile --output-file=llvm/utils/git/requirements_formatting.txt llvm/utils/git/requirements_formatting.txt.in
#
black==25.1.0 \
--hash=sha256:030b9759066a4ee5e5aca28c3c77f9c64789cdd4de8ac1df642c40b708be6171 \
--hash=sha256:055e59b198df7ac0b7efca5ad7ff2516bca343276c466be72eb04a3bcc1f82d7 \
--hash=sha256:0e519ecf93120f34243e6b0054db49c00a35f84f195d5bce7e9f5cfc578fc2da \
--hash=sha256:172b1dbff09f86ce6f4eb8edf9dede08b1fce58ba194c87d7a4f1a5aa2f5b3c2 \
--hash=sha256:1e2978f6df243b155ef5fa7e558a43037c3079093ed5d10fd84c43900f2d8ecc \
--hash=sha256:33496d5cd1222ad73391352b4ae8da15253c5de89b93a80b3e2c8d9a19ec2666 \
--hash=sha256:3b48735872ec535027d979e8dcb20bf4f70b5ac75a8ea99f127c106a7d7aba9f \
--hash=sha256:4b60580e829091e6f9238c848ea6750efed72140b91b048770b64e74fe04908b \
--hash=sha256:759e7ec1e050a15f89b770cefbf91ebee8917aac5c20483bc2d80a6c3a04df32 \
--hash=sha256:8f0b18a02996a836cc9c9c78e5babec10930862827b1b724ddfe98ccf2f2fe4f \
--hash=sha256:95e8176dae143ba9097f351d174fdaf0ccd29efb414b362ae3fd72bf0f710717 \
--hash=sha256:96c1c7cd856bba8e20094e36e0f948718dc688dba4a9d78c3adde52b9e6c2299 \
--hash=sha256:a1ee0a0c330f7b5130ce0caed9936a904793576ef4d2b98c40835d6a65afa6a0 \
--hash=sha256:a22f402b410566e2d1c950708c77ebf5ebd5d0d88a6a2e87c86d9fb48afa0d18 \
--hash=sha256:a39337598244de4bae26475f77dda852ea00a93bd4c728e09eacd827ec929df0 \
--hash=sha256:afebb7098bfbc70037a053b91ae8437c3857482d3a690fefc03e9ff7aa9a5fd3 \
--hash=sha256:bacabb307dca5ebaf9c118d2d2f6903da0d62c9faa82bd21a33eecc319559355 \
--hash=sha256:bce2e264d59c91e52d8000d507eb20a9aca4a778731a08cfff7e5ac4a4bb7096 \
--hash=sha256:d9e6827d563a2c820772b32ce8a42828dc6790f095f441beef18f96aa6f8294e \
--hash=sha256:db8ea9917d6f8fc62abd90d944920d95e73c83a5ee3383493e35d271aca872e9 \
--hash=sha256:ea0213189960bda9cf99be5b8c8ce66bb054af5e9e861249cd23471bd7b0b3ba \
--hash=sha256:f3df5f1bf91d36002b0a75389ca8663510cf0531cca8aa5c1ef695b46d98655f
black==23.9.1
# via
# -r requirements_formatting.txt.in
# -r llvm/utils/git/requirements_formatting.txt.in
# darker
certifi==2025.7.14 \
--hash=sha256:6b31f564a415d79ee77df69d757bb49a5bb53bd9f756cbbe24394ffd6fc1f4b2 \
--hash=sha256:8ea99dbdfaaf2ba2f9bac77b9249ef62ec5218e7c2b2e903378ed5fccf765995
# via
# -r requirements_formatting.txt.in
# requests
cffi==1.15.1 \
--hash=sha256:00a9ed42e88df81ffae7a8ab6d9356b371399b91dbdf0c3cb1e84c03a13aceb5 \
--hash=sha256:03425bdae262c76aad70202debd780501fabeaca237cdfddc008987c0e0f59ef \
--hash=sha256:04ed324bda3cda42b9b695d51bb7d54b680b9719cfab04227cdd1e04e5de3104 \
--hash=sha256:0e2642fe3142e4cc4af0799748233ad6da94c62a8bec3a6648bf8ee68b1c7426 \
--hash=sha256:173379135477dc8cac4bc58f45db08ab45d228b3363adb7af79436135d028405 \
--hash=sha256:198caafb44239b60e252492445da556afafc7d1e3ab7a1fb3f0584ef6d742375 \
--hash=sha256:1e74c6b51a9ed6589199c787bf5f9875612ca4a8a0785fb2d4a84429badaf22a \
--hash=sha256:2012c72d854c2d03e45d06ae57f40d78e5770d252f195b93f581acf3ba44496e \
--hash=sha256:21157295583fe8943475029ed5abdcf71eb3911894724e360acff1d61c1d54bc \
--hash=sha256:2470043b93ff09bf8fb1d46d1cb756ce6132c54826661a32d4e4d132e1977adf \
--hash=sha256:285d29981935eb726a4399badae8f0ffdff4f5050eaa6d0cfc3f64b857b77185 \
--hash=sha256:30d78fbc8ebf9c92c9b7823ee18eb92f2e6ef79b45ac84db507f52fbe3ec4497 \
--hash=sha256:320dab6e7cb2eacdf0e658569d2575c4dad258c0fcc794f46215e1e39f90f2c3 \
--hash=sha256:33ab79603146aace82c2427da5ca6e58f2b3f2fb5da893ceac0c42218a40be35 \
--hash=sha256:3548db281cd7d2561c9ad9984681c95f7b0e38881201e157833a2342c30d5e8c \
--hash=sha256:3799aecf2e17cf585d977b780ce79ff0dc9b78d799fc694221ce814c2c19db83 \
--hash=sha256:39d39875251ca8f612b6f33e6b1195af86d1b3e60086068be9cc053aa4376e21 \
--hash=sha256:3b926aa83d1edb5aa5b427b4053dc420ec295a08e40911296b9eb1b6170f6cca \
--hash=sha256:3bcde07039e586f91b45c88f8583ea7cf7a0770df3a1649627bf598332cb6984 \
--hash=sha256:3d08afd128ddaa624a48cf2b859afef385b720bb4b43df214f85616922e6a5ac \
--hash=sha256:3eb6971dcff08619f8d91607cfc726518b6fa2a9eba42856be181c6d0d9515fd \
--hash=sha256:40f4774f5a9d4f5e344f31a32b5096977b5d48560c5592e2f3d2c4374bd543ee \
--hash=sha256:4289fc34b2f5316fbb762d75362931e351941fa95fa18789191b33fc4cf9504a \
--hash=sha256:470c103ae716238bbe698d67ad020e1db9d9dba34fa5a899b5e21577e6d52ed2 \
--hash=sha256:4f2c9f67e9821cad2e5f480bc8d83b8742896f1242dba247911072d4fa94c192 \
--hash=sha256:50a74364d85fd319352182ef59c5c790484a336f6db772c1a9231f1c3ed0cbd7 \
--hash=sha256:54a2db7b78338edd780e7ef7f9f6c442500fb0d41a5a4ea24fff1c929d5af585 \
--hash=sha256:5635bd9cb9731e6d4a1132a498dd34f764034a8ce60cef4f5319c0541159392f \
--hash=sha256:59c0b02d0a6c384d453fece7566d1c7e6b7bae4fc5874ef2ef46d56776d61c9e \
--hash=sha256:5d598b938678ebf3c67377cdd45e09d431369c3b1a5b331058c338e201f12b27 \
--hash=sha256:5df2768244d19ab7f60546d0c7c63ce1581f7af8b5de3eb3004b9b6fc8a9f84b \
--hash=sha256:5ef34d190326c3b1f822a5b7a45f6c4535e2f47ed06fec77d3d799c450b2651e \
--hash=sha256:6975a3fac6bc83c4a65c9f9fcab9e47019a11d3d2cf7f3c0d03431bf145a941e \
--hash=sha256:6c9a799e985904922a4d207a94eae35c78ebae90e128f0c4e521ce339396be9d \
--hash=sha256:70df4e3b545a17496c9b3f41f5115e69a4f2e77e94e1d2a8e1070bc0c38c8a3c \
--hash=sha256:7473e861101c9e72452f9bf8acb984947aa1661a7704553a9f6e4baa5ba64415 \
--hash=sha256:8102eaf27e1e448db915d08afa8b41d6c7ca7a04b7d73af6514df10a3e74bd82 \
--hash=sha256:87c450779d0914f2861b8526e035c5e6da0a3199d8f1add1a665e1cbc6fc6d02 \
--hash=sha256:8b7ee99e510d7b66cdb6c593f21c043c248537a32e0bedf02e01e9553a172314 \
--hash=sha256:91fc98adde3d7881af9b59ed0294046f3806221863722ba7d8d120c575314325 \
--hash=sha256:94411f22c3985acaec6f83c6df553f2dbe17b698cc7f8ae751ff2237d96b9e3c \
--hash=sha256:98d85c6a2bef81588d9227dde12db8a7f47f639f4a17c9ae08e773aa9c697bf3 \
--hash=sha256:9ad5db27f9cabae298d151c85cf2bad1d359a1b9c686a275df03385758e2f914 \
--hash=sha256:a0b71b1b8fbf2b96e41c4d990244165e2c9be83d54962a9a1d118fd8657d2045 \
--hash=sha256:a0f100c8912c114ff53e1202d0078b425bee3649ae34d7b070e9697f93c5d52d \
--hash=sha256:a591fe9e525846e4d154205572a029f653ada1a78b93697f3b5a8f1f2bc055b9 \
--hash=sha256:a5c84c68147988265e60416b57fc83425a78058853509c1b0629c180094904a5 \
--hash=sha256:a66d3508133af6e8548451b25058d5812812ec3798c886bf38ed24a98216fab2 \
--hash=sha256:a8c4917bd7ad33e8eb21e9a5bbba979b49d9a97acb3a803092cbc1133e20343c \
--hash=sha256:b3bbeb01c2b273cca1e1e0c5df57f12dce9a4dd331b4fa1635b8bec26350bde3 \
--hash=sha256:cba9d6b9a7d64d4bd46167096fc9d2f835e25d7e4c121fb2ddfc6528fb0413b2 \
--hash=sha256:cc4d65aeeaa04136a12677d3dd0b1c0c94dc43abac5860ab33cceb42b801c1e8 \
--hash=sha256:ce4bcc037df4fc5e3d184794f27bdaab018943698f4ca31630bc7f84a7b69c6d \
--hash=sha256:cec7d9412a9102bdc577382c3929b337320c4c4c4849f2c5cdd14d7368c5562d \
--hash=sha256:d400bfb9a37b1351253cb402671cea7e89bdecc294e8016a707f6d1d8ac934f9 \
--hash=sha256:d61f4695e6c866a23a21acab0509af1cdfd2c013cf256bbf5b6b5e2695827162 \
--hash=sha256:db0fbb9c62743ce59a9ff687eb5f4afbe77e5e8403d6697f7446e5f609976f76 \
--hash=sha256:dd86c085fae2efd48ac91dd7ccffcfc0571387fe1193d33b6394db7ef31fe2a4 \
--hash=sha256:e00b098126fd45523dd056d2efba6c5a63b71ffe9f2bbe1a4fe1716e1d0c331e \
--hash=sha256:e229a521186c75c8ad9490854fd8bbdd9a0c9aa3a524326b55be83b54d4e0ad9 \
--hash=sha256:e263d77ee3dd201c3a142934a086a4450861778baaeeb45db4591ef65550b0a6 \
--hash=sha256:ed9cb427ba5504c1dc15ede7d516b84757c3e3d7868ccc85121d9310d27eed0b \
--hash=sha256:fa6693661a4c91757f4412306191b6dc88c1703f780c8234035eac011922bc01 \
--hash=sha256:fcd131dd944808b5bdb38e6f5b53013c5aa4f334c5cad0c72742f6eba4b73db0
certifi==2023.7.22
# via requests
cffi==1.15.1
# via
# cryptography
# pynacl
charset-normalizer==3.2.0 \
--hash=sha256:04e57ab9fbf9607b77f7d057974694b4f6b142da9ed4a199859d9d4d5c63fe96 \
--hash=sha256:09393e1b2a9461950b1c9a45d5fd251dc7c6f228acab64da1c9c0165d9c7765c \
--hash=sha256:0b87549028f680ca955556e3bd57013ab47474c3124dc069faa0b6545b6c9710 \
--hash=sha256:1000fba1057b92a65daec275aec30586c3de2401ccdcd41f8a5c1e2c87078706 \
--hash=sha256:1249cbbf3d3b04902ff081ffbb33ce3377fa6e4c7356f759f3cd076cc138d020 \
--hash=sha256:1920d4ff15ce893210c1f0c0e9d19bfbecb7983c76b33f046c13a8ffbd570252 \
--hash=sha256:193cbc708ea3aca45e7221ae58f0fd63f933753a9bfb498a3b474878f12caaad \
--hash=sha256:1a100c6d595a7f316f1b6f01d20815d916e75ff98c27a01ae817439ea7726329 \
--hash=sha256:1f30b48dd7fa1474554b0b0f3fdfdd4c13b5c737a3c6284d3cdc424ec0ffff3a \
--hash=sha256:203f0c8871d5a7987be20c72442488a0b8cfd0f43b7973771640fc593f56321f \
--hash=sha256:246de67b99b6851627d945db38147d1b209a899311b1305dd84916f2b88526c6 \
--hash=sha256:2dee8e57f052ef5353cf608e0b4c871aee320dd1b87d351c28764fc0ca55f9f4 \
--hash=sha256:2efb1bd13885392adfda4614c33d3b68dee4921fd0ac1d3988f8cbb7d589e72a \
--hash=sha256:2f4ac36d8e2b4cc1aa71df3dd84ff8efbe3bfb97ac41242fbcfc053c67434f46 \
--hash=sha256:3170c9399da12c9dc66366e9d14da8bf7147e1e9d9ea566067bbce7bb74bd9c2 \
--hash=sha256:3b1613dd5aee995ec6d4c69f00378bbd07614702a315a2cf6c1d21461fe17c23 \
--hash=sha256:3bb3d25a8e6c0aedd251753a79ae98a093c7e7b471faa3aa9a93a81431987ace \
--hash=sha256:3bb7fda7260735efe66d5107fb7e6af6a7c04c7fce9b2514e04b7a74b06bf5dd \
--hash=sha256:41b25eaa7d15909cf3ac4c96088c1f266a9a93ec44f87f1d13d4a0e86c81b982 \
--hash=sha256:45de3f87179c1823e6d9e32156fb14c1927fcc9aba21433f088fdfb555b77c10 \
--hash=sha256:46fb8c61d794b78ec7134a715a3e564aafc8f6b5e338417cb19fe9f57a5a9bf2 \
--hash=sha256:48021783bdf96e3d6de03a6e39a1171ed5bd7e8bb93fc84cc649d11490f87cea \
--hash=sha256:4957669ef390f0e6719db3613ab3a7631e68424604a7b448f079bee145da6e09 \
--hash=sha256:5e86d77b090dbddbe78867a0275cb4df08ea195e660f1f7f13435a4649e954e5 \
--hash=sha256:6339d047dab2780cc6220f46306628e04d9750f02f983ddb37439ca47ced7149 \
--hash=sha256:681eb3d7e02e3c3655d1b16059fbfb605ac464c834a0c629048a30fad2b27489 \
--hash=sha256:6c409c0deba34f147f77efaa67b8e4bb83d2f11c8806405f76397ae5b8c0d1c9 \
--hash=sha256:7095f6fbfaa55defb6b733cfeb14efaae7a29f0b59d8cf213be4e7ca0b857b80 \
--hash=sha256:70c610f6cbe4b9fce272c407dd9d07e33e6bf7b4aa1b7ffb6f6ded8e634e3592 \
--hash=sha256:72814c01533f51d68702802d74f77ea026b5ec52793c791e2da806a3844a46c3 \
--hash=sha256:7a4826ad2bd6b07ca615c74ab91f32f6c96d08f6fcc3902ceeedaec8cdc3bcd6 \
--hash=sha256:7c70087bfee18a42b4040bb9ec1ca15a08242cf5867c58726530bdf3945672ed \
--hash=sha256:855eafa5d5a2034b4621c74925d89c5efef61418570e5ef9b37717d9c796419c \
--hash=sha256:8700f06d0ce6f128de3ccdbc1acaea1ee264d2caa9ca05daaf492fde7c2a7200 \
--hash=sha256:89f1b185a01fe560bc8ae5f619e924407efca2191b56ce749ec84982fc59a32a \
--hash=sha256:8b2c760cfc7042b27ebdb4a43a4453bd829a5742503599144d54a032c5dc7e9e \
--hash=sha256:8c2f5e83493748286002f9369f3e6607c565a6a90425a3a1fef5ae32a36d749d \
--hash=sha256:8e098148dd37b4ce3baca71fb394c81dc5d9c7728c95df695d2dca218edf40e6 \
--hash=sha256:94aea8eff76ee6d1cdacb07dd2123a68283cb5569e0250feab1240058f53b623 \
--hash=sha256:95eb302ff792e12aba9a8b8f8474ab229a83c103d74a750ec0bd1c1eea32e669 \
--hash=sha256:9bd9b3b31adcb054116447ea22caa61a285d92e94d710aa5ec97992ff5eb7cf3 \
--hash=sha256:9e608aafdb55eb9f255034709e20d5a83b6d60c054df0802fa9c9883d0a937aa \
--hash=sha256:a103b3a7069b62f5d4890ae1b8f0597618f628b286b03d4bc9195230b154bfa9 \
--hash=sha256:a386ebe437176aab38c041de1260cd3ea459c6ce5263594399880bbc398225b2 \
--hash=sha256:a38856a971c602f98472050165cea2cdc97709240373041b69030be15047691f \
--hash=sha256:a401b4598e5d3f4a9a811f3daf42ee2291790c7f9d74b18d75d6e21dda98a1a1 \
--hash=sha256:a7647ebdfb9682b7bb97e2a5e7cb6ae735b1c25008a70b906aecca294ee96cf4 \
--hash=sha256:aaf63899c94de41fe3cf934601b0f7ccb6b428c6e4eeb80da72c58eab077b19a \
--hash=sha256:b0dac0ff919ba34d4df1b6131f59ce95b08b9065233446be7e459f95554c0dc8 \
--hash=sha256:baacc6aee0b2ef6f3d308e197b5d7a81c0e70b06beae1f1fcacffdbd124fe0e3 \
--hash=sha256:bf420121d4c8dce6b889f0e8e4ec0ca34b7f40186203f06a946fa0276ba54029 \
--hash=sha256:c04a46716adde8d927adb9457bbe39cf473e1e2c2f5d0a16ceb837e5d841ad4f \
--hash=sha256:c0b21078a4b56965e2b12f247467b234734491897e99c1d51cee628da9786959 \
--hash=sha256:c1c76a1743432b4b60ab3358c937a3fe1341c828ae6194108a94c69028247f22 \
--hash=sha256:c4983bf937209c57240cff65906b18bb35e64ae872da6a0db937d7b4af845dd7 \
--hash=sha256:c4fb39a81950ec280984b3a44f5bd12819953dc5fa3a7e6fa7a80db5ee853952 \
--hash=sha256:c57921cda3a80d0f2b8aec7e25c8aa14479ea92b5b51b6876d975d925a2ea346 \
--hash=sha256:c8063cf17b19661471ecbdb3df1c84f24ad2e389e326ccaf89e3fb2484d8dd7e \
--hash=sha256:ccd16eb18a849fd8dcb23e23380e2f0a354e8daa0c984b8a732d9cfaba3a776d \
--hash=sha256:cd6dbe0238f7743d0efe563ab46294f54f9bc8f4b9bcf57c3c666cc5bc9d1299 \
--hash=sha256:d62e51710986674142526ab9f78663ca2b0726066ae26b78b22e0f5e571238dd \
--hash=sha256:db901e2ac34c931d73054d9797383d0f8009991e723dab15109740a63e7f902a \
--hash=sha256:e03b8895a6990c9ab2cdcd0f2fe44088ca1c65ae592b8f795c3294af00a461c3 \
--hash=sha256:e1c8a2f4c69e08e89632defbfabec2feb8a8d99edc9f89ce33c4b9e36ab63037 \
--hash=sha256:e4b749b9cc6ee664a3300bb3a273c1ca8068c46be705b6c31cf5d276f8628a94 \
--hash=sha256:e6a5bf2cba5ae1bb80b154ed68a3cfa2fa00fde979a7f50d6598d3e17d9ac20c \
--hash=sha256:e857a2232ba53ae940d3456f7533ce6ca98b81917d47adc3c7fd55dad8fab858 \
--hash=sha256:ee4006268ed33370957f55bf2e6f4d263eaf4dc3cfc473d1d90baff6ed36ce4a \
--hash=sha256:eef9df1eefada2c09a5e7a40991b9fc6ac6ef20b1372abd48d2794a316dc0449 \
--hash=sha256:f058f6963fd82eb143c692cecdc89e075fa0828db2e5b291070485390b2f1c9c \
--hash=sha256:f25c229a6ba38a35ae6e25ca1264621cc25d4d38dca2942a7fce0b67a4efe918 \
--hash=sha256:f2a1d0fd4242bd8643ce6f98927cf9c04540af6efa92323e9d3124f57727bfc1 \
--hash=sha256:f7560358a6811e52e9c4d142d497f1a6e10103d3a6881f18d04dbce3729c0e2c \
--hash=sha256:f779d3ad205f108d14e99bb3859aa7dd8e9c68874617c72354d7ecaec2a054ac \
--hash=sha256:f87f746ee241d30d6ed93969de31e5ffd09a2961a051e60ae6bddde9ec3583aa
charset-normalizer==3.2.0
# via requests
click==8.1.7 \
--hash=sha256:ae74fb96c20a0277a1d615f1e4d73c8414f5a98db8b799a7931d1582f3390c28 \
--hash=sha256:ca9853ad459e787e2192211578cc907e7594e294c7ccc834310722b41b9ca6de
click==8.1.7
# via black
cryptography==45.0.5 \
--hash=sha256:0027d566d65a38497bc37e0dd7c2f8ceda73597d2ac9ba93810204f56f52ebc7 \
--hash=sha256:101ee65078f6dd3e5a028d4f19c07ffa4dd22cce6a20eaa160f8b5219911e7d8 \
--hash=sha256:12e55281d993a793b0e883066f590c1ae1e802e3acb67f8b442e721e475e6463 \
--hash=sha256:14d96584701a887763384f3c47f0ca7c1cce322aa1c31172680eb596b890ec30 \
--hash=sha256:1e1da5accc0c750056c556a93c3e9cb828970206c68867712ca5805e46dc806f \
--hash=sha256:206210d03c1193f4e1ff681d22885181d47efa1ab3018766a7b32a7b3d6e6afd \
--hash=sha256:2089cc8f70a6e454601525e5bf2779e665d7865af002a5dec8d14e561002e135 \
--hash=sha256:3a264aae5f7fbb089dbc01e0242d3b67dffe3e6292e1f5182122bdf58e65215d \
--hash=sha256:3af26738f2db354aafe492fb3869e955b12b2ef2e16908c8b9cb928128d42c57 \
--hash=sha256:3fcfbefc4a7f332dece7272a88e410f611e79458fab97b5efe14e54fe476f4fd \
--hash=sha256:460f8c39ba66af7db0545a8c6f2eabcbc5a5528fc1cf6c3fa9a1e44cec33385e \
--hash=sha256:57c816dfbd1659a367831baca4b775b2a5b43c003daf52e9d57e1d30bc2e1b0e \
--hash=sha256:5aa1e32983d4443e310f726ee4b071ab7569f58eedfdd65e9675484a4eb67bd1 \
--hash=sha256:6ff8728d8d890b3dda5765276d1bc6fb099252915a2cd3aff960c4c195745dd0 \
--hash=sha256:7259038202a47fdecee7e62e0fd0b0738b6daa335354396c6ddebdbe1206af2a \
--hash=sha256:72e76caa004ab63accdf26023fccd1d087f6d90ec6048ff33ad0445abf7f605a \
--hash=sha256:7760c1c2e1a7084153a0f68fab76e754083b126a47d0117c9ed15e69e2103492 \
--hash=sha256:8c4a6ff8a30e9e3d38ac0539e9a9e02540ab3f827a3394f8852432f6b0ea152e \
--hash=sha256:9024beb59aca9d31d36fcdc1604dd9bbeed0a55bface9f1908df19178e2f116e \
--hash=sha256:90cb0a7bb35959f37e23303b7eed0a32280510030daba3f7fdfbb65defde6a97 \
--hash=sha256:91098f02ca81579c85f66df8a588c78f331ca19089763d733e34ad359f474174 \
--hash=sha256:926c3ea71a6043921050eaa639137e13dbe7b4ab25800932a8498364fc1abec9 \
--hash=sha256:982518cd64c54fcada9d7e5cf28eabd3ee76bd03ab18e08a48cad7e8b6f31b18 \
--hash=sha256:9b4cf6318915dccfe218e69bbec417fdd7c7185aa7aab139a2c0beb7468c89f0 \
--hash=sha256:ad0caded895a00261a5b4aa9af828baede54638754b51955a0ac75576b831b27 \
--hash=sha256:b85980d1e345fe769cfc57c57db2b59cff5464ee0c045d52c0df087e926fbe63 \
--hash=sha256:b8fa8b0a35a9982a3c60ec79905ba5bb090fc0b9addcfd3dc2dd04267e45f25e \
--hash=sha256:b9e38e0a83cd51e07f5a48ff9691cae95a79bea28fe4ded168a8e5c6c77e819d \
--hash=sha256:bd4c45986472694e5121084c6ebbd112aa919a25e783b87eb95953c9573906d6 \
--hash=sha256:be97d3a19c16a9be00edf79dca949c8fa7eff621763666a145f9f9535a5d7f42 \
--hash=sha256:c648025b6840fe62e57107e0a25f604db740e728bd67da4f6f060f03017d5097 \
--hash=sha256:d05a38884db2ba215218745f0781775806bde4f32e07b135348355fe8e4991d9 \
--hash=sha256:dd420e577921c8c2d31289536c386aaa30140b473835e97f83bc71ea9d2baf2d \
--hash=sha256:e357286c1b76403dd384d938f93c46b2b058ed4dfcdce64a770f0537ed3feb6f \
--hash=sha256:e6c00130ed423201c5bc5544c23359141660b07999ad82e34e7bb8f882bb78e0 \
--hash=sha256:e74d30ec9c7cb2f404af331d5b4099a9b322a8a6b25c4632755c8757345baac5 \
--hash=sha256:f3562c2f23c612f2e4a6964a61d942f891d29ee320edb62ff48ffb99f3de9ae8
# via
# -r requirements_formatting.txt.in
# pyjwt
darker==2.1.1 \
--hash=sha256:a6e6a682c0604e76fe9aec7650e96a944f517563c69b28fcc076db9d957d98ea \
--hash=sha256:ead701414c45359fc0312bc285614d3285fc135476d43f3bc08d989ee19d9020
# via -r requirements_formatting.txt.in
darkgraylib==1.2.1 \
--hash=sha256:60c59de69842367ce0c78c32c451fa8e9d29500e681312d9864a7416bcdb7792 \
--hash=sha256:a5dd6a2015a470d9047278cdd01a91ccb1d746675f8fd4562b3b5f6b8cbda930
# via
# darker
# graylint
deprecated==1.2.14 \
--hash=sha256:6fac8b097794a90302bdbb17b9b815e732d3c4720583ff1b198499d78470466c \
--hash=sha256:e5323eb936458dccc2582dc6f9c322c852a775a27065ff2b0c4970b9d53d01b3
cryptography==41.0.3
# via pyjwt
darker==1.7.2
# via -r llvm/utils/git/requirements_formatting.txt.in
deprecated==1.2.14
# via pygithub
graylint==1.1.1 \
--hash=sha256:0fd8e02972ca03d0ef2bf0adea76b5343efcd492d7afb5f658f3e3a724f55a36 \
--hash=sha256:b7e0eab6c159684dbf5ef84e942c3340f6a6549b02a3d11b1a1763cc4f8f0593
# via darker
idna==3.10 \
--hash=sha256:12f65c9b470abda6dc35cf8e63cc574b1c52b11df2c86030af0ac09b01b13ea9 \
--hash=sha256:946d195a0d259cbba61165e88e65941f16e9b36ea6ddb97f00452bae8b1287d3
# via
# -r requirements_formatting.txt.in
# requests
mypy-extensions==1.0.0 \
--hash=sha256:4392f6c0eb8a5668a69e23d168ffa70f0be9ccfd32b5cc2d26a34ae5b844552d \
--hash=sha256:75dbf8955dc00442a438fc4d0666508a9a97b6bd41aa2f0ffe9d2f2725af0782
idna==3.4
# via requests
mypy-extensions==1.0.0
# via black
packaging==23.1 \
--hash=sha256:994793af429502c4ea2ebf6bf664629d07c1a9fe974af92966e4b8d2df7edc61 \
--hash=sha256:a392980d2b6cffa644431898be54b0045151319d1e7ec34f0cfed48767dd334f
packaging==23.1
# via black
pathspec==0.11.2 \
--hash=sha256:1d6ed233af05e679efb96b1851550ea95bbb64b7c490b0f5aa52996c11e92a20 \
--hash=sha256:e0d8d0ac2f12da61956eb2306b69f9469b42f4deb0f3cb6ed47b9cce9996ced3
pathspec==0.11.2
# via black
platformdirs==3.10.0 \
--hash=sha256:b45696dab2d7cc691a3226759c0d3b00c47c8b6e293d96f6436f733303f77f6d \
--hash=sha256:d7c24979f292f916dc9cbf8648319032f551ea8c49a4c9bf2fb556a02070ec1d
platformdirs==3.10.0
# via black
pycparser==2.21 \
--hash=sha256:8ee45429555515e1f6b185e78100aea234072576aa43ab53aefcae078162fca9 \
--hash=sha256:e644fdec12f7872f86c58ff790da456218b10f863970249516d60a5eaca77206
pycparser==2.21
# via cffi
pygithub==2.6.1 \
--hash=sha256:6f2fa6d076ccae475f9fc392cc6cdbd54db985d4f69b8833a28397de75ed6ca3 \
--hash=sha256:b5c035392991cca63959e9453286b41b54d83bf2de2daa7d7ff7e4312cebf3bf
# via -r requirements_formatting.txt.in
pyjwt==2.8.0 \
--hash=sha256:57e28d156e3d5c10088e0c68abb90bfac3df82b40a71bd0daa20c65ccd5c23de \
--hash=sha256:59127c392cc44c2da5bb3192169a91f429924e17aff6534d70fdc02ab3e04320
pygithub==1.59.1
# via -r llvm/utils/git/requirements_formatting.txt.in
pyjwt[crypto]==2.8.0
# via pygithub
pynacl==1.5.0 \
--hash=sha256:06b8f6fa7f5de8d5d2f7573fe8c863c051225a27b61e6860fd047b1775807858 \
--hash=sha256:0c84947a22519e013607c9be43706dd42513f9e6ae5d39d3613ca1e142fba44d \
--hash=sha256:20f42270d27e1b6a29f54032090b972d97f0a1b0948cc52392041ef7831fee93 \
--hash=sha256:401002a4aaa07c9414132aaed7f6836ff98f59277a234704ff66878c2ee4a0d1 \
--hash=sha256:52cb72a79269189d4e0dc537556f4740f7f0a9ec41c1322598799b0bdad4ef92 \
--hash=sha256:61f642bf2378713e2c2e1de73444a3778e5f0a38be6fee0fe532fe30060282ff \
--hash=sha256:8ac7448f09ab85811607bdd21ec2464495ac8b7c66d146bf545b0f08fb9220ba \
--hash=sha256:a36d4a9dda1f19ce6e03c9a784a2921a4b726b02e1c736600ca9c22029474394 \
--hash=sha256:a422368fc821589c228f4c49438a368831cb5bbc0eab5ebe1d7fac9dded6567b \
--hash=sha256:e46dae94e34b085175f8abb3b0aaa7da40767865ac82c928eeb9e57e1ea8a543
pynacl==1.5.0
# via pygithub
requests==2.32.4 \
--hash=sha256:27babd3cda2a6d50b30443204ee89830707d396671944c998b5975b031ac2b2c \
--hash=sha256:27d0316682c8a29834d3264820024b62a36942083d52caf2f14c0591336d3422
# via
# -r requirements_formatting.txt.in
# pygithub
toml==0.10.2 \
--hash=sha256:806143ae5bfb6a3c6e736a764057db0e6a0e05e338b5630894a5f779cabb4f9b \
--hash=sha256:b3bda1d108d5dd99f4a20d24d9c348e91c4db7ab1b749200bded2f839ccbe68f
# via
# darker
# darkgraylib
typing-extensions==4.14.1 \
--hash=sha256:38b39f4aeeab64884ce9f74c94263ef78f3c22467c8724005483154c26648d36 \
--hash=sha256:d1e1e3b58374dc93031d6eda2420a48ea44a36c2b4766a4fdeb3710755731d76
requests==2.31.0
# via pygithub
urllib3==2.5.0 \
--hash=sha256:3fc47733c7e419d4bc3f6b3dc2b4f890bb743906a30d56ba4a5bfa4bbff92760 \
--hash=sha256:e6b01673c0fa6a13e374b50871808eb3bf7046c4b125b216f6bf1cc604cff0dc
# via
# -r requirements_formatting.txt.in
# pygithub
# requests
wrapt==1.15.0 \
--hash=sha256:02fce1852f755f44f95af51f69d22e45080102e9d00258053b79367d07af39c0 \
--hash=sha256:077ff0d1f9d9e4ce6476c1a924a3332452c1406e59d90a2cf24aeb29eeac9420 \
--hash=sha256:078e2a1a86544e644a68422f881c48b84fef6d18f8c7a957ffd3f2e0a74a0d4a \
--hash=sha256:0970ddb69bba00670e58955f8019bec4a42d1785db3faa043c33d81de2bf843c \
--hash=sha256:1286eb30261894e4c70d124d44b7fd07825340869945c79d05bda53a40caa079 \
--hash=sha256:21f6d9a0d5b3a207cdf7acf8e58d7d13d463e639f0c7e01d82cdb671e6cb7923 \
--hash=sha256:230ae493696a371f1dbffaad3dafbb742a4d27a0afd2b1aecebe52b740167e7f \
--hash=sha256:26458da5653aa5b3d8dc8b24192f574a58984c749401f98fff994d41d3f08da1 \
--hash=sha256:2cf56d0e237280baed46f0b5316661da892565ff58309d4d2ed7dba763d984b8 \
--hash=sha256:2e51de54d4fb8fb50d6ee8327f9828306a959ae394d3e01a1ba8b2f937747d86 \
--hash=sha256:2fbfbca668dd15b744418265a9607baa970c347eefd0db6a518aaf0cfbd153c0 \
--hash=sha256:38adf7198f8f154502883242f9fe7333ab05a5b02de7d83aa2d88ea621f13364 \
--hash=sha256:3a8564f283394634a7a7054b7983e47dbf39c07712d7b177b37e03f2467a024e \
--hash=sha256:3abbe948c3cbde2689370a262a8d04e32ec2dd4f27103669a45c6929bcdbfe7c \
--hash=sha256:3bbe623731d03b186b3d6b0d6f51865bf598587c38d6f7b0be2e27414f7f214e \
--hash=sha256:40737a081d7497efea35ab9304b829b857f21558acfc7b3272f908d33b0d9d4c \
--hash=sha256:41d07d029dd4157ae27beab04d22b8e261eddfc6ecd64ff7000b10dc8b3a5727 \
--hash=sha256:46ed616d5fb42f98630ed70c3529541408166c22cdfd4540b88d5f21006b0eff \
--hash=sha256:493d389a2b63c88ad56cdc35d0fa5752daac56ca755805b1b0c530f785767d5e \
--hash=sha256:4ff0d20f2e670800d3ed2b220d40984162089a6e2c9646fdb09b85e6f9a8fc29 \
--hash=sha256:54accd4b8bc202966bafafd16e69da9d5640ff92389d33d28555c5fd4f25ccb7 \
--hash=sha256:56374914b132c702aa9aa9959c550004b8847148f95e1b824772d453ac204a72 \
--hash=sha256:578383d740457fa790fdf85e6d346fda1416a40549fe8db08e5e9bd281c6a475 \
--hash=sha256:58d7a75d731e8c63614222bcb21dd992b4ab01a399f1f09dd82af17bbfc2368a \
--hash=sha256:5c5aa28df055697d7c37d2099a7bc09f559d5053c3349b1ad0c39000e611d317 \
--hash=sha256:5fc8e02f5984a55d2c653f5fea93531e9836abbd84342c1d1e17abc4a15084c2 \
--hash=sha256:63424c681923b9f3bfbc5e3205aafe790904053d42ddcc08542181a30a7a51bd \
--hash=sha256:64b1df0f83706b4ef4cfb4fb0e4c2669100fd7ecacfb59e091fad300d4e04640 \
--hash=sha256:74934ebd71950e3db69960a7da29204f89624dde411afbfb3b4858c1409b1e98 \
--hash=sha256:75669d77bb2c071333417617a235324a1618dba66f82a750362eccbe5b61d248 \
--hash=sha256:75760a47c06b5974aa5e01949bf7e66d2af4d08cb8c1d6516af5e39595397f5e \
--hash=sha256:76407ab327158c510f44ded207e2f76b657303e17cb7a572ffe2f5a8a48aa04d \
--hash=sha256:76e9c727a874b4856d11a32fb0b389afc61ce8aaf281ada613713ddeadd1cfec \
--hash=sha256:77d4c1b881076c3ba173484dfa53d3582c1c8ff1f914c6461ab70c8428b796c1 \
--hash=sha256:780c82a41dc493b62fc5884fb1d3a3b81106642c5c5c78d6a0d4cbe96d62ba7e \
--hash=sha256:7dc0713bf81287a00516ef43137273b23ee414fe41a3c14be10dd95ed98a2df9 \
--hash=sha256:7eebcdbe3677e58dd4c0e03b4f2cfa346ed4049687d839adad68cc38bb559c92 \
--hash=sha256:896689fddba4f23ef7c718279e42f8834041a21342d95e56922e1c10c0cc7afb \
--hash=sha256:96177eb5645b1c6985f5c11d03fc2dbda9ad24ec0f3a46dcce91445747e15094 \
--hash=sha256:96e25c8603a155559231c19c0349245eeb4ac0096fe3c1d0be5c47e075bd4f46 \
--hash=sha256:9d37ac69edc5614b90516807de32d08cb8e7b12260a285ee330955604ed9dd29 \
--hash=sha256:9ed6aa0726b9b60911f4aed8ec5b8dd7bf3491476015819f56473ffaef8959bd \
--hash=sha256:a487f72a25904e2b4bbc0817ce7a8de94363bd7e79890510174da9d901c38705 \
--hash=sha256:a4cbb9ff5795cd66f0066bdf5947f170f5d63a9274f99bdbca02fd973adcf2a8 \
--hash=sha256:a74d56552ddbde46c246b5b89199cb3fd182f9c346c784e1a93e4dc3f5ec9975 \
--hash=sha256:a89ce3fd220ff144bd9d54da333ec0de0399b52c9ac3d2ce34b569cf1a5748fb \
--hash=sha256:abd52a09d03adf9c763d706df707c343293d5d106aea53483e0ec8d9e310ad5e \
--hash=sha256:abd8f36c99512755b8456047b7be10372fca271bf1467a1caa88db991e7c421b \
--hash=sha256:af5bd9ccb188f6a5fdda9f1f09d9f4c86cc8a539bd48a0bfdc97723970348418 \
--hash=sha256:b02f21c1e2074943312d03d243ac4388319f2456576b2c6023041c4d57cd7019 \
--hash=sha256:b06fa97478a5f478fb05e1980980a7cdf2712015493b44d0c87606c1513ed5b1 \
--hash=sha256:b0724f05c396b0a4c36a3226c31648385deb6a65d8992644c12a4963c70326ba \
--hash=sha256:b130fe77361d6771ecf5a219d8e0817d61b236b7d8b37cc045172e574ed219e6 \
--hash=sha256:b56d5519e470d3f2fe4aa7585f0632b060d532d0696c5bdfb5e8319e1d0f69a2 \
--hash=sha256:b67b819628e3b748fd3c2192c15fb951f549d0f47c0449af0764d7647302fda3 \
--hash=sha256:ba1711cda2d30634a7e452fc79eabcadaffedf241ff206db2ee93dd2c89a60e7 \
--hash=sha256:bbeccb1aa40ab88cd29e6c7d8585582c99548f55f9b2581dfc5ba68c59a85752 \
--hash=sha256:bd84395aab8e4d36263cd1b9308cd504f6cf713b7d6d3ce25ea55670baec5416 \
--hash=sha256:c99f4309f5145b93eca6e35ac1a988f0dc0a7ccf9ccdcd78d3c0adf57224e62f \
--hash=sha256:ca1cccf838cd28d5a0883b342474c630ac48cac5df0ee6eacc9c7290f76b11c1 \
--hash=sha256:cd525e0e52a5ff16653a3fc9e3dd827981917d34996600bbc34c05d048ca35cc \
--hash=sha256:cdb4f085756c96a3af04e6eca7f08b1345e94b53af8921b25c72f096e704e145 \
--hash=sha256:ce42618f67741d4697684e501ef02f29e758a123aa2d669e2d964ff734ee00ee \
--hash=sha256:d06730c6aed78cee4126234cf2d071e01b44b915e725a6cb439a879ec9754a3a \
--hash=sha256:d5fe3e099cf07d0fb5a1e23d399e5d4d1ca3e6dfcbe5c8570ccff3e9208274f7 \
--hash=sha256:d6bcbfc99f55655c3d93feb7ef3800bd5bbe963a755687cbf1f490a71fb7794b \
--hash=sha256:d787272ed958a05b2c86311d3a4135d3c2aeea4fc655705f074130aa57d71653 \
--hash=sha256:e169e957c33576f47e21864cf3fc9ff47c223a4ebca8960079b8bd36cb014fd0 \
--hash=sha256:e20076a211cd6f9b44a6be58f7eeafa7ab5720eb796975d0c03f05b47d89eb90 \
--hash=sha256:e826aadda3cae59295b95343db8f3d965fb31059da7de01ee8d1c40a60398b29 \
--hash=sha256:eef4d64c650f33347c1f9266fa5ae001440b232ad9b98f1f43dfe7a79435c0a6 \
--hash=sha256:f2e69b3ed24544b0d3dbe2c5c0ba5153ce50dcebb576fdc4696d52aa22db6034 \
--hash=sha256:f87ec75864c37c4c6cb908d282e1969e79763e0d9becdfe9fe5473b7bb1e5f09 \
--hash=sha256:fbec11614dba0424ca72f4e8ba3c420dba07b4a7c206c8c8e4e73f2e98f4c559 \
--hash=sha256:fd69666217b62fa5d7c6aa88e507493a34dec4fa20c5bd925e4bc12fce586639
toml==0.10.2
# via darker
urllib3==2.0.4
# via requests
wrapt==1.15.0
# via deprecated
@@ -1,8 +0,0 @@
black~=25.1
darker==2.1.1
PyGithub==2.6.1
cryptography>=43.0.1
urllib3>=2.5.0
requests>=2.32.4
idna>=3.7
certifi>=2024.7.4
+1 -1
Submodule External/range-v3 deleted from ca1388fb9d.
+1 -1
+1 -1
View File
@@ -78,6 +78,6 @@ install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
DESTINATION include
COMPONENT Development)
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
+170 -11
View File
@@ -118,6 +118,41 @@ def print_man_env_option(name, desc, default, no_json_key):
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
def print_man_options(options):
output_man.write(".Sh OPTIONS\n")
output_man.write(".Bl -tag -width -indent\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
default = op_vals["Default"]
value_type = op_vals["Type"]
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = op_vals["TextDefault"]
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
# Wrap the string argument in quotes
default = "'" + default + "'"
print_man_option(
short,
long,
op_vals["Desc"],
default
)
if (value_type == "strenum"):
Enums = op_vals["Enums"]
output_man.write("\\fBAvailable Options:\\fR\n")
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
output_man.write("\n.sp\n")
output_man.write(".El\n")
def print_man_environment(options):
output_man.write(".Sh ENVIRONMENT\n")
output_man.write(".Bl -tag -width -indent\n")
@@ -159,7 +194,7 @@ def print_man_environment_tail():
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
"This will override the full path",
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -173,7 +208,7 @@ def print_man_environment_tail():
"One must be careful with this option as it will override any applications that load with execve as well"
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -192,8 +227,8 @@ def print_man_environment_tail():
"PORTABLE",
[
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
"For FEX on Linux:",
"These files are instead read from <FEXPath>/fex-emu/ by default.",
"For FEXInterpreter on Linux:",
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
"For Arm64ec/Wow64 WINE builds:",
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
@@ -205,12 +240,20 @@ def print_man_header():
.Dt FEX
.Os Linux
.Sh NAME
.Nm FEX
.Nm FEXLoader
.Nm FEXInterpreter
.Nm FEXBash
.Nd Fast x86-64 and x86 emulation.
.Sh SYNOPSIS
.Nm
.Ar <args> ...
.Op options
.Op Ar --
.Ar Application
<args> ...
.Pp
.Nm FEXInterpreter
.Ar Application
<args> ...
.Pp
.Nm FEXBash
.Ar <args> ...
@@ -318,6 +361,82 @@ def print_config_option(type, group_name, json_name, default_value, short, choic
output_argloader.write("\n");
def print_argloader_options(options):
output_argloader.write("#ifdef BEFORE_PARSE\n")
output_argloader.write("#undef BEFORE_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
default = op_vals["Default"]
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
# Wrap the string argument in quotes
default = "\"" + default + "\""
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = "\"" + op_vals["TextDefault"] + "\""
short = None
choices = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if ("Choices" in op_vals):
choices = op_vals["Choices"]
print_config_option(
op_vals["Type"],
op_group,
op_key,
default,
short,
choices,
op_vals["Desc"])
output_argloader.write("\n")
output_argloader.write("#endif\n")
def print_parse_argloader_options(options):
output_argloader.write("#ifdef AFTER_PARSE\n")
output_argloader.write("#undef AFTER_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
output_argloader.write("if (Options.is_set_by_user(\"{0}\")) {{\n".format(op_key))
value_type = op_vals["Type"]
NeedsString = False
conversion_func = "fextl::fmt::format(\"{}\", "
if ("ArgumentHandler" in op_vals):
NeedsString = True
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
if (value_type == "str"):
NeedsString = True
conversion_func = "std::move("
if (value_type == "bool"):
# boolean values need a decimal specifier. Otherwise fmt prints strings.
conversion_func = "fextl::fmt::format(\"{:d}\", "
if (value_type == "strenum"):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key, op_key))
elif (value_type == "strarray"):
# these need a bit more help
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
output_argloader.write("\t}\n")
else:
if (NeedsString):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
else:
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_envloader_options(options):
output_argloader.write("#ifdef ENVLOADER\n")
output_argloader.write("#undef ENVLOADER\n")
@@ -328,13 +447,13 @@ def print_parse_envloader_options(options):
value_type = op_vals["Type"]
if (value_type == "strenum"):
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
output_argloader.write("\tValue = FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View);\n".format(op_key, op_key))
output_argloader.write("Value = FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View);\n".format(op_key, op_key, op_key))
output_argloader.write("}\n")
if ("ArgumentHandler" in op_vals):
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
output_argloader.write("\tValue = {0}(Value_View);\n".format(conversion_func))
output_argloader.write("Value = {0}(Value_View);\n".format(conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
@@ -348,15 +467,15 @@ def print_parse_jsonloader_options(options):
value_type = op_vals["Type"]
if (value_type == "strenum"):
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key))
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
output_argloader.write("}\n")
elif (value_type == "strarray"):
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
output_argloader.write("\tAppendStrArrayValue(KeyOption, ConfigString);\n")
output_argloader.write("}\n")
assert op_key is not None, "No options found in JSONLOADER"
output_argloader.write("else {\n")
output_argloader.write("\tSet(KeyOption, ConfigString);\n")
output_argloader.write("else {{\n".format(op_key))
output_argloader.write("Set(KeyOption, ConfigString);\n")
output_argloader.write("}\n")
output_argloader.write("#endif\n")
@@ -398,6 +517,41 @@ def print_parse_enum_options(options):
output_argloader.write("#endif\n")
def check_for_duplicate_options(options):
short_map = []
long_map = []
# Spin through all the items and see if we have a duplicate option
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
long_invert = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if (op_vals["Type"] == "bool"):
long_invert = "no-" + long
# Check for short key duplication
if (short != None):
if (short in short_map):
raise Exception("Short config '{0}' for option '{1}' has duplicate entry!".format(short, op_key))
else:
short_map.append(short)
# Check for long key duplication
if (long in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long))
else:
long_map.append(long)
# Check for long key duplication
if (long_invert != None):
if (long_invert in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long_invert))
else:
long_map.append(long_invert)
if (len(sys.argv) < 5):
sys.exit()
@@ -414,6 +568,8 @@ json_object = json.loads(json_text)
options = json_object["Options"]
unnamed_options = json_object["UnnamedOptions"]
check_for_duplicate_options(options)
# Generate config include file
output_file = open(output_filename, "w")
print_header()
@@ -425,6 +581,7 @@ output_file.close()
# Generate man file
output_man = open(output_man_page, "w")
print_man_header()
print_man_options(options)
print_man_environment(options)
print_man_tail()
@@ -432,6 +589,8 @@ output_man.close()
# Generate argument loader code
output_argloader = open(output_argumentloader_filename, "w")
print_argloader_options(options);
print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
+15 -42
View File
@@ -58,7 +58,6 @@ class OpDefinition:
JITDispatch: bool
JITDispatchOverride: str
TiedSource: int
Inline: list
Arguments: list
EmitValidation: list
Desc: list
@@ -279,12 +278,6 @@ def parse_ops(ops):
if "TiedSource" in op_val:
OpDef.TiedSource = op_val["TiedSource"]
# Pad Inline out to the argument count
OpDef.Inline = [''] * len(OpDef.Arguments)
if "Inline" in op_val:
Value = op_val["Inline"]
OpDef.Inline[0:len(Value)] = Value
# Do some fixups of the data here
if len(OpDef.EmitValidation) != 0:
for i in range(len(OpDef.EmitValidation)):
@@ -404,16 +397,16 @@ def print_ir_sizes():
// Make sure our array maps directly to the IROps enum
static_assert(IRSizes[IROps::OP_LAST] == -1ULL);
[[nodiscard]] inline size_t GetSize(IROps Op) { return IRSizes[Op]; }
[[nodiscard, gnu::const]] std::string_view const& GetName(IROps Op);
[[nodiscard, gnu::const]] uint8_t GetArgs(IROps Op);
[[nodiscard, gnu::const]] uint8_t GetRAArgs(IROps Op);
[[nodiscard, gnu::const]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
[[nodiscard, gnu::const]] bool HasSideEffects(IROps Op);
[[nodiscard, gnu::const]] bool ImplicitFlagClobber(IROps Op);
[[nodiscard, gnu::const]] bool GetHasDest(IROps Op);
[[nodiscard, gnu::const]] bool LoweredX87(IROps Op);
[[nodiscard, gnu::const]] int8_t TiedSource(IROps Op);
[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }
[[nodiscard, gnu::const, gnu::visibility("default")]] std::string_view const& GetName(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetArgs(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetRAArgs(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] bool HasSideEffects(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] bool ImplicitFlagClobber(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] bool GetHasDest(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] bool LoweredX87(IROps Op);
[[nodiscard, gnu::const, gnu::visibility("default")]] int8_t TiedSource(IROps Op);
#undef IROP_SIZES
#endif
@@ -595,13 +588,13 @@ def print_ir_arg_printer():
output_file.write("#endif\n")
def print_validation(op):
if len(op.EmitValidation) != 0:
output_file.write("#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
if op.EmitValidation != None:
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
for Validation in op.EmitValidation:
Sanitized = Validation.replace("\"", "\\\"")
output_file.write("\t\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
output_file.write("#endif\n")
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
output_file.write("\t\t#endif\n")
# Print out IR allocator helpers
def print_ir_allocator_helpers():
@@ -780,29 +773,9 @@ def print_ir_allocator_helpers():
output_file.write(") {\n")
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
idx = 0
for arg in op.Arguments:
if arg.IsSSA:
# Inline an immediate if we can
inline = op.Inline[idx]
idx += 1
if inline != '':
Sized = "Size" in [x.Name for x in op.Arguments]
P = ["Size" if Sized else "OpSize::i64Bit", arg.Name]
# A few cases need extra info plumbed.
if inline == "SubtractZero":
P += ["Src2"]
elif inline == "Mem":
P += ["OffsetType", "OffsetScale"]
elif inline == "Memtso":
P += ["OffsetType", "OffsetScale", "true /* TSO */"]
inline = "Mem"
output_file.write(f"\t\t{arg.Name} = Inline{inline}({', '.join(P)});\n")
output_file.write(f"\t\t{arg.Name}->AddUse();\n")
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
# Insert validation here. This is skipped for the
# OrderedNodeWrapper version because validation can depend on
+8 -3
View File
@@ -1,3 +1,4 @@
include(GNUInstallDirs)
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
set (FEXCORE_BASE_SRCS
@@ -18,12 +19,14 @@ set (SRCS
Common/JitSymbols.cpp
Interface/Context/Context.cpp
Interface/Core/LookupCache.cpp
Interface/Core/CodeCache.cpp
Interface/Core/Core.cpp
Interface/Core/CPUBackend.cpp
Interface/Core/Addressing.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/ObjectCache/JobHandling.cpp
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
Interface/Core/ObjectCache/ObjectCacheService.cpp
Interface/Core/OpcodeDispatcher/AVX_128.cpp
Interface/Core/OpcodeDispatcher/Crypto.cpp
Interface/Core/OpcodeDispatcher/Flags.cpp
@@ -31,6 +34,7 @@ set (SRCS
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher/X87F64.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/X86Tables.cpp
Interface/Core/X86HelperGen.cpp
Interface/Core/ArchHelpers/Arm64Emitter.cpp
Interface/Core/Dispatcher/Dispatcher.cpp
@@ -58,15 +62,16 @@ set (SRCS
Interface/Core/X86Tables/VEXTables.cpp
Interface/Core/X86Tables/X87Tables.cpp
Interface/GDBJIT/GDBJIT.cpp
Interface/IR/AOTIR.cpp
Interface/IR/IRDumper.cpp
Interface/IR/IREmitter.cpp
Interface/IR/PassManager.cpp
Interface/IR/Passes/ConstProp.cpp
Interface/IR/Passes/IRDumperPass.cpp
Interface/IR/Passes/IRValidation.cpp
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/x87StackOptimizationPass.cpp
Utils/LongJump.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/Profiler.cpp
@@ -203,7 +208,7 @@ add_custom_target(CONFIG_INC
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
# Add in diagnostic colours if the option is available.
# Ninja code generator will kill colours if this isn't here
+3 -1
View File
@@ -2,10 +2,12 @@
#pragma once
#include <FEXCore/fextl/memory.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <chrono>
#include <cstddef>
#include <cstdint>
#include <cstdio>
#include <memory>
#include <string_view>
namespace FEXCore {
+2 -58
View File
@@ -157,30 +157,7 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
/*
* Check for invalid operation cases first - Intel FPREM sets Invalid Operation
* for several cases including infinity dividend and zero divisor.
*/
X80SoftFloat result = 0;
if (HandleInfinityOp(state, lhs, result)) {
return result;
} else if (lhs.Exponent == 0x7FFF && (lhs.Significand & 0x7FFFFFFFFFFFFFFFULL)) { // NaN
// propagate NaN
state->exceptionFlags |= softfloat_flag_invalid;
return lhs;
}
// Check for zero divisor - fprem(x, 0) is invalid operation
if (rhs.Exponent == 0 && rhs.Significand == 0) {
state->exceptionFlags |= softfloat_flag_invalid;
// Return QNaN
result.Sign = 0;
result.Exponent = 0x7FFF;
result.Significand = 0xC000000000000000ULL;
return result;
}
/*
* FPREM is not an IEEE-754 remainder. From the Intel spec:
* FPREM is not an IEEE-754 remainder. From the spec:
*
* Computes the remainder obtained from dividing the value in the ST(0)
* register (the dividend) by the value in the ST(1) register (the divisor
@@ -294,11 +271,7 @@ struct FEX_PACKED X80SoftFloat {
FCMP(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs, bool* eq, bool* lt, bool* nan) {
*eq = extF80_eq(state, lhs, rhs);
*lt = extF80_lt(state, lhs, rhs);
// Use IEEE 754 semantics: unordered if neither <, =, nor > is true
// This is more reliable than custom NaN detection
bool gt = !(*eq) && !(*lt) && extF80_le(state, rhs, lhs);
*nan = !(*eq) && !(*lt) && !gt;
*nan = IsNan(lhs) || IsNan(rhs);
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSCALE(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
@@ -417,11 +390,6 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
X80SoftFloat result;
if (HandleInfinityOp(state, lhs, result)) {
return result;
}
BIGFLOAT Src_d = lhs.ToFMax(state);
Src_d = FEXCore::cephes_128bit::tanl(Src_d);
return X80SoftFloat(state, Src_d);
@@ -443,11 +411,6 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
X80SoftFloat result;
if (HandleInfinityOp(state, lhs, result)) {
return result;
}
BIGFLOAT Src_d = lhs.ToFMax(state);
Src_d = FEXCore::cephes_128bit::sinl(Src_d);
return X80SoftFloat(state, Src_d);
@@ -469,11 +432,6 @@ struct FEX_PACKED X80SoftFloat {
return Result;
#else
X80SoftFloat result;
if (HandleInfinityOp(state, lhs, result)) {
return result;
}
BIGFLOAT Src_d = lhs.ToFMax(state);
Src_d = FEXCore::cephes_128bit::cosl(Src_d);
return X80SoftFloat(state, Src_d);
@@ -633,20 +591,6 @@ private:
static constexpr uint64_t IntegerBit = (1ULL << 63);
static constexpr uint64_t Bottom62Significand = ((1ULL << 62) - 1);
static constexpr uint32_t ExponentBias = 16383;
// Helper function to check for infinity and set invalid operation flag.
// Returns true if infinity is dealt with, false otherwise.
FEXCORE_PRESERVE_ALL_ATTR static bool HandleInfinityOp(softfloat_state* state, const X80SoftFloat& arg, X80SoftFloat& result) {
if (arg.Exponent == 0x7FFF && arg.Significand == 0x8000000000000000ULL) {
state->exceptionFlags |= softfloat_flag_invalid;
// Return QNaN.
result.Sign = 0;
result.Exponent = 0x7FFF;
result.Significand = 0xC000000000000000ULL;
return true;
}
return false;
}
};
#ifndef _WIN32
+22 -11
View File
@@ -7,58 +7,69 @@
#include <optional>
namespace FEXCore::StrConv {
inline bool Conv(std::string_view Value, bool* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, bool* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint8_t* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, uint8_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int8_t* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, int8_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint16_t* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, uint16_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int16_t* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, int16_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint32_t* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, uint32_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int32_t* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, int32_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, uint64_t* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, uint64_t* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
inline bool Conv(std::string_view Value, int64_t* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, int64_t* Result) {
*Result = std::strtoll(Value.data(), nullptr, 0);
return true;
}
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
inline bool Conv(std::string_view Value, T* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, T* Result) {
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
return true;
}
inline bool Conv(std::string_view Value, fextl::string* Result) {
[[maybe_unused]]
static bool Conv(std::string_view Value, fextl::string* Result) {
*Result = Value;
return true;
}
-2
View File
@@ -4,8 +4,6 @@
#ifdef _M_X86_64
#include <xmmintrin.h>
#include <immintrin.h>
#else
#include <cstdint>
#endif
namespace FEXCore {
+62 -43
View File
@@ -4,6 +4,7 @@
"Multiblock": {
"Type": "bool",
"Default": "true",
"ShortArg": "m",
"Desc": [
"Controls multiblock code compilation",
"Can cause long JIT compilation times and stutter"
@@ -12,10 +13,22 @@
"MaxInst": {
"Type": "int32",
"Default": "5000",
"ShortArg": "n",
"Desc": [
"Maximum number of instruction to store in a block"
]
},
"CacheObjectCodeCompilation": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
"TextDefault": "none",
"Choices": [ "none", "read", "readwrite" ],
"ArgumentHandler": "CacheObjectCodeHandler",
"Desc": [
"Cache JIT object code to drive.",
"Allows JIT code to be shared between applications"
]
},
"HostFeatures": {
"Type": "strenum",
"Default": "FEXCore::Config::HostFeatures::OFF",
@@ -55,11 +68,7 @@
"ENABLESVEBITPERM": "enablesvebitperm",
"DISABLESVEBITPERM": "disablesvebitperm",
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
"DISABLEPRESERVEALLABI": "disablepreserveallabi",
"ENABLEWFXT": "enablewfxt",
"DISABLEWFXT": "disablewfxt",
"ENABLE3DNOW": "enable3dnow",
"DISABLE3DNOW": "disable3dnow"
"DISABLEPRESERVEALLABI": "disablepreserveallabi"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
@@ -80,9 +89,7 @@
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
"\t{enable,disable}3dnow: Will force enable or disable 3DNow even if the host doesn't support it"
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
]
},
"SmallTSCScale": {
@@ -97,6 +104,7 @@
"RootFS": {
"Type": "str",
"Default": "",
"ShortArg": "R",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
@@ -111,6 +119,7 @@
"ThunkHostLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
]
@@ -118,6 +127,7 @@
"ThunkGuestLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
@@ -125,6 +135,7 @@
"ThunkConfig": {
"Type": "str",
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks.",
"This can be a filesystem path",
@@ -139,6 +150,7 @@
"Env": {
"Type": "strarray",
"Default": "",
"ShortArg": "E",
"Desc": [
"Adds an environment variable to the emulated environment."
]
@@ -146,6 +158,7 @@
"HostEnv": {
"Type": "strarray",
"Default": "",
"ShortArg": "H",
"Desc": [
"Adds an environment variable to the host environment.",
"This can be useful for setting environment variables that thunks can pick up.",
@@ -164,6 +177,7 @@
"SingleStep": {
"Type": "bool",
"Default": "false",
"ShortArg": "S",
"Desc": [
"Single stepping configuration."
]
@@ -171,6 +185,7 @@
"GdbServer": {
"Type": "bool",
"Default": "false",
"ShortArg": "G",
"Desc": [
"Enables the GDB server."
]
@@ -204,6 +219,7 @@
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"ShortArg": "g",
"Desc": [
"When the test harness ends, print the GPR state."
]
@@ -211,6 +227,7 @@
"O0": {
"Type": "bool",
"Default": "false",
"ShortArg": "O0",
"Desc": [
"Disables optimizations passes for debugging."
]
@@ -300,6 +317,7 @@
"SilentLog": {
"Type": "bool",
"Default": "true",
"ShortArg": "s",
"Desc": [
"Disables logging"
]
@@ -307,6 +325,7 @@
"OutputLog": {
"Type": "str",
"Default": "server",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
"[stdout, stderr, server, <Filename>]"
@@ -327,13 +346,6 @@
"Enables FEX's low-overhead sampling profile statistics.",
"Requires a supported version of Mangohud to see the results"
]
},
"TraceProfiler": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enables FEX's trace profiler. Using gpuvis or tracy"
]
}
},
"Hacks": {
@@ -388,6 +400,14 @@
"This is required to ensure a split-lock doesn't tear inside the process"
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
"Desc": [
"Automatically enables TSO when shared memory is used.",
"Should work without issues in most cases."
]
},
"VolatileMetadata": {
"Type": "bool",
"Default": "true",
@@ -450,16 +470,32 @@
"Desc": [
"Contrains the startup sleep to only apply to processes that match this name."
]
},
"MonoHacks": {
"Type": "bool",
"Default": "true",
"Desc": [
"Permits a hook-based SMC approach and smaller JIT blocks when mono is detected."
]
}
},
"Misc": {
"AOTIRCapture": {
"Type": "bool",
"Default": "false",
"Desc": [
"Captures IR and generates an AOT IR cache.",
"Captures both the loaded executable and libraries it loads."
]
},
"AOTIRGenerate": {
"Type": "bool",
"Default": "false",
"Desc": [
"Scans file for executable code and generates an AOT IR cache.",
"Does not run the executable."
]
},
"AOTIRLoad": {
"Type": "bool",
"Default": "false",
"Desc": [
"Loads an AOT IR cache for the loaded executable."
]
},
"ServerSocketPath": {
"Type": "str",
"Default": "",
@@ -473,32 +509,15 @@
"Desc": [
"Disables inline syscalls in order to support seccomp handling"
]
},
"ExtendedVolatileMetadata": {
"Type": "str",
"Default": "",
"Desc": [
"Configuration provided volatile metadata. Only implemented for WoW64/arm64ec.",
"Limited in its use but can be handy.",
"Extends on top of what Microsoft has for volatile metadata, but also supported for WoW64.",
"Colon delimited modules, then semi-colon delimited instructions, then comma delimited ranges",
"Default disables TSO in the module, unless instructions overlap the range",
"<module>;<offset begin>-<offset-end>,...;<instruction offset to force TSO>,...:<another>",
"examples:",
" * Disable TSO for a full module: Just provide the module name:",
" `hl2_linux`",
" * Disable TSO for a part of the module:",
" `hl2_linux;<offset begin>-<offset-end>`",
" * Disable TSO for a part of the module, but enable TSO for some instructions within the module",
" `hl2_linux;<offset begin>-<offset-end>;<instruction offset>,<instruction offset>`",
" * Disable TSO for multiple modules",
" `hl2_linux:libsdl2.so`"
]
}
}
},
"UnnamedOptions": {
"Misc": {
"IS_INTERPRETER": {
"Type": "bool",
"Default": "false"
},
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false"
+8 -3
View File
@@ -1,7 +1,6 @@
// SPDX-License-Identifier: MIT
#include "Interface/Context/Context.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Core/CoreState.h>
@@ -9,12 +8,18 @@
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Core/Thunks.h>
#include "FEXCore/Debug/InternalThreadState.h"
#include <string.h>
#include <utility>
namespace FEXCore::Context {
void InitializeStaticTables(OperatingMode Mode) {
X86Tables::InitializeInfoTables(Mode);
IR::InstallOpcodeHandlers(Mode);
}
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext(const FEXCore::HostFeatures& Features) {
return fextl::make_unique<FEXCore::Context::ContextImpl>(Features);
}
+97 -70
View File
@@ -2,54 +2,65 @@
#pragma once
#include "Common/JitSymbols.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/X86HelperGen.h"
#include <Interface/IR/IntrusiveIRList.h>
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/AOTIR.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <stdint.h>
#include <atomic>
#include <cstddef>
#include <cstdint>
#include <mutex>
#include <optional>
#include <shared_mutex>
namespace FEXCore {
class SignalDelegator;
class CodeLoader;
class ThunkHandler;
namespace Core {
struct DebugData;
struct InternalThreadState;
} // namespace Core
namespace CodeSerialize {
class CodeObjectSerializeService;
}
namespace CPU {
class Arm64JITCore;
class Dispatcher;
} // namespace CPU
namespace HLE {
class SourcecodeResolver;
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
} // namespace HLE
} // namespace FEXCore
namespace FEXCore::IR {
struct IRListCopy;
class IRListView;
namespace Validation {
class IRValidation;
}
} // namespace FEXCore::IR
namespace FEXCore::Context {
struct FEX_PACKED ExitFunctionLinkData {
uint64_t HostCode;
uint64_t HostBranch;
uint64_t GuestRIP;
int64_t CallerOffset;
};
struct CustomIRResult {
@@ -64,23 +75,7 @@ struct CustomIRResult {
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
class CodeCache : public AbstractCodeCache {
public:
CodeCache(ContextImpl&);
~CodeCache();
ContextImpl& CTX;
bool IsGeneratingCache = false;
void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
void InitiateCacheGeneration() override {
IsGeneratingCache = true;
}
};
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
public:
// Context base class implementation.
bool InitCore() override;
@@ -94,7 +89,6 @@ public:
bool IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) override;
bool IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) override;
uint64_t GetGuestBlockEntry(FEXCore::Core::InternalThreadState* Thread) override;
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs, uint64_t PSTATE) override;
@@ -150,18 +144,35 @@ public:
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
CodeCache& GetCodeCache() override {
return CodeCache;
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
}
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
}
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
}
void FinalizeAOTIRCache() override {
IRCaptureCache.FinalizeAOTIRCache();
}
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start,
uint64_t Length) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
return CodeInvalidationMutex;
}
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
@@ -176,13 +187,14 @@ public:
void RemoveForceTSOInformation(uint64_t Address, uint64_t Size) override;
void MarkMonoDetected() override {
MonoDetected = true;
}
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
public:
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
friend class FEXCore::IR::Validation::IRValidation;
struct {
uint64_t VirtualMemSize {1ULL << 36};
uint64_t TSCScale = 0;
@@ -195,9 +207,13 @@ public:
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
@@ -206,12 +222,12 @@ public:
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
} Config;
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
@@ -225,23 +241,37 @@ public:
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
FEXCore::ThunkHandler* ThunkHandler {};
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CodeCache CodeCache;
SignalDelegator* SignalDelegation {};
X86GeneratedCode X86CodeGen;
ContextImpl(const FEXCore::HostFeatures& Features);
~ContextImpl();
static bool ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, ExitFunctionLinkData* Record) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
// This is used as a replacement for the SMC writes in the mono callsite backpatcher that avoids atomic operations
// (safe as the invalidation mutex is locked) and manually invalidates the modified range. Allowing SMC to be detected
// even if faulting is disabled.
static void MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint8_t Size, uint64_t Address, uint64_t Value);
return Fn(Frame, Record);
}
void RemoveCustomIREntrypoint(FEXCore::Core::InternalThreadState* Thread, uintptr_t Entrypoint);
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
// Must be called from owning thread
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
// NOTE: Other threads sharing the same CodeBuffer may reference
// invalidated data ranges through their L1/L2 caches. This is
// not currently a problem since FEX does not repurpose the
// invalidated CodeBuffer memory range currently.
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
struct GenerateIRResult {
std::optional<IR::IRListView> IRView;
@@ -249,23 +279,25 @@ public:
uint64_t TotalInstructionsLength;
uint64_t StartAddr;
uint64_t Length;
bool NeedsAddGuestCodeRanges;
};
[[nodiscard]]
GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
struct CompileCodeResult {
CPU::CPUBackend::CompiledCode CompiledCode;
void* CompiledCode;
fextl::unique_ptr<FEXCore::Core::DebugData> DebugData;
uint64_t StartAddr;
uint64_t Length;
bool NeedsAddGuestCodeRanges;
};
[[nodiscard]]
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
IR::OpSize GetGPROpSize() const {
return Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
}
FEXCore::JITSymbols Symbols;
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
@@ -300,10 +332,6 @@ public:
return ExitOnHLT;
}
bool AreMonoHacksActive() const {
return Config.MonoHacks && MonoDetected;
}
protected:
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
@@ -316,9 +344,12 @@ protected:
VectorAtomicTSOEmulationEnabled = true;
MemcpyAtomicTSOEmulationEnabled = true;
} else {
AtomicTSOEmulationEnabled = Config.TSOEnabled;
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
MemcpyAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.MemcpySetTSOEnabled;
// Atomic TSO emulation only enabled if the config option is enabled.
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
}
}
@@ -332,6 +363,10 @@ private:
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
bool IsMemoryShared = false;
bool SupportsHardwareTSO = false;
bool AtomicTSOEmulationEnabled = true;
bool VectorAtomicTSOEmulationEnabled = false;
@@ -342,16 +377,8 @@ private:
std::shared_mutex CustomIRMutex;
std::atomic<bool> HasCustomIRHandlers {};
struct CustomIRHandlerEntry final {
CustomIREntrypointHandler Handler;
void* Creator;
void* Data;
};
fextl::unordered_map<uint64_t, CustomIRHandlerEntry> CustomIRHandlers;
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void*, void*>> CustomIRHandlers;
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
fextl::set<uint64_t> ForceTSOInstructions;
bool MonoDetected = false;
std::atomic<uint64_t> MonoBackpatcherBlock;
};
} // namespace FEXCore::Context
+38 -24
View File
@@ -11,7 +11,8 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
Ref Tmp = A.Base;
if (A.Offset) {
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Offset) : IREmit->Constant(A.Offset);
Ref Offset = IREmit->_Constant(A.Offset);
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, Offset) : Offset;
}
if (A.Index) {
@@ -21,10 +22,10 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
if (Tmp) {
Tmp = IREmit->_AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
} else {
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->Constant(Log2));
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->_Constant(Log2));
}
} else {
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Index) : A.Index;
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Index) : A.Index;
}
}
@@ -40,28 +41,32 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
} else if (A.Offset) {
uint64_t X = A.Offset;
X &= (1ull << Bits) - 1;
Tmp = IREmit->Constant(X);
Tmp = IREmit->_Constant(X);
}
}
if (A.Segment && AddSegmentBase) {
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Segment) : A.Segment;
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Segment) : A.Segment;
}
return Tmp ?: IREmit->Constant(0);
return Tmp ?: IREmit->_Constant(0);
}
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
IR::OpSize AccessSize) {
auto SoftwareAddressCalculation = [IREmit, &A, GPRSize]() -> AddressMode {
return {
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
.Index = IREmit->Invalid(),
};
};
const auto Is32Bit = GPRSize == OpSize::i32Bit;
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
if (!GPRSizeMatchesAddrSize || OffsetIndexToLargeFor32Bit) {
// If address size doesn't match GPR size then no optimizations can occur.
return {
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
.Index = IREmit->Invalid(),
};
return SoftwareAddressCalculation();
}
// Loadstore rules:
@@ -95,31 +100,43 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
const bool OffsetIsSIMM9 = A.Offset && A.Offset >= -256 && A.Offset <= 255;
const bool OffsetIsUnsignedScaled = A.Offset > 0 && (A.Offset & (AccessSizeAsImm - 1)) == 0 && (A.Offset / AccessSizeAsImm) <= 4095;
if ((AtomicTSO && !Vector && HostSupportsTSOImm9 && OffsetIsSIMM9) || (!AtomicTSO && (OffsetIsSIMM9 || OffsetIsUnsignedScaled))) {
auto InlineImmOffsetLoadstore = [IREmit, &GPRSize](AddressMode A) -> AddressMode {
// Peel off the offset
AddressMode B = A;
B.Offset = 0;
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset),
.Index = IREmit->_Constant(A.Offset),
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = 1,
};
}
};
if (AtomicTSO) {
// TODO: LRCPC3 support for vector Imm9.
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
// ScaledRegisterLoadstore
auto ScaledRegisterLoadstore = [IREmit, GPRSize](AddressMode A) -> AddressMode {
if (A.Index && A.Segment) {
A.Base = IREmit->Add(GPRSize, A.Base, A.Segment);
A.Base = IREmit->_Add(GPRSize, A.Base, A.Segment);
} else if (A.Segment) {
A.Index = A.Segment;
A.IndexScale = 1;
}
return A;
};
if (AtomicTSO) {
if (!Vector) {
if (HostSupportsTSOImm9 && OffsetIsSIMM9) {
return InlineImmOffsetLoadstore(A);
}
} else {
// TODO: LRCPC3 support for vector Imm9.
}
} else {
if (OffsetIsSIMM9 || OffsetIsUnsignedScaled) {
return InlineImmOffsetLoadstore(A);
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
return ScaledRegisterLoadstore(A);
}
}
if (Vector || !AtomicTSO) {
@@ -133,7 +150,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset),
.Index = IREmit->_Constant(A.Offset),
.IndexType = MEM_OFFSET_SXTX,
.IndexScale = 1,
};
@@ -142,10 +159,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
}
// Fallback on software address calculation
return {
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
.Index = IREmit->Invalid(),
};
return SoftwareAddressCalculation();
}
@@ -73,13 +73,13 @@ namespace x64 {
ARMEmitter::Reg::r8, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
};
constexpr std::array<ARMEmitter::Register, 7> RA = {
constexpr std::array<ARMEmitter::Register, 8> RA = {
// All these callee saved
ARMEmitter::Reg::r20, ARMEmitter::Reg::r21, ARMEmitter::Reg::r22, ARMEmitter::Reg::r23,
ARMEmitter::Reg::r24, ARMEmitter::Reg::r30, ARMEmitter::Reg::r18,
ARMEmitter::Reg::r24, ARMEmitter::Reg::r25, ARMEmitter::Reg::r30, ARMEmitter::Reg::r18,
};
constexpr unsigned RAPairs = 4;
constexpr unsigned RAPairs = 6;
// Dynamic GPRs
constexpr std::array<ARMEmitter::Register, 2> PreserveAll_Dynamic = {
@@ -143,18 +143,18 @@ namespace x64 {
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r8,
};
constexpr std::array<ARMEmitter::Register, 6> RA = {
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15, ARMEmitter::Reg::r16, ARMEmitter::Reg::r30,
constexpr std::array<ARMEmitter::Register, 7> RA = {
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15,
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
};
constexpr std::array<ARMEmitter::Register, 5> PreserveAll_Dynamic = {ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r16,
ARMEmitter::Reg::r17, ARMEmitter::Reg::r30};
constexpr std::array<ARMEmitter::Register, 5> PreserveAll_Dynamic = {
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
};
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = {ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14,
ARMEmitter::Reg::r15, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
ARMEmitter::Reg::r30};
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = RA;
constexpr unsigned RAPairs = 4;
constexpr unsigned RAPairs = 6;
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
ARMEmitter::VReg::v0, ARMEmitter::VReg::v1, ARMEmitter::VReg::v2, ARMEmitter::VReg::v3,
@@ -245,12 +245,14 @@ namespace x32 {
REG_AF,
};
constexpr std::array<ARMEmitter::Register, 14> RA = {
constexpr std::array<ARMEmitter::Register, 15> RA = {
// All these callee saved
ARMEmitter::Reg::r20,
ARMEmitter::Reg::r21,
ARMEmitter::Reg::r22,
ARMEmitter::Reg::r23,
ARMEmitter::Reg::r24,
ARMEmitter::Reg::r25,
// Registers only available on 32-bit
// All these are caller saved (except for r19).
@@ -263,7 +265,6 @@ namespace x32 {
ARMEmitter::Reg::r29,
ARMEmitter::Reg::r30,
ARMEmitter::Reg::r24,
ARMEmitter::Reg::r19,
};
@@ -272,7 +273,7 @@ namespace x32 {
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
};
constexpr unsigned RAPairs = 10;
constexpr unsigned RAPairs = 12;
// All are caller saved
constexpr std::array<ARMEmitter::VRegister, 8> SRAFPR = {
@@ -369,8 +370,6 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
// Hardcode a 256-bit vector width if we are running in the simulator.
// Allow the user to override this.
Simulator.SetVectorLengthInBits(ForceSVEWidth() ? ForceSVEWidth() : 256);
// FEX doesn't support GCS.
Simulator.DisableGCSCheck();
#endif
#ifdef VIXL_DISASSEMBLER
// Only setup the disassembler if enabled.
@@ -496,7 +495,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
uint64_t AlignedPC = PC & ~0xFFFULL;
// Offset from aligned PC
auto AlignedOffset = std::bit_cast<int64_t>(Constant - AlignedPC);
int64_t AlignedOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(AlignedPC);
int NumMoves = 0;
@@ -512,7 +511,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
} else {
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
// 21-bit signed integer here
auto SmallOffset = std::bit_cast<int64_t>(Constant - PC);
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
if (ARMEmitter::Emitter::IsInt21(SmallOffset)) {
adr(Reg, SmallOffset);
} else {
@@ -695,8 +694,6 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
GPRSpillMask &= ~PFAFSpillMask;
str(REG_CALLRET_SP, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.callret_sp));
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i + 1];
@@ -712,7 +709,7 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
// Now handle PF/AF
if (PFAFSpillMask) {
auto PFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
[[maybe_unused]] auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
LOGMAN_THROW_A_FMT(AFOffset == PFOffset + 4, "PF/AF are together");
@@ -792,8 +789,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
ldr(STATE, TmpReg, CPU_AREA_EMULATOR_DATA_OFFSET);
#endif
ldr(REG_CALLRET_SP, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.callret_sp));
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
// is always static and was almost certainly clobbered.
//
@@ -2,7 +2,7 @@
#pragma once
#include "FEXCore/Utils/EnumUtils.h"
#include "Interface/Core/JIT/Relocations.h"
#include "Interface/Core/ObjectCache/Relocations.h"
#ifdef VIXL_DISASSEMBLER
#include <aarch64/disasm-aarch64.h>
@@ -43,8 +43,6 @@ constexpr bool TMP_ABIARGS = true;
constexpr auto REG_PF = ARMEmitter::Reg::r26;
constexpr auto REG_AF = ARMEmitter::Reg::r27;
constexpr auto REG_CALLRET_SP = ARMEmitter::XReg::x25;
// Vector temporaries
constexpr auto VTMP1 = ARMEmitter::VReg::v0;
constexpr auto VTMP2 = ARMEmitter::VReg::v1;
@@ -63,8 +61,6 @@ constexpr bool TMP_ABIARGS = false;
constexpr auto REG_PF = ARMEmitter::Reg::r9;
constexpr auto REG_AF = ARMEmitter::Reg::r24;
constexpr auto REG_CALLRET_SP = ARMEmitter::XReg::x17;
// Vector temporaries
constexpr auto VTMP1 = ARMEmitter::VReg::v16;
constexpr auto VTMP2 = ARMEmitter::VReg::v17;
@@ -88,8 +84,7 @@ constexpr uint64_t EC_CODE_BITMAP_MAX_ADDRESS = 1ULL << 47;
#endif
// Will force one single instruction block to be generated first if set when entering the JIT filling SRA.
// FillStaticRegs must preserve this
constexpr auto ENTRY_FILL_SRA_SINGLE_INST_REG = TMP2;
constexpr auto ENTRY_FILL_SRA_SINGLE_INST_REG = TMP1;
// Predicate to use in the X87 SVE optimization
constexpr ARMEmitter::PRegister PRED_X87_SVEOPT = ARMEmitter::PReg::p2;
+3 -2
View File
@@ -317,7 +317,7 @@ namespace CPU {
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
RegisterForSignalHandler(std::move(PrevCodeBuffer));
RegisterForSignalHandler(PrevCodeBuffer);
return CurrentCodeBuffer.get();
}
@@ -326,13 +326,14 @@ namespace CPU {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Keep a reference to the old code buffer to delay deallocation
SignalHandlerCodeBuffers.push_back(std::move(CodeBuffer));
SignalHandlerCodeBuffers.push_back(CodeBuffer);
} else {
SignalHandlerCodeBuffers.clear();
}
}
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
fextl::shared_ptr<CodeBuffer> OldCodeBuffer;
auto NewCodeBuffer = CodeBuffers.GetLatest();
if (CurrentCodeBuffer != NewCodeBuffer) {
RegisterForSignalHandler(CurrentCodeBuffer);
+21 -7
View File
@@ -13,14 +13,9 @@ $end_info$
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <FEXCore/fextl/map.h>
#include <cstdint>
namespace FEXCore::CPU {
union Relocation;
}
namespace FEXCore {
namespace IR {
@@ -99,7 +94,15 @@ namespace CPU {
struct CompiledCode {
// Where this code block begins.
uint8_t* BlockBegin;
fextl::map<uint64_t, uint8_t*> EntryPoints;
/**
* The function entrypoint to this codeblock.
*
* This may or may not equal `BlockBegin` above. Depending on the CPU backend, it may stick data
* prior to the BlockEntry.
*
* Is actually a function pointer of type `void (FEXCore::Core::ThreadState *Thread)`
*/
uint8_t* BlockEntry;
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
size_t Size;
};
@@ -161,7 +164,18 @@ namespace CPU {
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() = 0;
/**
* @brief Relocates a block of code from the JIT code object cache
*
* @param Entry - RIP of the entry
* @param SerializationData - Serialization data referring to the object cache for `Entry`
*
* @return An executable function pointer relocated from the cache object
*/
[[nodiscard]]
virtual void* RelocateJITObjectCode(uint64_t /* Entry */, const CodeSerialize::CodeObjectFileSection* /* SerializationData */) {
return nullptr;
}
virtual void ClearCache() {}
+93 -111
View File
@@ -43,15 +43,12 @@ namespace ProductNames {
static const char ARM_A715[] = "Cortex-A715";
static const char ARM_A720[] = "Cortex-A720";
static const char ARM_A725[] = "Cortex-A725";
static const char ARM_C1Pro[] = "C1-Pro";
static const char ARM_C1Premium[] = "C1-Premium";
static const char ARM_X1[] = "Cortex-X1";
static const char ARM_X1C[] = "Cortex-X1C";
static const char ARM_X2[] = "Cortex-X2";
static const char ARM_X3[] = "Cortex-X3";
static const char ARM_X4[] = "Cortex-X4";
static const char ARM_X925[] = "Cortex-X925";
static const char ARM_C1Ultra[] = "C1-Ultra";
static const char ARM_N1[] = "Neoverse N1";
static const char ARM_N2[] = "Neoverse N2";
static const char ARM_N3[] = "Neoverse N3";
@@ -62,7 +59,6 @@ namespace ProductNames {
static const char ARM_A65[] = "Cortex-A65";
static const char ARM_A510[] = "Cortex-A510";
static const char ARM_A520[] = "Cortex-A520";
static const char ARM_C1Nano[] = "C1-Nano";
static const char ARM_Kryo200[] = "Kryo 2xx";
static const char ARM_Kryo300[] = "Kryo 3xx";
@@ -74,7 +70,6 @@ namespace ProductNames {
static const char ARM_Denver[] = "Nvidia Denver";
static const char ARM_Carmel[] = "Nvidia Carmel";
static const char ARM_Olympus[] = "Nvidia Olympus";
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
@@ -90,9 +85,6 @@ namespace ProductNames {
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
static const char ARM_ORYON_1[] = "Oryon-1";
static const char ARM_Ampere_1[] = "AmpereOne";
static const char ARM_Ampere_1A[] = "AmpereOneA";
static const char ARM_Ampere_1B[] = "AmpereOneB";
#else
#endif
} // namespace ProductNames
@@ -178,7 +170,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
// CPU priority order
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
// Relative list so things they will commonly end up in big.little configurations sort of relate
static constexpr std::array<CPUMIDR, 66> CPUMIDRs = {{
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
// Typically big CPU cores
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
@@ -189,46 +181,38 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
{0x41, 0xd8b, 1, ProductNames::ARM_C1Pro}, // C1-Pro
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0xc0, 0xac3, 1, ProductNames::ARM_Ampere_1}, // AmpereOne
{0xc0, 0xac4, 1, ProductNames::ARM_Ampere_1A}, // AmpereOneA
{0xc0, 0xac5, 1, ProductNames::ARM_Ampere_1B}, // AmpereOneB
{0x4e, 0x010, 1, ProductNames::ARM_Olympus}, // Olympus
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
// Denver rated above A57 to match TX2 weirdness
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
@@ -243,7 +227,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
{0x41, 0xd8a, 1, ProductNames::ARM_C1Nano}, // C1-Nano
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
@@ -643,7 +626,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
// Only enable EnhancedREPMOVS if atomic memcpy tso emulation isn't enabled.
const uint32_t SupportsEnhancedREPMOVS = CTX->IsMemcpyAtomicTSOEnabled() == false;
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT;
// Number of subfunctions
Res.eax = 0x0;
@@ -663,39 +645,39 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(1 << 13) | // Deprecates FPU CS and DS
(0 << 14) | // Intel MPX
(0 << 15) | // Intel Resource Directory Technology Allocation
(0 << 16) | // AVX512-F
(0 << 17) | // AVX512-DQ
(0 << 16) | // Reserved
(0 << 17) | // Reserved
(CTX->HostFeatures.SupportsRAND << 18) | // RDSEED
(1 << 19) | // ADCX and ADOX instructions
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
(0 << 21) | // AVX512-IFMA
(0 << 22) | // PCOMMIT (deprecated?)
(0 << 21) | // Reserved
(0 << 22) | // Reserved
(1 << 23) | // CLFLUSHOPT instruction
(1 << 24) | // CLWB instruction
(0 << 25) | // Intel processor trace
(0 << 26) | // AVX512-PF
(0 << 27) | // AVX512-ER
(0 << 28) | // AVX512-CD
(0 << 26) | // Reserved
(0 << 27) | // Reserved
(0 << 28) | // Reserved
(Features.SHA << 29) | // SHA instructions
(0 << 30) | // AVX512-BW
(0 << 31); // AVX512-VL
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.ecx = (1 << 0) | // PREFETCHWT1
(0 << 1) | // AVX512VBMI
(0 << 2) | // Usermode instruction prevention
(0 << 3) | // Protection keys for user mode pages
(0 << 4) | // OS protection keys
(SupportsWFXT << 5) | // waitpkg
(0 << 6) | // AVX512-VBMI2
(0 << 5) | // waitpkg
(0 << 6) | // AVX512_VBMI2
(0 << 7) | // CET shadow stack
(0 << 8) | // GFNI
(CTX->HostFeatures.SupportsAES256 << 9) | // VAES
(SupportsVPCLMULQDQ << 10) | // VPCLMULQDQ
(0 << 11) | // AVX512-VNNI
(0 << 12) | // AVX512-BITALG
(0 << 11) | // AVX512_VNNI
(0 << 12) | // AVX512_BITALG
(0 << 13) | // Intel Total Memory Encryption
(0 << 14) | // AVX512-VPOPCNTDQ
(0 << 15) | // FZM (TDX)
(0 << 14) | // AVX512_VPOPCNTDQ
(0 << 15) | // Reserved
(0 << 16) | // 5 Level page tables
(0 << 17) | // MPX MAWAU
(0 << 18) | // MPX MAWAU
@@ -703,28 +685,28 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 20) | // MPX MAWAU
(0 << 21) | // MPX MAWAU
(1 << 22) | // RDPID Read Processor ID
(0 << 23) | // AES Key Locker
(1 << 24) | // bus-lock-detect
(0 << 23) | // Reserved
(0 << 24) | // Reserved
(0 << 25) | // CLDEMOTE
(0 << 26) | // MPRR (TDX)
(0 << 26) | // Reserved
(0 << 27) | // MOVDIRI
(0 << 28) | // MOVDIR64B
(0 << 29) | // ENQCMD
(0 << 29) | // Reserved
(0 << 30) | // SGX Launch configuration
(0 << 31); // PKS
(0 << 31); // Reserved
Res.edx = (0 << 0) | // SGX-TEM (TDX)
(0 << 1) | // SGX-KEYS
(0 << 2) | // AVX512-4VNNIW
(0 << 3) | // AVX512-4FMAPS
Res.edx = (0 << 0) | // Reserved
(0 << 1) | // Reserved
(0 << 2) | // AVX512_4VNNIW
(0 << 3) | // AVX512_4FMAPS
(1 << 4) | // Fast Short Rep Mov
(0 << 5) | // UINTR
(0 << 5) | // Reserved
(0 << 6) | // Reserved
(0 << 7) | // Reserved
(0 << 8) | // AVX512-VP2INTERSECT
(0 << 8) | // AVX512_VP2INTERSECT
(0 << 9) | // SRBDS_CTRL (Special Register Buffer Data Sampling Mitigations)
(0 << 10) | // VERW clears CPU buffers
(0 << 11) | // rtm-always-abort
(0 << 11) | // Reserved
(0 << 12) | // Reserved
(0 << 13) | // TSX Force Abort (TSX will force abort if attempted)
(0 << 14) | // SERIALIZE instruction
@@ -736,7 +718,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 20) | // Intel CET
(0 << 21) | // Reserved
(0 << 22) | // AMX-BF16 - Tile computation on bfloat16
(0 << 23) | // AVX512-FP16 - FP16 AVX512 instructions
(0 << 23) | // AVX512_FP16 - FP16 AVX512 instructions
(0 << 24) | // AMX-tile - If AMX is implemented
(0 << 25) | // AMX-int8 - AMX on 8-bit integers
(0 << 26) | // IBRS_IBPB - Speculation control
@@ -772,7 +754,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) const {
// XFeatureSupportedMask[63:32]
Res.edx = 0; // Upper 32-bits of XFeatureSupportedMask
} else if (Leaf == 1) {
Res.eax = (1 << 0) | // XSAVEOPT
Res.eax = (0 << 0) | // XSAVEOPT
(0 << 1) | // XSAVEC (and XRSTOR)
(0 << 2) | // XGETBV - XGETBV with ECX=1 supported
(0 << 3); // XSAVES - XSAVES, XRSTORS, and IA32_XSS supported
@@ -942,38 +924,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
(0 << 30) | // Reserved
(0 << 31); // Reserved
Res.edx = (1 << 0) | // FPU
(1 << 1) | // Virtual mode extensions
(1 << 2) | // Debugging extensions
(1 << 3) | // Page size extensions
(1 << 4) | // TSC
(1 << 5) | // MSR support
(1 << 6) | // PAE
(1 << 7) | // Machine Check Exception
(1 << 8) | // CMPXCHG8B
(1 << 9) | // APIC
(0 << 10) | // Reserved
(1 << 11) | // SYSCALL/SYSRET
(1 << 12) | // MTRR
(1 << 13) | // Page global extension
(1 << 14) | // Machine Check architecture
(1 << 15) | // CMOV
(1 << 16) | // Page attribute table
(1 << 17) | // Page-size extensions
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(1 << 20) | // NX
(0 << 21) | // Reserved
(1 << 22) | // MMXExt
(1 << 23) | // MMX
(1 << 24) | // FXSAVE/FXRSTOR
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
(0 << 26) | // 1 gigabit pages
(SUPPORTS_RDTSCP << 27) | // RDTSCP
(0 << 28) | // Reserved
(1 << 29) | // Long Mode
(CTX->HostFeatures.Supports3DNow << 30) | // 3DNow! Extensions
(CTX->HostFeatures.Supports3DNow << 31); // 3DNow!
Res.edx = (1 << 0) | // FPU
(1 << 1) | // Virtual mode extensions
(1 << 2) | // Debugging extensions
(1 << 3) | // Page size extensions
(1 << 4) | // TSC
(1 << 5) | // MSR support
(1 << 6) | // PAE
(1 << 7) | // Machine Check Exception
(1 << 8) | // CMPXCHG8B
(1 << 9) | // APIC
(0 << 10) | // Reserved
(1 << 11) | // SYSCALL/SYSRET
(1 << 12) | // MTRR
(1 << 13) | // Page global extension
(1 << 14) | // Machine Check architecture
(1 << 15) | // CMOV
(1 << 16) | // Page attribute table
(1 << 17) | // Page-size extensions
(0 << 18) | // Reserved
(0 << 19) | // Reserved
(1 << 20) | // NX
(0 << 21) | // Reserved
(1 << 22) | // MMXExt
(1 << 23) | // MMX
(1 << 24) | // FXSAVE/FXRSTOR
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
(0 << 26) | // 1 gigabit pages
(SUPPORTS_RDTSCP << 27) | // RDTSCP
(0 << 28) | // Reserved
(1 << 29) | // Long Mode
(1 << 30) | // 3DNow! Extensions
(1 << 31); // 3DNow!
return Res;
}
@@ -1,27 +0,0 @@
// SPDX-License-Identifier: MIT
#include <Interface/Context/Context.h>
#include <FEXCore/HLE/SourcecodeResolver.h>
namespace FEXCore {
ExecutableFileInfo::~ExecutableFileInfo() = default;
} // namespace FEXCore
namespace FEXCore::Context {
CodeCache::CodeCache(ContextImpl& CTX_)
: CTX(CTX_) {}
CodeCache::~CodeCache() = default;
void CodeCache::LoadData(Core::InternalThreadState& Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& GuestRIPLookup) {
// TODO
}
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
// TODO
return true;
}
} // namespace FEXCore::Context
+197 -173
View File
@@ -14,11 +14,11 @@ $end_info$
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <Interface/GDBJIT/GDBJIT.h>
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
@@ -46,13 +46,13 @@ $end_info$
#include "FEXCore/Utils/SignalScopeGuards.h"
#include <FEXCore/Utils/Threads.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/SHMStats.h>
#include <FEXCore/fextl/fmt.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <FEXHeaderUtils/TodoDefines.h>
#include <algorithm>
#include <array>
@@ -78,7 +78,10 @@ namespace FEXCore::Context {
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
: HostFeatures {Features}
, CPUID {this}
, CodeCache {*this} {
, IRCaptureCache {this} {
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
CodeObjectCacheService = fextl::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
}
if (!Config.Is64BitMode()) {
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
Config.VirtualMemSize = 1ULL << 32;
@@ -102,6 +105,14 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
UpdateAtomicTSOEmulationConfig();
}
ContextImpl::~ContextImpl() {
{
if (CodeObjectCacheService) {
CodeObjectCacheService->Shutdown();
}
}
}
struct GetFrameBlockInfoResult {
const CPU::CPUBackend::JITCodeHeader* InlineHeader;
const CPU::CPUBackend::JITCodeTail* InlineTail;
@@ -128,11 +139,6 @@ bool ContextImpl::IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* T
return InlineTail && InlineTail->SingleInst;
}
uint64_t ContextImpl::GetGuestBlockEntry(FEXCore::Core::InternalThreadState* Thread) {
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
return InlineTail ? InlineTail->RIP : 0;
}
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
const auto Frame = Thread->CurrentFrame;
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
@@ -343,9 +349,36 @@ bool ContextImpl::InitCore() {
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
// Set up the SignalDelegator config since core is initialized.
SignalDelegation->SetConfig(Dispatcher->MakeSignalDelegatorConfig());
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
.DispatcherBegin = Dispatcher->Start,
.DispatcherEnd = Dispatcher->End,
#if defined(_WIN32) && !defined(_M_ARM_64EC)
.AbsoluteLoopTopAddress = Dispatcher->AbsoluteLoopTopAddress,
.AbsoluteLoopTopAddressFillSRA = Dispatcher->AbsoluteLoopTopAddressFillSRA,
.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress,
.SignalHandlerReturnAddressRT = Dispatcher->SignalHandlerReturnAddressRT,
.PauseReturnInstruction = Dispatcher->PauseReturnInstruction,
.ThreadPauseHandlerAddressSpillSRA = Dispatcher->ThreadPauseHandlerAddressSpillSRA,
.ThreadPauseHandlerAddress = Dispatcher->ThreadPauseHandlerAddress,
// Stop handlers.
.ThreadStopHandlerAddressSpillSRA = Dispatcher->ThreadStopHandlerAddressSpillSRA,
.ThreadStopHandlerAddress = Dispatcher->ThreadStopHandlerAddress,
// SRA information.
.SRAGPRCount = Dispatcher->GetSRAGPRCount(),
.SRAFPRCount = Dispatcher->GetSRAFPRCount(),
};
Dispatcher->GetSRAGPRMapping(SignalConfig.SRAGPRMapping);
Dispatcher->GetSRAFPRMapping(SignalConfig.SRAFPRMapping);
// Give this configuration to the SignalDelegator.
SignalDelegation->SetConfig(SignalConfig);
#ifndef _WIN32
#elif !defined(_M_ARM_64EC)
// WOW64 always needs the interrupt fault check to be enabled.
Config.NeedsPendingInterruptFaultCheck = true;
#endif
@@ -365,15 +398,21 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
if (CodeObjectCacheService) {
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
// Use the thread's object cache ref counter for this
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
}
// If it is the parent thread that died then just leave
// TODO: This doesn't make sense when the parent thread doesn't outlive its children
FEX_TODO("This doesn't make sense when the parent thread doesn't outlive its children");
}
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(this);
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
@@ -465,6 +504,12 @@ void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer) {
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
if (CodeObjectCacheService) {
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
// Use the thread's object cache ref counter for this
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
}
if (NewCodeBuffer) {
// Allocate new CodeBuffer + L3 LookupCache and clear L1+L2 caches
Thread->CPUBackend->ClearCache();
@@ -472,7 +517,6 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, boo
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
Thread->LookupCache->ClearCache();
}
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
}
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
@@ -501,7 +545,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
if (Handler != CustomIRHandlers.end()) {
TotalInstructions = 1;
TotalInstructionsLength = 1;
Handler->second.Handler(GuestRIP, Thread->OpDispatcher.get());
std::get<0>(Handler->second)(GuestRIP, Thread->OpDispatcher.get());
HasCustomIR = true;
}
}
@@ -513,15 +557,19 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
bool HadDispatchError {false};
bool HadInvalidInst {false};
Thread->FrontendDecoder->DecodeInstructionsAtEntry(Thread, GuestCode, GuestRIP, MaxInst);
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, MaxInst,
[Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Thread, Start, Length);
}
});
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
auto CodeBlocks = &BlockInfo->Blocks;
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
AreMonoHacksActive() && MonoBackpatcherBlock.load(std::memory_order_relaxed) == GuestRIP);
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount);
const auto GPRSize = Thread->OpDispatcher->GetGPROpSize();
const auto GPRSize = GetGPROpSize();
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
@@ -547,7 +595,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
if (InstsInBlock == 0) {
// Special case for an empty instruction block.
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry - GuestRIP));
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
}
for (size_t i = 0; i < InstsInBlock; ++i) {
@@ -577,12 +625,11 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
Thread->OpDispatcher->_GuestOpcode(InstAddress - GuestRIP);
}
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL || Block.ForceFullSMCDetection) {
auto ExistingCodePtr = reinterpret_cast<uint8_t*>(Block.Entry + BlockInstructionsLength);
auto InstAddressReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
std::array<uint8_t, 0x10> CodeOriginal;
memcpy(CodeOriginal.data(), ExistingCodePtr, DecodedInfo->InstSize);
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CodeOriginal, InstAddressReg, DecodedInfo->InstSize);
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
auto ExistingCodePtr = reinterpret_cast<uint64_t*>(Block.Entry + BlockInstructionsLength);
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(ExistingCodePtr[0], ExistingCodePtr[1],
(uintptr_t)ExistingCodePtr - GuestRIP, DecodedInfo->InstSize);
auto InvalidateCodeCond = Thread->OpDispatcher->CondJump(CodeChanged);
@@ -592,7 +639,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP));
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
@@ -600,21 +647,15 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
}
if (TableInfo && TableInfo->OpcodeDispatcher.OpDispatch) {
auto Fn = TableInfo->OpcodeDispatcher.OpDispatch;
if (TableInfo && TableInfo->OpcodeDispatcher) {
auto Fn = TableInfo->OpcodeDispatcher;
Thread->OpDispatcher->ResetHandledLock();
Thread->OpDispatcher->ResetDecodeFailure();
IR::ForceTSOMode ForceTSO = IR::ForceTSOMode::NoOverride;
if (BlockInForceTSOValidRange) {
if (InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress) {
ForceTSO = IR::ForceTSOMode::ForceEnabled;
} else {
ForceTSO = IR::ForceTSOMode::ForceDisabled;
}
} else if (DecodedInfo->Flags & X86Tables::DecodeFlags::FLAG_FORCE_TSO) {
ForceTSO = IR::ForceTSOMode::ForceEnabled;
}
IR::ForceTSOMode ForceTSO =
BlockInForceTSOValidRange ?
(InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress ? IR::ForceTSOMode::ForceEnabled :
IR::ForceTSOMode::ForceDisabled) :
IR::ForceTSOMode::NoOverride;
Thread->OpDispatcher->SetForceTSO(ForceTSO);
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
if (Thread->OpDispatcher->HadDecodeFailure()) {
@@ -642,11 +683,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
}
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::NOEXEC_INST) {
Thread->OpDispatcher->NoExecOp(DecodedInfo);
} else {
Thread->OpDispatcher->InvalidOp(DecodedInfo);
}
Thread->OpDispatcher->InvalidOp(DecodedInfo);
}
HadInvalidInst = true;
@@ -664,8 +701,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
if (NeedsBlockEnd) {
// We had some instructions. Early exit
Thread->OpDispatcher->ExitFunction(
Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
break;
}
@@ -703,23 +739,37 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
.TotalInstructionsLength = TotalInstructionsLength,
.StartAddr = Thread->FrontendDecoder->DecodedMinAddress,
.Length = Thread->FrontendDecoder->DecodedMaxAddress - Thread->FrontendDecoder->DecodedMinAddress,
.NeedsAddGuestCodeRanges = !HasCustomIR,
};
}
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
// JIT Code object cache lookup
if (CodeObjectCacheService) {
auto CodeCacheEntry = CodeObjectCacheService->FetchCodeObjectFromCache(GuestRIP);
if (CodeCacheEntry) {
auto CompiledCode = Thread->CPUBackend->RelocateJITObjectCode(GuestRIP, CodeCacheEntry);
if (CompiledCode) {
return {
.CompiledCode = CompiledCode,
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
.StartAddr = 0, // Unused
.Length = 0, // Unused
};
}
}
}
if (SourcecodeResolver && Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
if (MappedSection) {
MappedSection->FileInfo.SourcecodeMap = SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, MappedSection->FileInfo.FileId);
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (AOTIRCacheEntry.Entry && !AOTIRCacheEntry.Entry->ContainsCode) {
AOTIRCacheEntry.Entry->SourcecodeMap = SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
}
}
// Generate IR + Meta Info
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, NeedsAddGuestCodeRanges] =
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
if (!IRView) {
return {{}, nullptr, 0, 0, false};
return {nullptr, nullptr, 0, 0};
}
// Attempt to get the CPU backend to compile this code
@@ -730,11 +780,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
if (MaxInst != 1) {
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
Thread->OpDispatcher->DelayedDisownBuffer();
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
.DebugData = nullptr,
.StartAddr = 0,
.Length = 0,
.NeedsAddGuestCodeRanges = false};
return {.CompiledCode = reinterpret_cast<uint8_t*>(Block), .DebugData = nullptr, .StartAddr = 0, .Length = 0};
}
}
@@ -749,11 +795,13 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
Thread->OpDispatcher->DelayedDisownBuffer();
return {
.CompiledCode = std::move(CompiledCode),
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
// In the future with code caching getting wired up, we will pass the rest of the data forward.
// TODO: Pass the data forward when code caching is wired up to this.
.CompiledCode = CompiledCode.BlockEntry,
.DebugData = std::move(DebugData),
.StartAddr = StartAddr,
.Length = Length,
.NeedsAddGuestCodeRanges = NeedsAddGuestCodeRanges,
};
}
@@ -773,8 +821,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
return HostCode;
}
auto [CompiledCode, DebugData, StartAddr, Length, NeedsAddGuestCodeRanges] = CompileCode(Thread, GuestRIP, MaxInst);
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
if (CodePtr == nullptr) {
return 0;
} else if (!DebugData) {
@@ -784,63 +831,58 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
// The core managed to compile the code.
if (Config.BlockJITNaming()) {
auto FragmentBasePtr = CompiledCode.BlockBegin;
auto FragmentBasePtr = reinterpret_cast<uint8_t*>(CodePtr);
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
if (DebugData) {
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
}
}
}
} else {
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, DebugData->HostCodeSize);
}
}
}
}
if (Config.LibraryJITNaming() || Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
if (MappedSection) {
if (Config.LibraryJITNaming()) {
Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, MappedSection->FileInfo.Filename);
}
if (Config.GDBSymbols()) {
GDBJITRegister(MappedSection->FileInfo, MappedSection->FileStartVA, GuestRIP, (uintptr_t)CodePtr, *DebugData);
}
}
// Tell the object cache service to serialize the code if enabled
if (CodeObjectCacheService && Config.CacheObjectCodeCompilation == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_READWRITE && DebugData) {
CodeObjectCacheService->AsyncAddSerializationJob(
fextl::make_unique<CodeSerialize::AsyncJobHandler::SerializationJobData>(CodeSerialize::AsyncJobHandler::SerializationJobData {
.GuestRIP = GuestRIP,
.GuestCodeLength = Length,
.GuestCodeHash = 0,
.HostCodeBegin = CodePtr,
.HostCodeLength = DebugData->HostCodeSize,
.HostCodeHash = 0,
.ThreadJobRefCount = &Thread->ObjectCacheRefCounter,
.Relocations = std::move(*DebugData->Relocations),
}));
}
// Clear any relocations that might have been generated
if (!CodeCache.IsGeneratingCache) {
Thread->CPUBackend->ClearRelocations();
}
Thread->CPUBackend->ClearRelocations();
if (NeedsAddGuestCodeRanges) {
// Track in the guest to host map all entrypoints for all pages the compiled block touches, if any page didn't previously
// contain code, inform the frontend so it can setup SMC detection.
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
for (auto CodePage : BlockInfo->CodePages) {
if (Thread->LookupCache->AddBlockExecutableRange(BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) {
// Early exit
return (uintptr_t)CodePtr;
}
// Insert to lookup cache
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
Thread->LookupCache->AddBlockMapping(GuestAddr, HostAddr);
}
// Pages containing this block are added via AddBlockExecutableRange before each page gets accessed in the frontend
Thread->LookupCache->AddBlockMapping(GuestRIP, CodePtr);
return (uintptr_t)CodePtr;
}
@@ -854,8 +896,7 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
auto [CompiledCode, DebugData, StartAddr, Length, _] = CompileCode(Thread, GuestRIP, 1);
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
if (CodePtr == nullptr) {
return 0;
}
@@ -866,51 +907,46 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
return (uintptr_t)CodePtr;
}
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
uint64_t Start, uint64_t Length) {
// Ensures now-modified mappings aren't cached as being in their previous non-executable state.
// Accessing FrontendDecoder is safe as the thread's code invalidation mutex must be locked here.
Thread->FrontendDecoder->ResetExecutableRangeCache();
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
auto lk = Thread->LookupCache->AcquireLock();
auto& CodePages = Thread->LookupCache->Shared->CodePages;
auto lower = CodePages.lower_bound(Start >> 12);
auto upper = CodePages.upper_bound((Start + Length - 1) >> 12);
auto lower = Thread->LookupCache->CodePages.lower_bound(Start >> 12);
auto upper = Thread->LookupCache->CodePages.upper_bound((Start + Length - 1) >> 12);
for (auto it = lower; it != upper; it++) {
Accumulator.emplace_back(std::move(it->second));
for (auto Address : it->second) {
ContextImpl::ThreadRemoveCodeEntry(Thread, Address);
}
it->second.clear();
}
}
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
InvalidateGuestThreadCodeRange(Thread, Start, Length);
}
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
if (!Thread) {
return;
}
bool InvalidatedAnyEntries = false;
for (const auto& PageEntries : Accumulator) {
for (const auto& Entry : PageEntries) {
if (ContextImpl::ThreadRemoveCodeEntry(Thread, Entry)) {
InvalidatedAnyEntries = true;
}
if (!IsMemoryShared) {
IsMemoryShared = true;
UpdateAtomicTSOEmulationConfig();
if (Config.TSOAutoMigration) {
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
Thread->LookupCache->ClearCache();
}
}
if (InvalidatedAnyEntries) {
// This may cause access violations in the thread on Windows as zeroing is not atomic, this is handled by the frontend
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
}
}
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
uint64_t Start, uint64_t Length) {
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
}
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
void ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
"be unique_locked here");
return Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
}
void ContextImpl::ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
static_cast<ContextImpl*>(Frame->Thread->CTX)->SyscallHandler->InvalidateGuestCodeRange(Frame->Thread, GuestRIP, 1);
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
}
std::optional<CustomIRResult>
@@ -919,7 +955,7 @@ ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandl
std::unique_lock lk(CustomIRMutex);
auto InsertedIterator = CustomIRHandlers.emplace(Entrypoint, CustomIRHandlerEntry {Handler, Creator, Data});
auto InsertedIterator = CustomIRHandlers.emplace(Entrypoint, std::tuple(Handler, Creator, Data));
HasCustomIRHandlers = true;
if (!InsertedIterator.second) {
@@ -943,21 +979,21 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
auto Result = AddCustomIREntrypoint(
Entrypoint,
[this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) {
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0, 0, 0);
auto Block = emit->CreateCodeNode(true, 0);
IRHeader.first->Blocks = emit->WrapNode(Block);
emit->SetCurrentCodeBlock(Block);
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0, 0, 0);
auto Block = emit->CreateCodeNode();
IRHeader.first->Blocks = emit->WrapNode(Block);
emit->SetCurrentCodeBlock(Block);
const auto GPRSize = this->Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
const auto GPRSize = GetGPROpSize();
if (GPRSize == IR::OpSize::i64Bit) {
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
} else {
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
offsetof(Core::CPUState, mm[0][0]));
}
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
if (GPRSize == IR::OpSize::i64Bit) {
IR::Ref R = emit->_StoreRegister(emit->_Constant(Entrypoint), GPRSize);
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
} else {
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->_Constant(Entrypoint)),
offsetof(Core::CPUState, mm[0][0]));
}
emit->_ExitFunction(IR::OpSize::i64Bit, emit->_Constant(GuestThunkEntrypoint));
},
ThunkHandler, (void*)GuestThunkEntrypoint);
@@ -986,36 +1022,24 @@ void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
ForceTSOInstructions.erase(ForceTSOInstructions.lower_bound(Address), ForceTSOInstructions.upper_bound(Address + Size));
}
void ContextImpl::MarkMonoBackpatcherBlock(uint64_t BlockEntry) {
MonoBackpatcherBlock.store(BlockEntry, std::memory_order_relaxed);
}
void ContextImpl::RemoveCustomIREntrypoint(FEXCore::Core::InternalThreadState* Thread, uintptr_t Entrypoint) {
void ContextImpl::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
std::scoped_lock lk(CustomIRMutex);
InvalidateGuestCodeRange(nullptr, Entrypoint, 1);
CustomIRHandlers.erase(Entrypoint);
HasCustomIRHandlers = !CustomIRHandlers.empty();
SyscallHandler->InvalidateGuestCodeRange(Thread, Entrypoint, 1);
}
void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint8_t Size, uint64_t Address, uint64_t Value) {
auto Thread = Frame->Thread;
auto CTX = static_cast<ContextImpl*>(Thread->CTX);
{
auto lk = GuardSignalDeferringSection(CTX->CodeInvalidationMutex, Thread);
IR::AOTIRCacheEntry* ContextImpl::LoadAOTIRCacheEntry(const fextl::string& filename) {
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
return rv;
}
if (Size == 8) {
*reinterpret_cast<uint64_t*>(Address) = Value;
} else if (Size == 4) {
*reinterpret_cast<uint32_t*>(Address) = Value;
} else {
ERROR_AND_DIE_FMT("Unexpected write size for backpatcher: {}", Size);
}
}
CTX->SyscallHandler->InvalidateGuestCodeRange(Thread, Address, Size);
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
}
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
@@ -1,8 +1,7 @@
// SPDX-License-Identifier: MIT
#include "Common/VectorRegType.h"
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/X86HelperGen.h"
@@ -17,18 +16,14 @@
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <CodeEmitter/Emitter.h>
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#endif
#include <array>
#include <bit>
#include <atomic>
#include <condition_variable>
#include <csignal>
#include <cstring>
#include <signal.h>
namespace FEXCore::CPU {
@@ -86,14 +81,11 @@ void Dispatcher::EmitDispatcher() {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::rsp, 0);
str(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
ARMEmitter::ForwardLabel CompileSingleStep;
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
FillStaticRegs();
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
ARMEmitter::BiDirectionalLabel LoopTop {};
ARMEmitter::ForwardLabel CompileSingleStep;
#ifdef _M_ARM_64EC
b(&LoopTop);
@@ -119,59 +111,38 @@ void Dispatcher::EmitDispatcher() {
add(ARMEmitter::Size::i64Bit, StaticRegisters[X86State::REG_RSP], ARMEmitter::Reg::rsp, 0);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, TMP1, 0);
ldr(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
FillSpecialRegs(TMP1, TMP2, false, true);
// As ARM64EC uses this as an entrypoint for both guest calls and host returns, opportunistically try to return
// using the call-ret stack to avoid unbalancing it.
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, REG_CALLRET_SP);
// EC_CALL_CHECKER_PC_REG is REG_PF which isn't touched by any of the above
sub(ARMEmitter::Size::i64Bit, TMP1, EC_CALL_CHECKER_PC_REG, TMP1);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
// If the entry at the TOS is for the target address, pop it and return to the JIT code
add(ARMEmitter::Size::i64Bit, REG_CALLRET_SP, REG_CALLRET_SP, 0x10);
ret(TMP2);
// Enter JIT
#endif
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
ARMEmitter::BiDirectionalLabel FullLookup {};
ARMEmitter::BiDirectionalLabel CallBlock {};
(void)Bind(&LoopTop);
Bind(&LoopTop);
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
// Load in our RIP
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
#ifdef _M_ARM_64EC
// Clobbers TMP1/2
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
ARMEmitter::ForwardLabel l_NotECCode;
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
tbz(TMP1, 0, &l_NotECCode);
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
mov(EC_CALL_CHECKER_PC_REG, RipReg);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
(void)!Bind(&l_NotECCode);
#endif
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
// L1 Cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP1, TMP1, 0);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
br(TMP4);
// L1C check failed, do a full lookup
Bind(&FullLookup);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
@@ -196,7 +167,7 @@ void Dispatcher::EmitDispatcher() {
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
// If page pointer is zero then we have no block
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
@@ -211,11 +182,10 @@ void Dispatcher::EmitDispatcher() {
// If the guest address doesn't match, Compile the block.
sub(TMP2, TMP2, RipReg);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
// Check the host address to see if it matches, else compile the block.
(void)cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
// If we've made it here then we have a real compiled block
{
@@ -301,9 +271,37 @@ void Dispatcher::EmitDispatcher() {
br(TMP1);
}
#ifdef _M_ARM_64EC
// Clobbers TMP1/2
auto EmitECExitCheck = [&]() {
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
ARMEmitter::ForwardLabel l_NotECCode;
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
tbz(TMP1, 0, &l_NotECCode);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
mov(EC_CALL_CHECKER_PC_REG, RipReg);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
Bind(&l_NotECCode);
};
#endif
// Need to create the block
{
(void)Bind(&NoBlock);
Bind(&NoBlock);
#ifdef _M_ARM_64EC
EmitECExitCheck();
#endif
EmitSignalGuardedRegion([&]() {
SpillStaticRegs(TMP1);
@@ -337,7 +335,11 @@ void Dispatcher::EmitDispatcher() {
}
{
(void)Bind(&CompileSingleStep);
Bind(&CompileSingleStep);
#ifdef _M_ARM_64EC
EmitECExitCheck();
#endif
EmitSignalGuardedRegion([&]() {
SpillStaticRegs(TMP1);
@@ -496,10 +498,9 @@ void Dispatcher::EmitDispatcher() {
// load static regs
FillStaticRegs();
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
// Now go back to the regular dispatcher loop
(void)b(&LoopTop);
b(&LoopTop);
}
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
@@ -549,8 +550,8 @@ void Dispatcher::EmitDispatcher() {
FABI_F80_I16_I32_PTR,
FABI_F32_I16_F80_PTR,
FABI_F64_I16_F80_PTR,
FABI_F64_F64_PTR,
FABI_F64_F64_F64_PTR,
FABI_F64_I16_F64_PTR,
FABI_F64_I16_F64_F64_PTR,
FABI_I16_I16_F80_PTR,
FABI_I32_I16_F80_PTR,
FABI_I64_I16_F80_PTR,
@@ -558,7 +559,7 @@ void Dispatcher::EmitDispatcher() {
FABI_F80_I16_F80_PTR,
FABI_F80_I16_F80_F80_PTR,
FABI_F80x2_I16_F80_PTR,
FABI_F64x2_F64_PTR,
FABI_F64x2_I16_F64_PTR,
FABI_I32_I64_I64_V128_V128_I16,
FABI_I32_V128_V128_I16,
}};
@@ -568,15 +569,14 @@ void Dispatcher::EmitDispatcher() {
}
}
(void)Bind(&l_CTX);
Bind(&l_CTX);
dc64(reinterpret_cast<uintptr_t>(CTX));
(void)Bind(&l_Sleep);
Bind(&l_Sleep);
dc64(reinterpret_cast<uint64_t>(SleepThread));
(void)Bind(&l_CompileBlock);
Bind(&l_CompileBlock);
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
dc64(PMFCompileBlock.GetConvertedPointer());
(void)Bind(&l_CompileSingleStep);
Bind(&l_CompileSingleStep);
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
dc64(PMFCompileSingleStep.GetConvertedPointer());
@@ -607,7 +607,6 @@ void Dispatcher::EmitDispatcher() {
#ifdef VIXL_SIMULATOR
void Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.WriteXRegister(1, 0);
Simulator.RunFrom(reinterpret_cast< const vixl::aarch64::Instruction*>(DispatchPtr));
}
@@ -758,7 +757,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
if (!TMP_ABIARGS) {
mov(VABI1.Q(), VTMP1.Q());
fmov(VABI1.D(), VTMP1.D());
}
mov(ARMEmitter::XReg::x1, STATE);
@@ -791,7 +790,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
FillF64Result();
} break;
case FABI_F64_F64_PTR: {
case FABI_F64_I16_F64_PTR: {
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
@@ -801,17 +800,18 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
if (!TMP_ABIARGS) {
fmov(VABI1.D(), VTMP1.D());
}
mov(ARMEmitter::XReg::x0, STATE);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
mov(ARMEmitter::XReg::x1, STATE);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<double, double, uint64_t>(FallbackPointerReg);
GenerateIndirectRuntimeCall<double, uint16_t, double, uint64_t>(FallbackPointerReg);
} else {
blr(FallbackPointerReg);
}
FillF64Result();
} break;
case FABI_F64_F64_F64_PTR: {
case FABI_F64_I16_F64_F64_PTR: {
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
@@ -824,9 +824,10 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
fmov(VABI2.D(), VTMP2.D());
}
mov(ARMEmitter::XReg::x0, STATE);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
mov(ARMEmitter::XReg::x1, STATE);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<double, double, double, uint64_t>(FallbackPointerReg);
GenerateIndirectRuntimeCall<double, uint16_t, double, double, uint64_t>(FallbackPointerReg);
} else {
blr(FallbackPointerReg);
}
@@ -986,7 +987,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
FillF80x2Result();
} break;
case FABI_F64x2_F64_PTR: {
case FABI_F64x2_I16_F64_PTR: {
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
@@ -995,13 +996,14 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
SpillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, TMP3, true);
mov(ARMEmitter::XReg::x0, STATE);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
mov(ARMEmitter::XReg::x1, STATE);
if (!TMP_ABIARGS) {
fmov(VABI1.D(), VTMP1.D());
}
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
// GenerateIndirectRuntimeCall<FEXCore::VectorScalarF64Pair, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
// GenerateIndirectRuntimeCall<FEXCore::VectorScalarF64Pair, uint16_t, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
} else {
blr(FallbackPointerReg);
}
@@ -1103,53 +1105,6 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
}
}
SignalDelegatorConfig Dispatcher::MakeSignalDelegatorConfig() const {
// PF/AF are the final two SRA registers. We only want GPRs
const auto GPRCount = uint16_t(StaticRegisters.size() - 2);
const auto FPRCount = uint16_t(StaticFPRegisters.size());
const auto GetSRAGPRMapping = [GPRCount, this] {
SignalDelegatorConfig::SRAIndexMapping Mapping {};
for (size_t i = 0; i < GPRCount; ++i) {
Mapping[i] = StaticRegisters[i].Idx();
}
return Mapping;
};
const auto GetSRAFPRMapping = [FPRCount, this] {
SignalDelegatorConfig::SRAIndexMapping Mapping {};
for (size_t i = 0; i < FPRCount; ++i) {
Mapping[i] = StaticFPRegisters[i].Idx();
}
return Mapping;
};
return FEXCore::SignalDelegatorConfig {
.DispatcherBegin = Start,
.DispatcherEnd = End,
.AbsoluteLoopTopAddress = AbsoluteLoopTopAddress,
.AbsoluteLoopTopAddressFillSRA = AbsoluteLoopTopAddressFillSRA,
.SignalHandlerReturnAddress = SignalHandlerReturnAddress,
.SignalHandlerReturnAddressRT = SignalHandlerReturnAddressRT,
.PauseReturnInstruction = PauseReturnInstruction,
.ThreadPauseHandlerAddressSpillSRA = ThreadPauseHandlerAddressSpillSRA,
.ThreadPauseHandlerAddress = ThreadPauseHandlerAddress,
// Stop handlers.
.ThreadStopHandlerAddressSpillSRA = ThreadStopHandlerAddressSpillSRA,
.ThreadStopHandlerAddress = ThreadStopHandlerAddress,
// SRA information.
.SRAGPRCount = GPRCount,
.SRAFPRCount = FPRCount,
.SRAGPRMapping = GetSRAGPRMapping(),
.SRAFPRMapping = GetSRAFPRMapping(),
};
}
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl* CTX) {
return fextl::make_unique<Dispatcher>(CTX);
}
@@ -2,18 +2,25 @@
#pragma once
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/fextl/memory.h>
#include <array>
#include <cstddef>
#ifdef VIXL_SIMULATOR
#include <aarch64/simulator-aarch64.h>
#endif
#include <cstdint>
#include <signal.h>
#include <stddef.h>
#include <stack>
#include <tuple>
namespace FEXCore {
struct GuestSigAction;
struct SignalDelegatorConfig;
} // namespace FEXCore
}
namespace FEXCore::Core {
struct CpuStateFrame;
@@ -35,32 +42,6 @@ public:
Dispatcher(FEXCore::Context::ContextImpl* ctx);
~Dispatcher();
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
#ifdef VIXL_SIMULATOR
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
#else
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
DispatchPtr(Frame, false);
}
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
CallbackPtr(Frame, RIP);
}
#endif
SignalDelegatorConfig MakeSignalDelegatorConfig() const;
protected:
FEXCore::Context::ContextImpl* CTX;
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame, bool SingleInst);
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
AsmDispatch DispatchPtr;
JITCallback CallbackPtr;
private:
/**
* @name Dispatch Helper functions
* @{ */
@@ -78,14 +59,62 @@ private:
uint64_t GuestSignal_SIGILL {};
uint64_t GuestSignal_SIGTRAP {};
uint64_t GuestSignal_SIGSEGV {};
uint64_t IntCallbackReturnAddress {};
uint64_t PauseReturnInstruction {};
std::array<uint64_t, FallbackABI::FABI_UNKNOWN> ABIPointers {};
/** @} */
uint64_t Start {};
uint64_t End {};
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
#ifdef VIXL_SIMULATOR
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
#else
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
DispatchPtr(Frame);
}
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
CallbackPtr(Frame, RIP);
}
#endif
uint16_t GetSRAGPRCount() const {
// PF/AF are the final two SRA registers.
// Only return the SRA for GPRs.
return StaticRegisters.size() - 2;
}
uint16_t GetSRAFPRCount() const {
return StaticFPRegisters.size();
}
void GetSRAGPRMapping(uint8_t Mapping[16]) const {
for (size_t i = 0; i < StaticRegisters.size() - 2; ++i) {
Mapping[i] = StaticRegisters[i].Idx();
}
}
void GetSRAFPRMapping(uint8_t Mapping[16]) const {
for (size_t i = 0; i < StaticFPRegisters.size(); ++i) {
Mapping[i] = StaticFPRegisters[i].Idx();
}
}
protected:
FEXCore::Context::ContextImpl* CTX;
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame);
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
AsmDispatch DispatchPtr;
JITCallback CallbackPtr;
private:
// Long division helpers
uint64_t LUDIVHandlerAddress {};
uint64_t LDIVHandlerAddress {};
+115 -314
View File
@@ -9,8 +9,6 @@ $end_info$
#include "Interface/Context/Context.h"
#include "Interface/Core/Frontend.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Core/LookupCache.h"
#include <array>
#include <algorithm>
@@ -23,7 +21,6 @@ $end_info$
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/fextl/set.h>
namespace FEXCore::Frontend {
@@ -67,90 +64,29 @@ static uint32_t MapVEXToReg(uint8_t vvvv, bool HasXMM) {
}
}
Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
: Thread {Thread}
, CTX {static_cast<FEXCore::Context::ContextImpl*>(Thread->CTX)}
, OSABI {CTX->SyscallHandler ? CTX->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN}
, PoolObject {CTX->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
if (ReducedPrecision) {
X87Table = &FEXCore::X86Tables::X87F64Ops;
} else {
X87Table = &FEXCore::X86Tables::X87F80Ops;
}
if (CTX->HostFeatures.SupportsAVX && CTX->HostFeatures.SupportsSVE256) {
VEXTable = &FEXCore::X86Tables::VEXTableOps;
VEXTableGroup = &FEXCore::X86Tables::VEXTableGroupOps;
} else if (CTX->HostFeatures.SupportsAVX) {
VEXTable = &FEXCore::X86Tables::VEXTableOps_AVX128;
VEXTableGroup = &FEXCore::X86Tables::VEXTableGroupOps_AVX128;
}
}
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
// Treat FEX-internal X86 callbacks as always executable
if (EntryPoint == CTX->X86CodeGen.CallbackReturn) {
return true;
}
while (Address < ExecutableRangeBase || Address + Size > ExecutableRangeEnd) {
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, Address);
ExecutableRangeBase = RangeInfo.Base;
ExecutableRangeEnd = RangeInfo.Base + RangeInfo.Size;
ExecutableRangeWritable = RangeInfo.Writable;
if (RangeInfo.Size == 0) {
return false;
}
uint64_t RangeRemainingSize = ExecutableRangeEnd - Address;
if (Size > RangeRemainingSize) {
Size -= RangeRemainingSize;
Address += RangeRemainingSize;
}
}
return true;
}
Decoder::Decoder(FEXCore::Context::ContextImpl* ctx)
: CTX {ctx}
, OSABI {ctx->SyscallHandler ? ctx->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN}
, PoolObject {ctx->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {}
uint8_t Decoder::ReadByte() {
uint8_t Byte = InstStream[InstructionSize];
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
std::optional<uint8_t> Byte = PeekByte(0);
if (!Byte) {
HitNonExecutableRange = true;
// Pretend we read 0, the main decode loop will see HitNonExecutableRange and rollback the instruction.
return 0;
}
Instruction[InstructionSize] = *Byte;
Instruction[InstructionSize] = Byte;
InstructionSize++;
return *Byte;
return Byte;
}
std::optional<uint8_t> Decoder::PeekByte(uint8_t Offset) {
uint64_t ByteAddress = reinterpret_cast<uint64_t>(InstStream + InstructionSize + Offset);
if (CheckRangeExecutable(ByteAddress, 1)) {
return InstStream[InstructionSize + Offset];
} else {
return std::nullopt;
}
uint8_t Decoder::PeekByte(uint8_t Offset) const {
uint8_t Byte = InstStream[InstructionSize + Offset];
return Byte;
}
uint64_t Decoder::ReadData(uint8_t Size) {
LOGMAN_THROW_A_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
uint64_t Res = 0;
uint64_t Address = reinterpret_cast<uint64_t>(InstStream + InstructionSize);
if (CheckRangeExecutable(Address, Size)) {
std::memcpy(&Res, &InstStream[InstructionSize], Size);
} else {
HitNonExecutableRange = true;
// See PeekByte, this specific case may cause some executable memory to read as 0 but it doesn't matter as the entire instruction will be rolled back anyway.
Res = 0;
}
std::memcpy(&Res, &InstStream[InstructionSize], Size);
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
for (size_t i = 0; i < Size; ++i) {
@@ -259,13 +195,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
if (HasSIB) {
FEXCore::X86Tables::SIBDecoded SIB;
if (DecodeInst->Flags & DecodeFlags::FLAG_DECODED_SIB) {
if (DecodeInst->DecodedSIB) {
SIB.Hex = DecodeInst->SIB;
} else {
// Haven't yet grabbed SIB, pull it now
DecodeInst->SIB = ReadByte();
SIB.Hex = DecodeInst->SIB;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_SIB;
DecodeInst->DecodedSIB = true;
}
// If the SIB base is 0b101, aka BP or R13 then we have a 32bit displacement
@@ -331,13 +267,6 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
}
bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options) {
if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) [[unlikely]] {
// Dispatcher Op.
// TODO: Move this in to `NormalOpHeader`, Dispatch tables have a bug currently where some subtables don't inherit flags correctly.
// Can be seen by running FEX asm tests if this is removed.
return NormalOp(&Info->OpcodeDispatcher.Indirect[BlockInfo.Is64BitMode ? 1 : 0], Op);
}
DecodeInst->OP = Op;
DecodeInst->TableInfo = Info;
@@ -357,7 +286,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
uint8_t DestSize {};
const bool HasWideningDisplacement =
(FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST) != 0 ||
(Options.w && BlockInfo.Is64BitMode);
(Options.w && CTX->Config.Is64BitMode);
const bool HasNarrowingDisplacement =
(FEXCore::X86Tables::DecodeFlags::GetOpAddr(DecodeInst->Flags, 0) & FEXCore::X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST) != 0;
@@ -375,21 +304,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
const bool HasMODRM = !!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM);
const bool HasREX = !!(DecodeInst->Flags & DecodeFlags::FLAG_REX_PREFIX);
const bool Has16BitAddressing = !BlockInfo.Is64BitMode && DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
if (Options.w && (Info->Flags & InstFlags::FLAGS_REX_W_0)) {
return false;
} else if (!Options.w && (Info->Flags & InstFlags::FLAGS_REX_W_1)) {
return false;
}
if (Options.L && (Info->Flags & InstFlags::FLAGS_VEX_L_0)) {
return false;
} else if (!Options.L && (Info->Flags & InstFlags::FLAGS_VEX_L_1)) {
return false;
}
const bool UseVEXL = Options.L && !(Info->Flags & InstFlags::FLAGS_VEX_L_IGNORE);
const bool Has16BitAddressing = !CTX->Config.Is64BitMode && DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
// This is used for ModRM register modification
// For both modrm.reg and modrm.rm(when mod == 0b11) when value is >= 0b100
@@ -401,9 +316,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// If we require ModRM and haven't decoded it yet, do it now
// Some instructions have to read modrm upfront, others do it later
if (HasMODRM && !(DecodeInst->Flags & DecodeFlags::FLAG_DECODED_MODRM)) {
if (HasMODRM && !DecodeInst->DecodedModRM) {
DecodeInst->ModRM = ReadByte();
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
}
// New instruction size decoding
@@ -420,7 +335,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_16BIT);
DestSize = 2;
} else if (DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_128BIT) {
if (UseVEXL) {
if (Options.L) {
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_256BIT);
DestSize = 32;
} else {
@@ -436,8 +351,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_16BIT);
DestSize = 2;
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) && (HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
} else if ((HasXMMDst || HasMMDst || CTX->Config.Is64BitMode) &&
(HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_64BIT);
DestSize = 8;
} else {
@@ -452,7 +368,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
} else if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_16BIT) {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
} else if (SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_128BIT) {
if (UseVEXL) {
if (Options.L) {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_256BIT);
} else {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_128BIT);
@@ -464,8 +380,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// See table 1-2. Operand-Size Overrides for this decoding
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) && (HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
} else if ((HasXMMSrc || HasMMSrc || CTX->Config.Is64BitMode) &&
(HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_64BIT);
} else {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_32BIT);
@@ -563,9 +480,6 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
size_t CurrentSrc = 0;
const auto VEXOperand = Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_SRC_MASK;
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_NO_OPERAND && Options.vvvv) {
return false;
}
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) {
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
@@ -634,20 +548,11 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
Literal = static_cast<int32_t>(Literal);
}
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
DecodeInst->Src[CurrentSrc].Data.Literal.SignExtend = true;
}
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
++CurrentSrc;
if (Bytes == 8) [[unlikely]] {
DecodeInst->Src[CurrentSrc].Data.Literal.Size = 4;
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal >> 32;
}
Bytes = 0;
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
}
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
@@ -657,7 +562,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
}
bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op) {
DecodeInst->OPRaw = DecodeInst->OP = Op;
DecodeInst->OP = Op;
DecodeInst->TableInfo = Info;
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
@@ -673,13 +578,10 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
// A normal instruction is the most likely.
if (Info->Type == FEXCore::X86Tables::TYPE_INST) [[likely]] {
return NormalOp(Info, Op);
} else if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) [[unlikely]] {
// Dispatcher Op.
return NormalOp(&Info->OpcodeDispatcher.Indirect[BlockInfo.Is64BitMode ? 1 : 0], Op);
} else if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -696,24 +598,24 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
constexpr uint16_t PF_F2 = 3;
uint16_t PrefixType = PF_NONE;
if (LastEscapePrefix == 0xF3) {
if (DecodeInst->LastEscapePrefix == 0xF3) {
PrefixType = PF_F3;
} else if (LastEscapePrefix == 0xF2) {
} else if (DecodeInst->LastEscapePrefix == 0xF2) {
PrefixType = PF_F2;
} else if (LastEscapePrefix == 0x66) {
} else if (DecodeInst->LastEscapePrefix == 0x66) {
PrefixType = PF_66;
}
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
uint16_t LocalOp = OPD(Info->Type, PrefixType, ModRM.reg);
const FEXCore::X86Tables::X86InstInfo* LocalInfo = &SecondInstGroupOps[LocalOp];
FEXCore::X86Tables::X86InstInfo* LocalInfo = &SecondInstGroupOps[LocalOp];
#undef OPD
if (LocalInfo->Type == FEXCore::X86Tables::TYPE_SECOND_GROUP_MODRM && ModRM.mod == 0b11) {
// Everything in this group is privileged instructions aside from XGETBV
@@ -734,23 +636,18 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
return NormalOp(&(*X87Table)[X87Op], X87Op);
return NormalOp(&X87Ops[X87Op], X87Op);
} else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
if (!VEXTable) {
// AVX not enabled.
return false;
}
uint16_t map_select = 1;
uint16_t pp = 0;
const uint8_t Byte1 = ReadByte();
DecodedHeader options {};
if ((Byte1 & 0b10000000) == 0) {
if (!BlockInfo.Is64BitMode) {
if (!CTX->Config.Is64BitMode) {
return false;
}
@@ -769,18 +666,19 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
options.w = (Byte2 & 0b10000000) != 0;
options.L = (Byte2 & 0b100) != 0;
if ((Byte1 & 0b01000000) == 0) {
if (!BlockInfo.Is64BitMode) {
if (!CTX->Config.Is64BitMode) {
return false;
}
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
}
if (BlockInfo.Is64BitMode && (Byte1 & 0b00100000) == 0) {
if (CTX->Config.Is64BitMode && (Byte1 & 0b00100000) == 0) {
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
}
if (options.w) {
DecodeInst->Flags |= DecodeFlags::FLAG_OPTION_AVX_W;
}
if (!(map_select >= 1 && map_select <= 3)) {
LogMan::Msg::EFmt("We don't understand a map_select of: {}", map_select);
return false;
}
}
@@ -790,13 +688,13 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
Op = OPD(map_select, pp, VEXOp);
#undef OPD
const FEXCore::X86Tables::X86InstInfo* LocalInfo = &(*VEXTable)[Op];
FEXCore::X86Tables::X86InstInfo* LocalInfo = &VEXTableOps[Op];
if (LocalInfo->Type >= FEXCore::X86Tables::TYPE_VEX_GROUP_12 && LocalInfo->Type <= FEXCore::X86Tables::TYPE_VEX_GROUP_17) {
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -804,7 +702,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
Op = OPD(LocalInfo->Type, pp, ModRM.reg);
#undef OPD
return NormalOp(&(*VEXTableGroup)[Op], Op, options);
return NormalOp(&VEXTableGroupOps[Op], Op, options);
} else {
return NormalOp(LocalInfo, Op, options);
}
@@ -818,9 +716,8 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
FEX_UNREACHABLE;
}
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
bool Decoder::DecodeInstruction(uint64_t PC) {
InstructionSize = 0;
LastEscapePrefix = 0;
Instruction.fill(0);
DecodeInst = &DecodedBuffer[DecodedSize];
@@ -842,12 +739,12 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
// Decode ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
const bool Has16BitAddressing = !BlockInfo.Is64BitMode && DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
const bool Has16BitAddressing = !CTX->Config.Is64BitMode && DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
// All 3DNow! instructions have the second argument as the rm handler
// We need to decode it upfront to get the displacement out of the way
@@ -881,7 +778,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
uint16_t LocalOp = (Prefix << 8) | ReadByte();
bool NoOverlay66 = (FEXCore::X86Tables::H0F38TableOps[LocalOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
if (LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
if (DecodeInst->LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather than modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
@@ -897,7 +794,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
constexpr uint16_t PF_3A_REX = (1 << 1);
uint16_t Prefix = PF_3A_NONE;
if (LastEscapePrefix == 0x66) { // Operand Size
if (DecodeInst->LastEscapePrefix == 0x66) { // Operand Size
Prefix = PF_3A_66;
}
@@ -923,17 +820,17 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
if (NoOverlay) { // This section of the table ignores prefix extention
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0xF3) { // REP
} else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0xF2) { // REPNE
} else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
} else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
@@ -949,57 +846,61 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
}
case 0x66: // Operand Size prefix
DecodeInst->Flags |= DecodeFlags::FLAG_OPERAND_SIZE;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
break;
case 0x67: // Address Size override prefix
DecodeInst->Flags |= DecodeFlags::FLAG_ADDRESS_SIZE;
break;
case 0x26: // ES legacy prefix
if (!BlockInfo.Is64BitMode) {
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_ES_PREFIX;
if (!CTX->Config.Is64BitMode) {
DecodeInst->Flags |= DecodeFlags::FLAG_ES_PREFIX;
}
break;
case 0x2E: // CS legacy prefix
if (!BlockInfo.Is64BitMode) {
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_CS_PREFIX;
if (!CTX->Config.Is64BitMode) {
DecodeInst->Flags |= DecodeFlags::FLAG_CS_PREFIX;
}
break;
case 0x36: // SS legacy prefix
if (!BlockInfo.Is64BitMode) {
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_SS_PREFIX;
if (!CTX->Config.Is64BitMode) {
DecodeInst->Flags |= DecodeFlags::FLAG_SS_PREFIX;
}
break;
case 0x3E: // DS legacy prefix
if (!BlockInfo.Is64BitMode) {
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_DS_PREFIX;
// Annoyingly GCC generates NOP ops with these prefixes
// Just ignore them for now
// eg. 66 2e 0f 1f 84 00 00 00 00 00 nop WORD PTR cs:[rax+rax*1+0x0]
if (!CTX->Config.Is64BitMode) {
DecodeInst->Flags |= DecodeFlags::FLAG_DS_PREFIX;
}
break;
break;
case 0xF0: // LOCK prefix
DecodeInst->Flags |= DecodeFlags::FLAG_LOCK;
break;
case 0xF2: // REPNE prefix
DecodeInst->Flags |= DecodeFlags::FLAG_REPNE_PREFIX;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
break;
case 0xF3: // REP prefix
DecodeInst->Flags |= DecodeFlags::FLAG_REP_PREFIX;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
break;
case 0x64: // FS prefix
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_FS_PREFIX;
DecodeInst->Flags |= DecodeFlags::FLAG_FS_PREFIX;
break;
case 0x65: // GS prefix
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_GS_PREFIX;
DecodeInst->Flags |= DecodeFlags::FLAG_GS_PREFIX;
break;
default:
[[likely]] { // Default base table
const X86InstInfo* Info = &FEXCore::X86Tables::BaseOps[Op];
if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) {
Info = &Info->OpcodeDispatcher.Indirect[BlockInfo.Is64BitMode ? 1 : 0];
}
auto Info = &FEXCore::X86Tables::BaseOps[Op];
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
if (!CTX->Config.Is64BitMode) {
return false;
}
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
// Widening displacement
@@ -1030,47 +931,6 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
}
}
}
if (DecodeInst->Dest.IsGPR()) {
return false;
}
return true;
}
Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
// Will be set if DecodeInstructionImpl tries to read non-executable memory
HitNonExecutableRange = false;
bool ErrorDuringDecoding = !DecodeInstructionImpl(PC);
if (ErrorDuringDecoding || HitNonExecutableRange) [[unlikely]] {
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
// Error while decoding instruction. We don't know the table or instruction size
DecodeInst->TableInfo = nullptr;
DecodeInst->InstSize = 0;
return ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST : DecodedBlockStatus::NOEXEC_INST;
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
return DecodedBlockStatus::INVALID_INST;
}
if (CTX->AreMonoHacksActive()) {
// Unity uses a standard SPSC ringbuffer with cached read/write pointers and thread waiting flags at the following
// offsets, which are consistent between 32-bit and 64-bit Unity versions from 2015 onwards.
auto IsKnownAtomicDisplacement = [](uint64_t Displacement) {
return Displacement == 0x80 || Displacement == 0x84 || Displacement == 0xC0 || Displacement == 0xC4;
};
if (DecodeInst->OP == 0x8b && DecodeInst->Src[0].IsGPRIndirect() &&
IsKnownAtomicDisplacement(DecodeInst->Src[0].Data.GPRIndirect.Displacement)) {
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
}
if (DecodeInst->OP == 0x89 && DecodeInst->Dest.IsGPRIndirect() && IsKnownAtomicDisplacement(DecodeInst->Dest.Data.GPRIndirect.Displacement)) {
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
}
}
return DecodedBlockStatus::SUCCESS;
}
void Decoder::BranchTargetInMultiblockRange() {
@@ -1080,30 +940,16 @@ void Decoder::BranchTargetInMultiblockRange() {
// If the RIP setting is conditional AND within our symbol range then it can be considered for multiblock
uint64_t TargetRIP = 0;
const auto GPRSize = GetGPROpSize();
const auto GPRSize = CTX->GetGPROpSize();
bool Conditional = true;
const auto InstEnd = DecodeInst->PC + DecodeInst->InstSize;
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_CALL) {
if (ExecutableRangeWritable && CTX->AreMonoHacksActive()) {
// Mono generated code often contains noreturn calls with garbage following them, and calls are always backpatched
// after CIL compilation leading to n recompiles for a multiblock with n calls. Choose to minimize stutters over
// raw performance and disable tracking past calls for mono generated code.
return;
}
AddBranchTarget(InstEnd);
BlockInfo.EntryPoints.emplace(InstEnd);
return;
}
// Calls are handled above
switch (DecodeInst->OP) {
case 0x70 ... 0x7F: // Conditional JUMP
case 0x80 ... 0x8F: { // More conditional
// Source is a literal
// auto RIPOffset = LoadSource(Op, Op->Src[0], Op->Flags);
// auto RIPTargetConst = Constant(Op->PC + Op->InstSize);
// auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
// Target offset is PC + InstSize + Literal
TargetRIP = InstEnd + DecodeInst->Src[0].Literal();
break;
@@ -1113,6 +959,11 @@ void Decoder::BranchTargetInMultiblockRange() {
TargetRIP = InstEnd + DecodeInst->Src[0].Literal();
Conditional = false;
break;
case 0xE8: // Call - Immediate target, We don't want to inline calls
if (ExternalBranches) {
ExternalBranches->insert(InstEnd);
}
[[fallthrough]];
case 0xC2: // RET imm
case 0xC3: // RET
default: return; break;
@@ -1123,16 +974,10 @@ void Decoder::BranchTargetInMultiblockRange() {
TargetRIP &= 0xFFFFFFFFU;
}
if (Conditional) {
// If we are conditional then a target can be the instruction past the conditional instruction
AddBranchTarget(InstEnd);
}
// If the target RIP is x86 code within the symbol ranges then we are golden
// Forbid distant branches to have the cost code better match the guest code layout, avoiding massive (range-wise) code
// blocks in highly fragmented guest code. Such branches are often not-taken branches to garbage in obfuscated code.
constexpr uint64_t MAX_FORWARD_BRANCH_DIST = FEXCore::Utils::FEX_PAGE_SIZE * 4;
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < std::min(InstEnd + MAX_FORWARD_BRANCH_DIST, SymbolMaxAddress);
// Forbid cross-page branches to both avoid massive (range-wise) code blocks in highly fragmented code and trying to decode unmapped branch targets
bool ValidMultiblockMember =
TargetRIP >= SymbolMinAddress && TargetRIP < std::min(FEXCore::AlignUp(InstEnd, FEXCore::Utils::FEX_PAGE_SIZE), SymbolMaxAddress);
#ifdef _M_ARM_64EC
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
@@ -1143,6 +988,9 @@ void Decoder::BranchTargetInMultiblockRange() {
if (Conditional) {
MaxCondBranchForward = std::max(MaxCondBranchForward, TargetRIP);
MaxCondBranchBackwards = std::min(MaxCondBranchBackwards, TargetRIP);
// If we are conditional then a target can be the instruction past the conditional instruction
AddBranchTarget(InstEnd);
}
AddBranchTarget(TargetRIP);
@@ -1153,60 +1001,6 @@ void Decoder::BranchTargetInMultiblockRange() {
}
}
bool Decoder::IsBranchMonoTailcall(uint64_t NumInstructions) const {
// While the mono call backpatching block can easily be detected due it being the only one to contain SMC-faulting
// atomics, that can't be said for the tailcall jump backpatcher which has changed several times across versions and
// can be partially inlined. To work around this, instead detect the tailcall site itself and force full non-signal-based
// SMC detection for that single block.
if (!ExecutableRangeWritable) {
// We only care about jitted code
return false;
}
// See mini-{amd64,x86}.c in the mono codebase, specifically where METHOD_JUMP patches are emitted.
if (GetGPROpSize() == IR::OpSize::i32Bit) {
// Matches:
// LEAVE
// <none> / NOP / MOV EAX, EAX / LEA EBP, [EBP+0]
// JMP imm32
if (DecodeInst->OP != 0xE9 || NumInstructions < 2) {
return false;
}
auto PrevInst = std::prev(DecodeInst);
if (PrevInst->OP == 0xC9) {
return true;
}
if (NumInstructions < 3 || std::prev(PrevInst)->OP != 0xC9) {
return false;
}
return PrevInst->OP == 0x90 || (PrevInst->OP == 0x8B && PrevInst->ModRM == 0xC0) ||
(PrevInst->OP == 0x8D && PrevInst->ModRM == 0x6D && PrevInst->Src[1].IsLiteral() && PrevInst->Src[1].Literal() == 0);
} else {
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
if (DecodeInst->OPRaw == 0xFF && ModRM.reg == 4 && DecodeInst->Src[0].IsGPR()) {
if (DecodeInst->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
// Found in versions of mono from 2024 onwards - matches:
// REX.W JMP rax
return (DecodeInst->Flags & (DecodeFlags::FLAG_REX_PREFIX | DecodeFlags::FLAG_REX_WIDENING | DecodeFlags::FLAG_REX_XGPR_B |
DecodeFlags::FLAG_REX_XGPR_X | DecodeFlags::FLAG_REX_XGPR_R)) ==
(DecodeFlags::FLAG_REX_PREFIX | DecodeFlags::FLAG_REX_WIDENING);
} else if (NumInstructions > 1 && DecodeInst->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_R11) {
// Found in older versions of mono - match:
// MOV r11, imm64
// JMP r11
auto PrevInst = std::prev(DecodeInst);
return PrevInst->OP == 0xBB && PrevInst->Dest.IsGPR() && PrevInst->Dest.Data.GPR.GPR == FEXCore::X86State::REG_R11;
}
}
}
return false;
}
bool Decoder::InstCanContinue() const {
if (DecodeInst->PC + DecodeInst->InstSize == NextBlockStartAddress) {
return false;
@@ -1217,7 +1011,7 @@ bool Decoder::InstCanContinue() const {
}
uint64_t TargetRIP = 0;
const auto GPRSize = GetGPROpSize();
const auto GPRSize = CTX->GetGPROpSize();
if (DecodeInst->OP == 0xE8) { // Call - immediate target
const uint64_t NextRIP = DecodeInst->PC + DecodeInst->InstSize;
@@ -1269,7 +1063,7 @@ void Decoder::AddBranchTarget(uint64_t Target) {
.Size = BlockIt->Size - SplitOffset,
.NumInstructions = BlockIt->NumInstructions - SplitIdx,
.DecodedInstructions = BlockIt->DecodedInstructions + SplitIdx,
.BlockStatus = BlockIt->BlockStatus,
.HasInvalidInstruction = BlockIt->HasInvalidInstruction,
};
BlockIt->Size = SplitOffset;
@@ -1309,10 +1103,12 @@ const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, u
return _InstStream - EntryPoint + RIP;
}
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst,
std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
BlockInfo.TotalInstructionCount = 0;
BlockInfo.Blocks.clear();
BlocksToDecode.clear();
VisitedBlocks.clear();
// Reset internal state management
DecodedSize = 0;
@@ -1320,15 +1116,9 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
MaxCondBranchBackwards = ~0ULL;
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
// Decode operating mode from thread's CS segment.
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
BlockInfo.Is64BitMode = CSSegment->L == 1;
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
// XXX: Load symbol data
SymbolAvailable = false;
EntryPoint = PC;
BlockInfo.EntryPoints = {PC};
InstStream = _InstStream;
uint64_t TotalInstructions {};
@@ -1344,11 +1134,13 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
DecodedMaxAddress = EntryPoint;
// Entry is a jump target
BlocksToDecode = {PC};
BlocksToDecode.emplace(PC);
uint64_t CurrentCodePage = PC & FEXCore::Utils::FEX_PAGE_MASK;
BlockInfo.CodePages = {CurrentCodePage};
fextl::set<uint64_t> CodePages = {CurrentCodePage};
AddContainedCodePage(PC, CurrentCodePage, FEXCore::Utils::FEX_PAGE_SIZE);
if (MaxInst == 0) {
MaxInst = CTX->Config.MaxInstPerBlock;
@@ -1383,7 +1175,6 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
BlockIt->Entry = RIPToDecode;
BlockIt->Size = 0;
BlockIt->IsEntryPoint = EntryBlock;
uint64_t PCOffset = 0;
uint64_t BlockStartOffset = DecodedSize;
@@ -1405,7 +1196,7 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
auto OpMinPage = OpAddress & FEXCore::Utils::FEX_PAGE_MASK;
auto OpMaxPage = OpMaxAddress & FEXCore::Utils::FEX_PAGE_MASK;
if (!EntryBlock && OpMinPage == OpMaxPage && PeekByte(0).value_or(0) == 0 && PeekByte(1).value_or(0) == 0) [[unlikely]] {
if (!EntryBlock && OpMinPage == OpMaxPage && PeekByte(0) == 0 && PeekByte(1) == 0) [[unlikely]] {
// End the multiblock early if we hit 2 consecutive null bytes (add [rax], al) in the same page with the
// assumption we are most likely trying to explore garbage code.
break;
@@ -1413,17 +1204,31 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
if (OpMinPage != CurrentCodePage) {
CurrentCodePage = OpMinPage;
BlockInfo.CodePages.insert(CurrentCodePage);
CodePages.insert(CurrentCodePage);
}
if (OpMaxPage != CurrentCodePage) {
CurrentCodePage = OpMaxPage;
BlockInfo.CodePages.insert(CurrentCodePage);
CodePages.insert(CurrentCodePage);
}
BlockIt->BlockStatus = DecodeInstruction(OpAddress);
bool ErrorDuringDecoding = !DecodeInstruction(OpAddress);
uint64_t OpEndAddress = OpAddress + DecodeInst->InstSize;
if (ErrorDuringDecoding) [[unlikely]] {
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
BlockIt->HasInvalidInstruction = true;
// Error while decoding instruction. We don't know the table or instruction size
DecodeInst->TableInfo = nullptr;
DecodeInst->InstSize = 0;
} else {
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
auto TableInfo = DecodeInst->TableInfo;
if (!TableInfo || !TableInfo->OpcodeDispatcher) {
BlockIt->HasInvalidInstruction = true;
}
}
DecodedMinAddress = std::min(DecodedMinAddress, OpAddress);
DecodedMaxAddress = std::max(DecodedMaxAddress, OpEndAddress);
@@ -1439,7 +1244,7 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
BlockIt->Size += DecodeInst->InstSize;
// Can not continue this block at all on invalid instruction
if (BlockIt->BlockStatus != DecodedBlockStatus::SUCCESS) [[unlikely]] {
if (BlockIt->HasInvalidInstruction) [[unlikely]] {
if (!EntryBlock) {
// In multiblock configurations, we can early terminate any non-entrypoint blocks with the expectation that this won't get hit.
// Improves compile-times.
@@ -1448,9 +1253,6 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
DecodedSize = BlockStartOffset;
InstStream -= PCOffset;
EraseBlock = true;
} else {
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" : "NoExec", OpAddress);
}
break;
}
@@ -1467,7 +1269,6 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
// If the branch target is within our multiblock range then we can keep going on
// We don't want to short circuit this since we want to calculate our ranges still
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
BlockIt->ForceFullSMCDetection = CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions);
BranchTargetInMultiblockRange();
}
@@ -1491,8 +1292,8 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
BlockInfo.TotalInstructionCount = TotalInstructions;
for (auto& Block : BlockInfo.Blocks) {
Block.IsEntryPoint = BlockInfo.EntryPoints.contains(Block.Entry);
for (auto CodePage : CodePages) {
AddContainedCodePage(PC, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
}
}
+9 -46
View File
@@ -2,54 +2,40 @@
#pragma once
#include "Interface/Core/X86Tables/X86Tables.h"
#include "Interface/IR/IR.h"
#include <FEXCore/Utils/ThreadPoolAllocator.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/vector.h>
#include <array>
#include <cstddef>
#include <cstdint>
#include <optional>
#include <stddef.h>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::HLE {
enum class SyscallOSABI;
}
namespace FEXCore::Frontend {
class Decoder final {
public:
enum class DecodedBlockStatus {
SUCCESS,
INVALID_INST,
NOEXEC_INST,
};
// New Frontend decoding
struct DecodedBlocks final {
uint64_t Entry {};
uint64_t Size {};
uint64_t NumInstructions {};
FEXCore::X86Tables::DecodedInst* DecodedInstructions;
DecodedBlockStatus BlockStatus;
bool IsEntryPoint {};
bool ForceFullSMCDetection {};
bool HasInvalidInstruction {};
};
struct DecodedBlockInformation final {
uint64_t TotalInstructionCount;
bool Is64BitMode {};
fextl::vector<DecodedBlocks> Blocks;
fextl::set<uint64_t> EntryPoints;
fextl::set<uint64_t> CodePages; // Start addresses of all pages touching the block
};
Decoder(FEXCore::Core::InternalThreadState* Thread);
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
Decoder(FEXCore::Context::ContextImpl* ctx);
void DecodeInstructionsAtEntry(const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst,
std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
const DecodedBlockInformation* GetDecodedBlockInfo() const {
return &BlockInfo;
@@ -69,10 +55,6 @@ public:
PoolObject.DelayedDisownBuffer();
}
void ResetExecutableRangeCache() {
ExecutableRangeBase = ExecutableRangeEnd = 0;
}
private:
// To pass any information from instruction prefixes
// down into the actual instruction handling machinery.
@@ -82,23 +64,18 @@ private:
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
};
FEXCore::Core::InternalThreadState* Thread;
FEXCore::Context::ContextImpl* CTX;
const FEXCore::HLE::SyscallOSABI OSABI {};
bool DecodeInstructionImpl(uint64_t PC);
DecodedBlockStatus DecodeInstruction(uint64_t PC);
bool DecodeInstruction(uint64_t PC);
void BranchTargetInMultiblockRange();
bool IsBranchMonoTailcall(uint64_t NumInstructions) const;
bool InstCanContinue() const;
void AddBranchTarget(uint64_t Target);
bool CheckRangeExecutable(uint64_t Address, uint64_t Size);
uint8_t ReadByte();
std::optional<uint8_t> PeekByte(uint8_t Offset);
uint8_t PeekByte(uint8_t Offset) const;
uint64_t ReadData(uint8_t Size);
void SkipBytes(uint8_t Size) {
InstructionSize += Size;
@@ -112,20 +89,11 @@ private:
Utils::PoolBufferWithTimedRetirement<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
size_t DecodedSize {};
uint64_t ExecutableRangeBase {};
uint64_t ExecutableRangeEnd {};
bool ExecutableRangeWritable {};
bool HitNonExecutableRange {};
const uint8_t* InstStream {};
IR::OpSize GetGPROpSize() const {
return BlockInfo.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
}
static constexpr size_t MAX_INST_SIZE = 15;
uint8_t InstructionSize {};
std::array<uint8_t, MAX_INST_SIZE> Instruction;
uint8_t LastEscapePrefix {};
FEXCore::X86Tables::DecodedInst* DecodeInst;
// This is for multiblock data tracking
@@ -154,11 +122,6 @@ private:
&FEXCore::Frontend::Decoder::DecodeModRM_16,
};
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_X87_TABLE_SIZE>* X87Table;
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_TABLE_SIZE>* VEXTable {};
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_GROUP_TABLE_SIZE>* VEXTableGroup {};
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
};
} // namespace FEXCore::Frontend
@@ -6,7 +6,7 @@
#include "Interface/IR/IR.h"
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/SHMStats.h>
#include <FEXCore/Utils/Profiler.h>
namespace FEXCore::CPU {
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
@@ -36,48 +36,18 @@ FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t
return State;
}
FEXCORE_PRESERVE_ALL_ATTR static void HandleX87Exception(const softfloat_state& State, FEXCore::Core::CpuStateFrame* Frame) {
// Check for Invalid Operation exception (bit 0 of X87 status word)
if (State.exceptionFlags & softfloat_flag_invalid) {
Frame->State.flags[FEXCore::X86State::X87FLAG_IE_LOC] = 1;
}
}
// Wrapper for SoftFloat state to handle X87 exceptions
class ScopedSoftFloatState {
public:
FEXCORE_PRESERVE_ALL_ATTR ScopedSoftFloatState(uint16_t FCW, FEXCore::Core::CpuStateFrame* Frame, bool Force80BitPrecision = false)
: State(SoftFloatStateFromFCW(FCW, Force80BitPrecision))
, Frame(Frame) {}
FEXCORE_PRESERVE_ALL_ATTR ~ScopedSoftFloatState() {
HandleX87Exception(State, Frame);
}
// Disable copy and move to ensure RAII semantics
ScopedSoftFloatState(const ScopedSoftFloatState&) = delete;
ScopedSoftFloatState& operator=(const ScopedSoftFloatState&) = delete;
ScopedSoftFloatState(ScopedSoftFloatState&&) = delete;
ScopedSoftFloatState& operator=(ScopedSoftFloatState&&) = delete;
softfloat_state State;
private:
FEXCore::Core::CpuStateFrame* Frame;
};
template<>
struct OpHandlers<IR::OP_F80CVTTO> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle4(uint16_t FCW, float src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(&State.State, src);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(&State, src);
}
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle8(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(&State.State, src);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(&State, src);
}
};
@@ -85,12 +55,12 @@ template<>
struct OpHandlers<IR::OP_F80CMP> {
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
softfloat_state State = SoftFloatStateFromFCW(FCW);
bool eq, lt, nan;
uint64_t ResultFlags = 0;
X80SoftFloat::FCMP(&State.State, Src1, Src2, &eq, &lt, &nan);
X80SoftFloat::FCMP(&State, Src1, Src2, &eq, &lt, &nan);
if (lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
@@ -108,14 +78,14 @@ template<>
struct OpHandlers<IR::OP_F80CVT> {
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToF32(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToF32(&State);
}
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToF64(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToF64(&State);
}
};
@@ -123,26 +93,26 @@ template<>
struct OpHandlers<IR::OP_F80CVTINT> {
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToI16(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToI16(&State);
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToI32(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToI32(&State);
}
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat(src).ToI64(&State.State);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat(src).ToI64(&State);
}
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
auto rv = extF80_to_i32(&State.State, X80SoftFloat(src), softfloat_round_minMag, false);
softfloat_state State = SoftFloatStateFromFCW(FCW);
auto rv = extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
if (rv > INT16_MAX || rv < INT16_MIN) {
///< Indefinite value for 16-bit conversions.
@@ -154,14 +124,14 @@ struct OpHandlers<IR::OP_F80CVTINT> {
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return extF80_to_i32(&State.State, X80SoftFloat(src), softfloat_round_minMag, false);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
}
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return extF80_to_i64(&State.State, X80SoftFloat(src), softfloat_round_minMag, false);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return extF80_to_i64(&State, X80SoftFloat(src), softfloat_round_minMag, false);
}
};
@@ -182,8 +152,8 @@ template<>
struct OpHandlers<IR::OP_F80ROUND> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FRNDINT(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FRNDINT(&State, Src1);
}
};
@@ -191,8 +161,8 @@ template<>
struct OpHandlers<IR::OP_F80F2XM1> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::F2XM1(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::F2XM1(&State, Src1);
}
};
@@ -200,8 +170,8 @@ template<>
struct OpHandlers<IR::OP_F80TAN> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FTAN(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FTAN(&State, Src1);
}
};
@@ -209,8 +179,8 @@ template<>
struct OpHandlers<IR::OP_F80SQRT> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FSQRT(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FSQRT(&State, Src1);
}
};
@@ -218,8 +188,8 @@ template<>
struct OpHandlers<IR::OP_F80SIN> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FSIN(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FSIN(&State, Src1);
}
};
@@ -227,8 +197,8 @@ template<>
struct OpHandlers<IR::OP_F80COS> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FCOS(&State.State, Src1);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FCOS(&State, Src1);
}
};
@@ -236,8 +206,8 @@ template<>
struct OpHandlers<IR::OP_F80SINCOS> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegPairType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return FEXCore::MakeVectorRegPair(X80SoftFloat::FSIN(&State.State, Src1), X80SoftFloat::FCOS(&State.State, Src1));
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return FEXCore::MakeVectorRegPair(X80SoftFloat::FSIN(&State, Src1), X80SoftFloat::FCOS(&State, Src1));
}
};
@@ -261,8 +231,8 @@ template<>
struct OpHandlers<IR::OP_F80ADD> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FADD(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FADD(&State, Src1, Src2);
}
};
@@ -270,8 +240,8 @@ template<>
struct OpHandlers<IR::OP_F80SUB> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FSUB(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FSUB(&State, Src1, Src2);
}
};
@@ -279,8 +249,8 @@ template<>
struct OpHandlers<IR::OP_F80MUL> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FMUL(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FMUL(&State, Src1, Src2);
}
};
@@ -288,8 +258,8 @@ template<>
struct OpHandlers<IR::OP_F80DIV> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
return X80SoftFloat::FDIV(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW);
return X80SoftFloat::FDIV(&State, Src1, Src2);
}
};
@@ -297,8 +267,8 @@ template<>
struct OpHandlers<IR::OP_F80FYL2X> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FYL2X(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FYL2X(&State, Src1, Src2);
}
};
@@ -306,8 +276,8 @@ template<>
struct OpHandlers<IR::OP_F80ATAN> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FATAN(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FATAN(&State, Src1, Src2);
}
};
@@ -315,8 +285,8 @@ template<>
struct OpHandlers<IR::OP_F80FPREM1> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FREM1(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FREM1(&State, Src1, Src2);
}
};
@@ -324,8 +294,8 @@ template<>
struct OpHandlers<IR::OP_F80FPREM> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FREM(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FREM(&State, Src1, Src2);
}
};
@@ -333,14 +303,14 @@ template<>
struct OpHandlers<IR::OP_F80SCALE> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame, true};
return X80SoftFloat::FSCALE(&State.State, Src1, Src2);
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
return X80SoftFloat::FSCALE(&State, Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F64SIN> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return sin(src);
}
@@ -348,7 +318,7 @@ struct OpHandlers<IR::OP_F64SIN> {
template<>
struct OpHandlers<IR::OP_F64COS> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return cos(src);
}
@@ -356,7 +326,7 @@ struct OpHandlers<IR::OP_F64COS> {
template<>
struct OpHandlers<IR::OP_F64SINCOS> {
FEXCORE_PRESERVE_ALL_ATTR static VectorScalarF64Pair handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static VectorScalarF64Pair handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
double sin, cos;
#ifdef _WIN32
@@ -371,7 +341,7 @@ struct OpHandlers<IR::OP_F64SINCOS> {
template<>
struct OpHandlers<IR::OP_F64TAN> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return tan(src);
}
@@ -379,7 +349,7 @@ struct OpHandlers<IR::OP_F64TAN> {
template<>
struct OpHandlers<IR::OP_F64F2XM1> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return exp2(src) - 1.0;
}
@@ -387,7 +357,7 @@ struct OpHandlers<IR::OP_F64F2XM1> {
template<>
struct OpHandlers<IR::OP_F64ATAN> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return atan2(src1, src2);
}
@@ -395,7 +365,7 @@ struct OpHandlers<IR::OP_F64ATAN> {
template<>
struct OpHandlers<IR::OP_F64FPREM> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return fmod(src1, src2);
}
@@ -403,7 +373,7 @@ struct OpHandlers<IR::OP_F64FPREM> {
template<>
struct OpHandlers<IR::OP_F64FPREM1> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return remainder(src1, src2);
}
@@ -411,7 +381,7 @@ struct OpHandlers<IR::OP_F64FPREM1> {
template<>
struct OpHandlers<IR::OP_F64FYL2X> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
return src2 * log2(src1);
}
@@ -419,7 +389,7 @@ struct OpHandlers<IR::OP_F64FYL2X> {
template<>
struct OpHandlers<IR::OP_F64SCALE> {
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
if (src1 == 0.0) { // src1 might be +/- zero
return src1; // this will return negative or positive zero if when appropriate
@@ -434,17 +404,18 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1q, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
X80SoftFloat Src1 = Src1q;
ScopedSoftFloatState State {FCW, Frame};
softfloat_state State = SoftFloatStateFromFCW(FCW);
bool Negative = Src1.Sign;
Src1 = X80SoftFloat::FRNDINT(&State.State, Src1);
Src1 = X80SoftFloat::FRNDINT(&State, Src1);
// Clear the Sign bit
Src1.Sign = 0;
uint64_t Tmp = Src1.ToI64(&State.State);
uint64_t Tmp = Src1.ToI64(&State);
X80SoftFloat Rv;
uint8_t* BCD = reinterpret_cast<uint8_t*>(&Rv);
memset(BCD, 0, 10);
for (size_t i = 0; i < 9; ++i) {
if (Tmp == 0) {
@@ -82,22 +82,24 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle)};
// Double Precision Unary
Info[Core::OPINDEX_F64SIN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle)};
Info[Core::OPINDEX_F64COS] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle)};
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_F64_PTR],
Info[Core::OPINDEX_F64SIN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle)};
Info[Core::OPINDEX_F64COS] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle)};
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_I16_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SINCOS>::handle)};
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_I16_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
// Double Precision Binary
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_F64_F64_PTR],
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle)};
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle)};
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_F64_F64_PTR],
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle)};
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_F64_F64_PTR],
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle)};
// SSE4.2 string instructions
@@ -218,21 +220,21 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
return true; \
}
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
}
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64x2_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64x2_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
}
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_I16_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
}
// Unary
@@ -19,8 +19,8 @@ enum FallbackABI {
FABI_F80_I16_I32_PTR,
FABI_F32_I16_F80_PTR,
FABI_F64_I16_F80_PTR,
FABI_F64_F64_PTR,
FABI_F64_F64_F64_PTR,
FABI_F64_I16_F64_PTR,
FABI_F64_I16_F64_F64_PTR,
FABI_I16_I16_F80_PTR,
FABI_I32_I16_F80_PTR,
FABI_I64_I16_F80_PTR,
@@ -28,7 +28,7 @@ enum FallbackABI {
FABI_F80_I16_F80_PTR,
FABI_F80_I16_F80_F80_PTR,
FABI_F80x2_I16_F80_PTR,
FABI_F64x2_F64_PTR,
FABI_F64x2_I16_F64_PTR,
FABI_I32_I64_I64_V128_V128_I16,
FABI_I32_V128_V128_I16,
FABI_UNKNOWN,
+51 -73
View File
@@ -515,12 +515,6 @@ DEF_OP(AndWithFlags) {
}
}
DEF_OP(AndShift) {
auto Op = IROp->C<IR::IROp_XorShift>();
and_(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
}
DEF_OP(XorShift) {
auto Op = IROp->C<IR::IROp_XorShift>();
@@ -588,7 +582,7 @@ DEF_OP(ShiftFlags) {
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, TMP1, &Done);
cbz(EmitSize, TMP1, &Done);
{
// PF/SF/ZF/OF
if (OpSize >= IR::OpSize::i32Bit) {
@@ -652,7 +646,7 @@ DEF_OP(ShiftFlags) {
msr(ARMEmitter::SystemRegister::NZCV, TMP2);
}
}
(void)Bind(&Done);
Bind(&Done);
// TODO: Make RA less dumb so this can't happen (e.g. with late-kill).
if (PFOutput != PFTemp) {
@@ -669,7 +663,7 @@ DEF_OP(RotateFlags) {
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, Shift, &Done);
cbz(EmitSize, Shift, &Done);
{
// Extract the last bit shifted in to CF
const auto BitSize = IR::OpSizeToSize(Op->Size) * 8;
@@ -701,7 +695,7 @@ DEF_OP(RotateFlags) {
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
}
}
(void)Bind(&Done);
Bind(&Done);
}
DEF_OP(Extr) {
@@ -767,14 +761,14 @@ DEF_OP(PDep) {
// Now, they're copied, so we can start setting Dest (even if it overlaps with
// one of them). Handle early exit case
mov(EmitSize, Dest, 0);
(void)cbz(EmitSize, OrigMask, &Done);
cbz(EmitSize, OrigMask, &Done);
// Setup for first iteration
neg(EmitSize, T0, Mask);
and_(EmitSize, T0, T0, Mask);
// Main loop
(void)Bind(&NextBit);
Bind(&NextBit);
sbfx(EmitSize, T1, Input, 0, 1);
eor(EmitSize, Mask, Mask, T0);
and_(EmitSize, T0, T1, T0);
@@ -782,10 +776,10 @@ DEF_OP(PDep) {
orr(EmitSize, Dest, Dest, T0);
lsr(EmitSize, Input, Input, 1);
and_(EmitSize, T0, Mask, T1);
(void)cbnz(EmitSize, T0, &NextBit);
cbnz(EmitSize, T0, &NextBit);
// All done with nothing to do.
(void)Bind(&Done);
Bind(&Done);
}
}
@@ -821,27 +815,27 @@ DEF_OP(PExt) {
ARMEmitter::BackwardLabel NextBit;
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, Mask, &EarlyExit);
cbz(EmitSize, Mask, &EarlyExit);
mov(EmitSize, MaskReg, Mask);
mov(EmitSize, ValueReg, Input);
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
// Main loop
(void)Bind(&NextBit);
(void)cbz(EmitSize, MaskReg, &Done);
Bind(&NextBit);
cbz(EmitSize, MaskReg, &Done);
clz(EmitSize, BitReg, MaskReg);
lslv(EmitSize, ValueReg, ValueReg, BitReg);
lslv(EmitSize, MaskReg, MaskReg, BitReg);
extr(EmitSize, Dest, Dest, ValueReg, OpSizeBitsM1);
bfc(EmitSize, MaskReg, OpSizeBitsM1, 1);
(void)b(&NextBit);
b(&NextBit);
// Early exit
(void)Bind(&EarlyExit);
Bind(&EarlyExit);
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
// All done with nothing to do.
(void)Bind(&Done);
Bind(&Done);
}
}
@@ -909,7 +903,7 @@ DEF_OP(Div) {
eor(EmitSize, TMP1, TMP1, Upper);
// If the sign bit matches then the result is zero
(void)cbz(EmitSize, TMP1, &Only64Bit);
cbz(EmitSize, TMP1, &Only64Bit);
// Long divide
{
@@ -928,17 +922,17 @@ DEF_OP(Div) {
mov(EmitSize, Remainder, TMP2);
// Skip 64-bit path
(void)b(&LongDIVRet);
b(&LongDIVRet);
}
(void)Bind(&Only64Bit);
Bind(&Only64Bit);
// 64-Bit only
{
sdiv(EmitSize, Quotient, Lower, Divisor);
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
}
(void)Bind(&LongDIVRet);
Bind(&LongDIVRet);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown DIV Size: {}", OpSize); break;
@@ -992,7 +986,7 @@ DEF_OP(UDiv) {
// Check the upper bits for zero
// If the upper bits are zero then we can do a 64-bit divide
(void)cbz(EmitSize, Upper, &Only64Bit);
cbz(EmitSize, Upper, &Only64Bit);
// Long divide
{
@@ -1011,17 +1005,17 @@ DEF_OP(UDiv) {
mov(EmitSize, Remainder, TMP2);
// Skip 64-bit path
(void)b(&LongDIVRet);
b(&LongDIVRet);
}
(void)Bind(&Only64Bit);
Bind(&Only64Bit);
// 64-Bit only
{
udiv(EmitSize, Quotient, Lower, Divisor);
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
}
(void)Bind(&LongDIVRet);
Bind(&LongDIVRet);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUDIV Size: {}", OpSize); break;
@@ -1044,50 +1038,34 @@ DEF_OP(Popcount) {
const auto Dst = GetReg(Node);
const auto Src = GetReg(Op->Src);
if (CTX->HostFeatures.SupportsCSSC) {
switch (OpSize) {
case IR::OpSize::i8Bit:
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i16Bit:
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i32Bit: cnt(ARMEmitter::Size::i32Bit, Dst, Src); break;
case IR::OpSize::i64Bit: cnt(ARMEmitter::Size::i64Bit, Dst, Src); break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
}
} else {
switch (OpSize) {
case IR::OpSize::i8Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
// only use lowest byte
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i16Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// only count two lowest bytes
addp(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i32Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// fmov has zero extended, unused bytes are zero
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i64Bit:
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// fmov has zero extended, unused bytes are zero
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
}
umov<ARMEmitter::SubRegSize::i8Bit>(Dst, VTMP1, 0);
switch (OpSize) {
case IR::OpSize::i8Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
// only use lowest byte
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i16Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// only count two lowest bytes
addp(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i32Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// fmov has zero extended, unused bytes are zero
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
case IR::OpSize::i64Bit:
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), Src);
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
// fmov has zero extended, unused bytes are zero
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
}
umov<ARMEmitter::SubRegSize::i8Bit>(Dst, VTMP1, 0);
}
DEF_OP(FindLSB) {
@@ -1369,12 +1347,12 @@ DEF_OP(VExtractToGPR) {
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
const auto OpSize = IROp->Size;
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
[[maybe_unused]] constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
const auto ElementSizeBits = IR::OpSizeAsBits(Op->Header.ElementSize);
const auto Offset = ElementSizeBits * Op->Index;
const auto Is256Bit = Offset >= SSERegBitSize;
[[maybe_unused]] const auto Is256Bit = Offset >= SSERegBitSize;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetReg(Node);
@@ -33,7 +33,7 @@ void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, false);
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
@@ -63,7 +63,7 @@ void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
auto CurrentCursor = GetCursorAddress<uint8_t*>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
BindOrRestart(&Lit.Loc);
Bind(&Lit.Loc);
dc64(Lit.Lit);
Relocations.emplace_back(Lit.MoveABI);
}
@@ -77,36 +77,39 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
MoveABI.GuestRIPMove.GuestRIP = Constant;
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, false);
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation> Relocations) {
const auto OrigBase = GetBufferBase();
const auto OrigSize = GetBufferSize();
const auto OrigOffset = GetCursorOffset();
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
const char* EntryRelocations) {
size_t DataIndex {};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
SetBuffer(reinterpret_cast<std::uint8_t*>(Code.data()), Code.size_bytes());
for (auto& Reloc : Relocations) {
switch (Reloc.Header.Type) {
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc.NamedSymbolLiteral.Symbol);
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
SetCursorOffset(Reloc.NamedSymbolLiteral.Offset);
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
dc64(Pointer);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(Reloc.NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
@@ -114,27 +117,18 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Co
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
SetBuffer(OrigBase, OrigSize);
SetCursorOffset(OrigOffset);
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(Reloc.GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIPMove.RegisterIndex), Pointer, true);
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
}
}
SetBuffer(OrigBase, OrigSize);
SetCursorOffset(OrigOffset);
return true;
}
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations() {
return std::move(Relocations);
}
} // namespace FEXCore::CPU
+154 -50
View File
@@ -62,27 +62,27 @@ DEF_OP(CASPair) {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
(void)Bind(&LoopTop);
Bind(&LoopTop);
// This instruction sequence must be synced with HandleCASPAL_Armv8.
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
cmp(EmitSize, TMP2, Expected0);
ccmp(EmitSize, TMP3, Expected1, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
stlxp(EmitSize, TMP2, Desired0, Desired1, MemSrc);
(void)cbnz(EmitSize, TMP2, &LoopTop);
cbnz(EmitSize, TMP2, &LoopTop);
mov(EmitSize, Dst0, Expected0);
mov(EmitSize, Dst1, Expected1);
(void)b(&LoopExpected);
b(&LoopExpected);
(void)Bind(&LoopNotExpected);
Bind(&LoopNotExpected);
mov(EmitSize, Dst0, TMP2.R());
mov(EmitSize, Dst1, TMP3.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
(void)Bind(&LoopExpected);
Bind(&LoopExpected);
// Restore
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
@@ -114,7 +114,7 @@ DEF_OP(CAS) {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
if (IROp->Size == IR::OpSize::i8Bit) {
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
@@ -123,18 +123,120 @@ DEF_OP(CAS) {
} else {
cmp(EmitSize, TMP2, Expected);
}
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
stlxr(SubEmitSize, TMP3, Desired, MemSrc);
(void)cbnz(EmitSize, TMP3, &LoopTop);
cbnz(EmitSize, TMP3, &LoopTop);
mov(EmitSize, Dst, Expected);
(void)b(&LoopExpected);
b(&LoopExpected);
(void)Bind(&LoopNotExpected);
Bind(&LoopNotExpected);
mov(EmitSize, Dst, TMP2.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
(void)Bind(&LoopExpected);
Bind(&LoopExpected);
}
}
DEF_OP(AtomicAdd) {
auto Op = IROp->C<IR::IROp_AtomicAdd>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
auto MemSrc = GetReg(Op->Addr);
auto Src = GetReg(Op->Value);
if (CTX->HostFeatures.SupportsAtomics) {
staddl(SubEmitSize, Src, MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
add(EmitSize, TMP2, TMP2, Src);
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
cbnz(EmitSize, TMP2, &LoopTop);
}
}
DEF_OP(AtomicSub) {
auto Op = IROp->C<IR::IROp_AtomicSub>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
auto MemSrc = GetReg(Op->Addr);
auto Src = GetReg(Op->Value);
if (CTX->HostFeatures.SupportsAtomics) {
neg(EmitSize, TMP2, Src);
staddl(SubEmitSize, TMP2, MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
sub(EmitSize, TMP2, TMP2, Src);
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
cbnz(EmitSize, TMP2, &LoopTop);
}
}
DEF_OP(AtomicAnd) {
auto Op = IROp->C<IR::IROp_AtomicAnd>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
auto MemSrc = GetReg(Op->Addr);
auto Src = GetReg(Op->Value);
if (CTX->HostFeatures.SupportsAtomics) {
mvn(EmitSize, TMP2, Src);
stclrl(SubEmitSize, TMP2, MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
and_(EmitSize, TMP2, TMP2, Src);
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
cbnz(EmitSize, TMP2, &LoopTop);
}
}
DEF_OP(AtomicCLR) {
auto Op = IROp->C<IR::IROp_AtomicCLR>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
auto MemSrc = GetReg(Op->Addr);
auto Src = GetReg(Op->Value);
if (CTX->HostFeatures.SupportsAtomics) {
stclrl(SubEmitSize, Src, MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
bic(EmitSize, TMP2, TMP2, Src);
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
cbnz(EmitSize, TMP2, &LoopTop);
}
}
DEF_OP(AtomicOr) {
auto Op = IROp->C<IR::IROp_AtomicOr>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
auto MemSrc = GetReg(Op->Addr);
auto Src = GetReg(Op->Value);
if (CTX->HostFeatures.SupportsAtomics) {
stsetl(SubEmitSize, Src, MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
orr(EmitSize, TMP2, TMP2, Src);
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
cbnz(EmitSize, TMP2, &LoopTop);
}
}
@@ -150,14 +252,29 @@ DEF_OP(AtomicXor) {
steorl(SubEmitSize, Src, MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
eor(EmitSize, TMP2, TMP2, Src);
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
(void)cbnz(EmitSize, TMP2, &LoopTop);
cbnz(EmitSize, TMP2, &LoopTop);
}
}
DEF_OP(AtomicNeg) {
auto Op = IROp->C<IR::IROp_AtomicNeg>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
auto MemSrc = GetReg(Op->Addr);
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
neg(EmitSize, TMP3, TMP2);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
cbnz(EmitSize, TMP4, &LoopTop);
}
DEF_OP(AtomicSwap) {
auto Op = IROp->C<IR::IROp_AtomicSwap>();
const auto OpSize = IROp->Size;
@@ -179,10 +296,10 @@ DEF_OP(AtomicSwap) {
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
stlxr(SubEmitSize, TMP4, Src, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
ubfm(EmitSize, GetReg(Node), TMP2, 0, IR::OpSizeAsBits(OpSize) - 1);
}
}
@@ -199,11 +316,11 @@ DEF_OP(AtomicFetchAdd) {
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
add(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -221,11 +338,11 @@ DEF_OP(AtomicFetchSub) {
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
sub(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -243,11 +360,11 @@ DEF_OP(AtomicFetchAnd) {
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
and_(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -264,11 +381,11 @@ DEF_OP(AtomicFetchCLR) {
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
bic(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -285,11 +402,11 @@ DEF_OP(AtomicFetchOr) {
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
orr(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -306,11 +423,11 @@ DEF_OP(AtomicFetchXor) {
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
eor(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -322,26 +439,13 @@ DEF_OP(AtomicFetchNeg) {
auto MemSrc = GetReg(Op->Addr);
if (CTX->HostFeatures.SupportsAtomics) {
// Use a CAS loop to avoid needing to emulate unaligned LLSC atomics
ldr(SubEmitSize, TMP2, MemSrc);
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
mov(EmitSize, TMP4, TMP2);
neg(EmitSize, TMP3, TMP2);
casal(SubEmitSize, TMP2, TMP3, MemSrc);
sub(EmitSize, TMP3, TMP2, TMP4);
(void)cbnz(EmitSize, TMP3, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
neg(EmitSize, TMP3, TMP2);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
neg(EmitSize, TMP3, TMP2);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
DEF_OP(TelemetrySetValue) {
@@ -359,11 +463,11 @@ DEF_OP(TelemetrySetValue) {
stsetl(ARMEmitter::SubRegSize::i64Bit, TMP1, TMP2);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
}
#endif
}
+66 -173
View File
@@ -58,120 +58,30 @@ DEF_OP(ExitFunction) {
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
#ifdef _M_ARM_64EC
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
LoadConstant(ARMEmitter::Size::i64Bit, EC_CALL_CHECKER_PC_REG, NewRIP);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
} else {
#endif
// In order to support direct branches without constantly hitting the L1 cache, we emit a call to a block linker,
// this will compile the branch target block when it is hit and replace the branch to the linker at the callsite
// with a direct branch to the destination block. Upon invalidation of the target block the backpatch is undone.
//
// In addition, to avoid needing to lookup in the cache for returns and any indirect branch prediction penalty,
// a shadow stack of <GuestReturnRIP, HostReturnPC> pairs is maintained, acting as a first level cache for any
// return operations. As the guest may not balance calls and returns exactly, an exception handler is expected to
// be installed by the frontend, to reset the shadow stack to the middle of its valid bounds on overflow/underflow.
// This shadow stack is also cleared on block invalidation operations or codebuffer switches, to ensure all pointed-to
// host code is always valid.
// This code will be backpatched by Arm64JITCore_ExitFunctionLink, below is an enumeration of all the possible cases.
// Jump thunks are emitted in JIT.cpp after compilation of the entire multiblock.
//
// Call with known return block - unlinked
// 00: adr TMP1, 0xC
// 04: stp RetReg, TMP1, [SpReg, -0x10]!
// 08: bl JmpThunk00
// JmpThunk00:
// 00: b 0x8
// 04: br TMP1
// 08: ldr TMP1, <Shared exit linker>
// 0c: blr TMP1
// 10: HostCode
// 18: GuestRIP
// 20: CallerOffset
//
// Call with known return block after backpatching - linked in branch immediate range
// 00: adr TMP1, 0xC
// 04: stp RetReg, TMP1, [SpReg, -0x10]!
// 08: bl HostCode - MODIFIED
//
// Call with known return block after backpatching - linked out of range
// 00: adr TMP1, 0xC
// 04: stp RetReg, TMP1, [SpReg, -0x10]!
// 08: bl JmpThunk00
// JmpThunk00:
// 00: ldr TMP1, 0x10 - MODIFIED 2nd
// 04: br TMP1
// 08: ldr TMP1, <Shared exit linker>
// 0c: blr TMP1
// 10: HostCode - MODIFIED 1st
// 18: GuestRIP
// 20: CallerOffset
//
// Jump - unlinked
// 00: b JmpThunk00
// JmpThunk00:
// 00: b 0x8
// 04: br TMP1
// 08: ldr TMP1, <Shared exit linker>
// 0c: blr TMP1
// 10: HostCode
// 18: GuestRIP
// 20: CallerOffset
//
// Jump after backpatching - linked in branch immediate range
// 00: b HostCode - MODIFIED
//
// Jump after backpatching - linked out of range
// 00: b JmpThunk00
// JmpThunk00:
// 00: ldr TMP1, 0x10 - MODIFIED 2nd
// 04: br TMP1
// 08: ldr TMP1, <Shared exit linker>
// 0c: blr TMP1
// 10: HostCode - MODIFIED 1st
// 18: GuestRIP
// 20: CallerOffset
// Align to 16 byte to allow atomic patching of the following 16 byte
// of code (excluding the RIP data) on platforms that support LSE2
Align16B();
ARMEmitter::ForwardLabel l_BranchHost;
ARMEmitter::ForwardLabel l_CallReturn;
if (Op->Hint == IR::BranchHint::Call) {
if (!Op->CallReturnBlock.IsInvalid()) {
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
(void)adr(TMP1, &l_CallReturn);
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
} else {
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
}
} else if (Op->Hint == IR::BranchHint::CheckTF) {
ARMEmitter::ForwardLabel TFUnset;
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, NewRIP);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
blr(TMP2);
(void)Bind(&TFUnset);
}
ldr(TMP1, &l_BranchHost);
blr(TMP1);
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
(void)Bind(&l_CallReturn);
Bind(&l_BranchHost);
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
dc64(NewRIP);
#ifdef _M_ARM_64EC
}
#endif
} else {
ARMEmitter::ForwardLabel SkipFullLookup;
auto RipReg = GetReg(Op->NewRIP);
if (Op->Hint == IR::BranchHint::Return) {
// First try to pop from the call-ret stack, otherwise follow the normal path (but ending in a ret)
ldp<ARMEmitter::IndexType::POST>(TMP1, TMP2, REG_CALLRET_SP, 0x10);
sub(TMP1, TMP1, RipReg.X());
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
}
ARMEmitter::ForwardLabel FullLookup;
auto RipReg = GetReg(Op->NewRIP);
// L1 Cache
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
@@ -183,51 +93,36 @@ DEF_OP(ExitFunction) {
ubfiz(ARMEmitter::Size::i64Bit, TMP4, RipReg, 4, 20);
add(TMP1, TMP1, TMP4);
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
// Note: sub+cbnz used over cmp+br to preserve flags.
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
sub(TMP1, TMP1, RipReg.X());
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
br(TMP2);
(void)Bind(&SkipFullLookup);
if (Op->Hint == IR::BranchHint::Call) {
ARMEmitter::ForwardLabel l_CallReturn;
if (!Op->CallReturnBlock.IsInvalid()) {
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
(void)adr(TMP1, &l_CallReturn);
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
} else {
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
}
blr(TMP2);
(void)Bind(&l_CallReturn);
} else if (Op->Hint == IR::BranchHint::Return) {
ret(TMP2);
} else {
br(TMP2);
}
Bind(&FullLookup);
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
br(TMP1);
}
}
DEF_OP(Jump) {
const auto Op = IROp->C<IR::IROp_Jump>();
const auto Target = Op->TargetBlock;
PendingTargetLabel = JumpTarget(Op->TargetBlock);
PendingTargetLabel = &JumpTargets.try_emplace(Target.ID()).first->second;
}
DEF_OP(CondJump) {
auto Op = IROp->C<IR::IROp_CondJump>();
auto TrueTargetLabel = JumpTarget(Op->TrueBlock);
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
if (Op->FromNZCV) {
b_OrRestart(MapCC(Op->Cond), TrueTargetLabel);
b(MapCC(Op->Cond), TrueTargetLabel);
} else {
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
[[maybe_unused]] uint64_t Const;
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
auto Reg = GetReg(Op->Cmp1);
const auto Size = Op->CompareSize == IR::OpSize::i32Bit ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
@@ -237,22 +132,22 @@ DEF_OP(CondJump) {
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
cbz_OrRestart(Size, Reg, TrueTargetLabel);
cbz(Size, Reg, TrueTargetLabel);
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
cbnz_OrRestart(Size, Reg, TrueTargetLabel);
cbnz(Size, Reg, TrueTargetLabel);
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
tbz_OrRestart(Reg, Const, TrueTargetLabel);
tbz(Reg, Const, TrueTargetLabel);
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
tbnz_OrRestart(Reg, Const, TrueTargetLabel);
tbnz(Reg, Const, TrueTargetLabel);
} else {
LOGMAN_THROW_A_FMT(false, "CondJump expected simple condition");
}
}
PendingTargetLabel = JumpTarget(Op->FalseBlock);
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
}
DEF_OP(Syscall) {
@@ -446,50 +341,48 @@ DEF_OP(Thunk) {
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
auto OldCode = Op->CodeOriginal.data();
auto Base = GetReg(Op->Header.Args[0]).X();
const auto* OldCode = (const uint8_t*)&Op->CodeOriginalLow;
int len = Op->CodeLength;
int Offset = 0;
ARMEmitter::ForwardLabel Fail;
int idx = 0;
LoadConstant(ARMEmitter::Size::i64Bit, GetReg(Node), 0);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Entry + Op->Offset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, 1);
const auto Dst = GetReg(Node);
auto EmitCheck = [&](size_t Size, auto&& LoadData) {
while (len >= Size) {
LoadData();
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
len -= Size;
Offset += Size;
}
};
EmitCheck(8, [&]() {
ldr(TMP1, Base, Offset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset));
});
EmitCheck(4, [&]() {
ldr(TMP1.W(), Base, Offset);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset));
});
EmitCheck(2, [&]() {
ldrh(TMP1.W(), Base, Offset);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset));
});
EmitCheck(1, [&]() {
ldrb(TMP1.W(), Base, Offset);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset));
});
ARMEmitter::ForwardLabel End;
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
b_OrRestart(&End);
BindOrRestart(&Fail);
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
BindOrRestart(&End);
while (len >= 8) {
ldr(ARMEmitter::XReg::x2, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint64_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i64Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
len -= 8;
idx += 8;
}
while (len >= 4) {
ldr(ARMEmitter::WReg::w2, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
len -= 4;
idx += 4;
}
while (len >= 2) {
ldrh(TMP3, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint16_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
len -= 2;
idx += 2;
}
while (len >= 1) {
ldrb(TMP3, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint8_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
len -= 1;
idx += 1;
}
}
DEF_OP(ThreadRemoveCodeEntry) {
@@ -1,35 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/fextl/vector.h>
#include <cstdint>
namespace FEXCore::CPU {
union Relocation;
} // namespace FEXCore::CPU
namespace FEXCore::Core {
struct DebugDataSubblock {
uint32_t HostCodeOffset;
uint32_t HostCodeSize;
};
struct DebugDataGuestOpcode {
uint64_t GuestEntryOffset;
ptrdiff_t HostEntryOffset;
};
/**
* @brief Contains debug data for a block of code for later debugger analysis
*
* Needs to remain around for as long as the code could be executed at least
*/
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
fextl::vector<DebugDataSubblock> Subblocks;
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
};
} // namespace FEXCore::Core
@@ -16,7 +16,7 @@ DEF_OP(VAESImc) {
DEF_OP(VAESEnc) {
const auto Op = IROp->C<IR::IROp_VAESEnc>();
const auto OpSize = IROp->Size;
[[maybe_unused]] const auto OpSize = IROp->Size;
const auto Dst = GetVReg(Node);
const auto Key = GetVReg(Op->Key);
@@ -41,7 +41,7 @@ DEF_OP(VAESEnc) {
DEF_OP(VAESEncLast) {
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
const auto OpSize = IROp->Size;
[[maybe_unused]] const auto OpSize = IROp->Size;
const auto Dst = GetVReg(Node);
const auto Key = GetVReg(Op->Key);
@@ -64,7 +64,7 @@ DEF_OP(VAESEncLast) {
DEF_OP(VAESDec) {
const auto Op = IROp->C<IR::IROp_VAESDec>();
const auto OpSize = IROp->Size;
[[maybe_unused]] const auto OpSize = IROp->Size;
const auto Dst = GetVReg(Node);
const auto Key = GetVReg(Op->Key);
@@ -89,7 +89,7 @@ DEF_OP(VAESDec) {
DEF_OP(VAESDecLast) {
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
const auto OpSize = IROp->Size;
[[maybe_unused]] const auto OpSize = IROp->Size;
const auto Dst = GetVReg(Node);
const auto Key = GetVReg(Op->Key);
@@ -322,7 +322,7 @@ DEF_OP(VSha256U1) {
DEF_OP(PCLMUL) {
const auto Op = IROp->C<IR::IROp_PCLMUL>();
const auto OpSize = IROp->Size;
[[maybe_unused]] const auto OpSize = IROp->Size;
const auto Dst = GetVReg(Node);
const auto Src1 = GetVReg(Op->Src1);
+132 -257
View File
@@ -11,12 +11,15 @@ desc: Main glue logic of the arm64 splatter backend
$end_info$
*/
#include "Common/SoftFloat.h"
#include "FEXCore/Utils/Telemetry.h"
#include "FEXCore/Utils/TypeDefines.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/JIT/DebugData.h"
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Utils/MemberFunctionToPointer.h"
@@ -27,16 +30,15 @@ $end_info$
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/LongJump.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <cstdio>
#include <cstring>
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include <stdio.h>
#include <unistd.h>
#include <string.h>
#include <limits>
namespace {
struct DivRem {
@@ -220,7 +222,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
FillF64Result();
} break;
case FABI_F64_F64_PTR: {
case FABI_F64_I16_F64_PTR: {
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
@@ -237,7 +239,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
FillF64Result();
} break;
case FABI_F64x2_F64_PTR: {
case FABI_F64x2_I16_F64_PTR: {
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
@@ -262,7 +264,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
FillF64x2Result(DstLo, DstHi);
} break;
case FABI_F64_F64_F64_PTR: {
case FABI_F64_I16_F64_F64_PTR: {
// Linux Reg/Win32 Reg:
// tmp4 (x4/x13): FallbackHandler
// x30: return
@@ -493,118 +495,59 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
}
}
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record, bool Call) {
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
uintptr_t CallerAddress = JumpThunkStartAddress + Record->CallerOffset;
auto BranchOffset = JumpThunkStartAddress / 4 - CallerAddress / 4;
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
// Emit new 16 bytes of code to a temporary patch, then atomically apply it
__uint128_t Patch;
ARMEmitter::Emitter emit((uint8_t*)&Patch, sizeof(Patch));
emit.ldr(TMP1, 8); // PC-relative value pointing to constant after blr
emit.blr(TMP1);
emit.dc64(Frame->Pointers.Common.ExitFunctionLinker);
// Replace the patched callsite with a branch to the jump thunk.
uint32_t BranchInst = 0;
ARMEmitter::Emitter BranchEmit(reinterpret_cast<uint8_t*>(&BranchInst), 4);
if (Call) {
BranchEmit.bl(BranchOffset);
} else {
BranchEmit.b(BranchOffset);
}
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(CallerAddress)).store(BranchInst, std::memory_order::relaxed);
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(CallerAddress), 4);
auto branch = reinterpret_cast<__uint128_t*>((uintptr_t)Record - 8);
std::atomic_ref<__uint128_t>(*branch).store(Patch, std::memory_order::relaxed);
ARMEmitter::Emitter::ClearICache((void*)branch, sizeof(*branch));
}
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
uint32_t BranchInst = 0;
ARMEmitter::Emitter BranchEmit(reinterpret_cast<uint8_t*>(&BranchInst), 4);
BranchEmit.b(0x8);
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(BranchInst, std::memory_order::relaxed);
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
// No need to reset HostCode here as the exit linker pointer is stored separately, and if the block is relinked it will be updated.
}
uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
auto Thread = Frame->Thread;
auto Lock = Thread->LookupCache->AcquireLock();
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
uintptr_t HostCode {};
auto GuestRip = Record->GuestRIP;
if (TFSet) {
if (!TFSet) {
HostCode = Thread->LookupCache->FindBlock(GuestRip);
}
if (TFSet || !HostCode) {
// If TF is set, the cache must be skipped as different code needs to be generated.
Frame->State.rip = GuestRip;
return Frame->Pointers.Common.DispatcherLoopTop;
} else {
{
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
auto lk_inval =
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
HostCode = Thread->LookupCache->FindBlock(GuestRip);
}
if (!HostCode) {
// Hold a reference to the code buffer, to avoid linking unmapped code if compilation triggers a recreation.
auto CodeBuffer = static_cast<Arm64JITCore*>(Thread->CPUBackend.get())->CurrentCodeBuffer;
HostCode = static_cast<Context::ContextImpl*>(Thread->CTX)->CompileBlock(Frame, GuestRip, 0);
if (Thread->LookupCache->Shared != CodeBuffer->LookupCache.get()) {
return HostCode;
}
}
}
// See ExitFunction in BranchOps.cpp for an assembly level view of the handled cases.
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
uintptr_t CallerAddress = JumpThunkStartAddress + Record->CallerOffset;
auto BranchOffset = HostCode / 4 - CallerAddress / 4;
uintptr_t branch = (uintptr_t)(Record)-8;
LOGMAN_THROW_A_FMT((branch % 16) == 0, "Incorrect alignment for block linking record");
uint32_t ExpectedKnownCallMarkerInst = 0;
ARMEmitter::Emitter ExpectedKnownCallMarkerEmit(reinterpret_cast<uint8_t*>(&ExpectedKnownCallMarkerInst), 4);
ExpectedKnownCallMarkerEmit.adr(TMP1, 0xC);
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
// Lock here is necessary to prevent simultaneous linking and delinking
auto lk = Thread->LookupCache->AcquireLock();
// For non-calls, this would extend into the block's code, however that's fine as an out-of-range adr would never
// be generated avoiding any false positives.
uintptr_t KnownCallMarkerAddr = CallerAddress - 0x8;
uint32_t KnownCallMarkerInst = *reinterpret_cast<uint32_t*>(KnownCallMarkerAddr);
if (ARMEmitter::Emitter::IsInt26(BranchOffset)) {
// Directly patch the callsite with the appropriate branch instruction.
uint32_t BranchInst = 0;
ARMEmitter::Emitter BranchEmit(reinterpret_cast<uint8_t*>(&BranchInst), 4);
if (KnownCallMarkerInst == ExpectedKnownCallMarkerInst) {
BranchEmit.bl(BranchOffset);
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
DirectBlockDelinker(Frame, Record, true);
});
} else {
BranchEmit.b(BranchOffset);
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
DirectBlockDelinker(Frame, Record, false);
});
}
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(CallerAddress)).store(BranchInst, std::memory_order::relaxed);
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(CallerAddress), 4);
auto offset = HostCode / 4 - branch / 4;
if (ARMEmitter::Emitter::IsInt26(offset)) {
// This is the optimal case, where the target can be encoded in a single instruction.
// Atomically patch the code with a relative branch.
const uint32_t Patch = (0b0001'01 << 26) | (offset & ((1u << 26) - 1));
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(branch)).store(Patch, std::memory_order::relaxed);
ARMEmitter::Emitter::ClearICache((void*)branch, 4);
} else {
// This case is common between calls and jumps as the thunk callsite can be left untouched.
std::atomic_ref<uint64_t>(Record->HostCode).store(HostCode, std::memory_order::seq_cst);
// fallback case - do a soft-er link by patching the pointer
std::atomic_ref<uint64_t>(Record->HostBranch).store(HostCode, std::memory_order::seq_cst);
#ifdef _M_ARM_64
// Make memory write visible to other threads reading the same location
asm volatile("dc cvau, %0; dsb ish" : : "r"(Record->HostCode) :);
asm volatile("dc cvau, %0; dsb ish" : : "r"(Record->HostBranch) :);
#endif
uint32_t LdrInst = 0;
ARMEmitter::Emitter LdrEmit(reinterpret_cast<uint8_t*>(&LdrInst), 4);
LdrEmit.ldr(TMP1, reinterpret_cast<uint64_t>(&Record->HostCode) - JumpThunkStartAddress);
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(LdrInst, std::memory_order::relaxed);
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker);
}
// Add de-linking handler
Thread->LookupCache->AddBlockLink(GuestRip, Record, DirectBlockDelinker);
return HostCode;
}
@@ -638,7 +581,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
Common.MonoBackpatcherWrite = reinterpret_cast<uint64_t>(&Context::ContextImpl::MonoBackpatcherWrite);
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
{
@@ -656,7 +598,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
Common.SyscallHandlerFunc = PMF.GetVTableEntry(CTX->SyscallHandler);
}
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore::ExitFunctionLink);
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
// Platform Specific
auto& AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
@@ -667,6 +609,15 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
CurrentCodeBuffer = CodeBuffers.GetLatest();
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
// Setup dynamic dispatch.
if (ParanoidTSO()) {
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
} else {
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
}
}
void Arm64JITCore::EmitDetectionString() {
@@ -730,48 +681,48 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
}
}
void Arm64JITCore::EmitTFCheck() {
ARMEmitter::ForwardLabel l_TFUnset;
ARMEmitter::ForwardLabel l_TFBlocked;
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
if (CheckTF) {
ARMEmitter::ForwardLabel l_TFUnset;
ARMEmitter::ForwardLabel l_TFBlocked;
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
(void)tbz(TMP1, 1, &l_TFBlocked);
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
tbz(TMP1, 1, &l_TFBlocked);
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
.FaultToTopAndGeneratedException = 1,
.Signal = Core::FAULT_SIGTRAP,
.TrapNo = X86State::X86_TRAPNO_DB,
.si_code = 2,
.err_code = 0,
};
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
.FaultToTopAndGeneratedException = 1,
.Signal = Core::FAULT_SIGTRAP,
.TrapNo = X86State::X86_TRAPNO_DB,
.si_code = 2,
.err_code = 0,
};
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
(void)Bind(&l_TFBlocked);
// If TF was blocked for this instruction, unblock it for the next.
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)Bind(&l_TFUnset);
}
Bind(&l_TFBlocked);
// If TF was blocked for this instruction, unblock it for the next.
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
Bind(&l_TFUnset);
}
void Arm64JITCore::EmitSuspendInterruptCheck() {
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
// Trigger a fault if there are any pending interrupts
// Used only for suspend on WIN32 at the moment
@@ -786,58 +737,23 @@ void Arm64JITCore::EmitSuspendInterruptCheck() {
ARMEmitter::ForwardLabel l_NoSuspend;
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
brk(SuspendMagic);
(void)Bind(&l_NoSuspend);
Bind(&l_NoSuspend);
#endif
}
void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF) {
// Get the address of the JITCodeHeader and store in to the core state.
// Two instruction cost, each 1 cycle.
adr_OrRestart(TMP1, &HeaderLabel);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
if (CheckTF) {
EmitTFCheck();
}
if (SpillSlots) {
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) {
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
}
}
}
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
JumpTargets.clear();
uint32_t SSACount = IR->GetSSACount();
this->Entry = Entry;
this->DebugData = DebugData;
this->IR = IR;
RequiresFarARM64Jumps = false;
switch (static_cast<RestartOptions::Control>(FEXCore::LongJump::SetJump(RestartControl.RestartJump))) {
case RestartOptions::Control::Incoming:
// Nothing
break;
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
default: ERROR_AND_DIE_FMT("Unhandled Arm64 restart condition!");
}
uint32_t SSACount = IR->GetSSACount();
JumpTargets.clear();
CallReturnTargets.clear();
PendingJumpThunks.clear();
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
CodeData.EntryPoints.clear();
// Fairly excessive buffer range to make sure we don't overflow
uint32_t BufferRange = 0x1000 + SSACount * 24;
uint32_t BufferRange = 0x100 + SSACount * 24;
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
// This minimizes lock contention of CodeBufferWriteMutex.
@@ -848,11 +764,13 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Put the code header at the start of the data block.
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
(void)Bind(&JITCodeHeaderLabel);
Bind(&JITCodeHeaderLabel);
JITCodeHeader* CodeHeader = GetCursorAddress<JITCodeHeader*>();
CursorIncrement(sizeof(JITCodeHeader));
auto CodeBegin = GetCursorAddress<uint8_t*>();
#ifdef VIXL_DISASSEMBLER
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
#endif
// AAPCS64
// r30 = LR
@@ -874,60 +792,49 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// X1-X3 = Temp
// X4-r18 = RA
CodeData.BlockEntry = GetCursorAddress<uint8_t*>();
// Get the address of the JITCodeHeader and store in to the core state.
// Two instruction cost, each 1 cycle.
adr(TMP1, &JITCodeHeaderLabel);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
EmitInterruptChecks(CheckTF);
SpillSlots = IR->SpillSlots();
if (SpillSlots) {
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) {
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
}
}
PendingTargetLabel = nullptr;
PendingCallReturnTargetLabel = nullptr;
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
using namespace FEXCore::IR;
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
#endif
auto BlockStartHostCode = GetCursorAddress<uint8_t*>();
{
const auto Node = IR->GetID(BlockNode);
const auto Target = &JumpTargets[BlockIROp->ID];
const auto IsTarget = JumpTargets.try_emplace(Node).first;
// if there's a pending branch, and it is not fall-through
if (PendingTargetLabel && PendingTargetLabel != Target) {
if (PendingTargetLabel->Backward.Location) {
EmitSuspendInterruptCheck();
}
b_OrRestart(PendingTargetLabel);
PendingTargetLabel = nullptr;
}
if (BlockIROp->EntryPoint) {
uint64_t BlockStartRIP = Entry + BlockIROp->GuestEntryOffset;
const auto IsReturnTarget = CallReturnTargets.try_emplace(Node).first;
if (PendingTargetLabel) {
// If there is a fallthrough branch to this block, skip over the entrypoint code.
b_OrRestart(Target);
} else if (PendingCallReturnTargetLabel && PendingCallReturnTargetLabel != &IsReturnTarget->second) {
// If we just emitted a call, but the block we're now emitting is not the return block so don't fallthrough.
b_OrRestart(PendingCallReturnTargetLabel);
}
PendingCallReturnTargetLabel = nullptr;
BindOrRestart(&IsReturnTarget->second);
CodeData.EntryPoints.emplace(BlockStartRIP, GetCursorAddress<uint8_t*>());
DebugData->GuestOpcodes.push_back({BlockIROp->GuestEntryOffset, GetCursorAddress<uint8_t*>() - CodeData.BlockBegin});
EmitEntryPoint(JITCodeHeaderLabel, CheckTF);
}
if (PendingCallReturnTargetLabel) {
// If there is still a pending call return target, then the block we're emitting is not the return block so don't fallthrough.
b_OrRestart(PendingCallReturnTargetLabel);
PendingCallReturnTargetLabel = nullptr;
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second) {
b(PendingTargetLabel);
}
PendingTargetLabel = nullptr;
BindOrRestart(Target);
Bind(&IsTarget->second);
}
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
@@ -945,45 +852,18 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
}
}
DebugData->Subblocks.push_back({static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockBegin),
DebugData->Subblocks.push_back({static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockEntry),
static_cast<uint32_t>(GetCursorAddress<uint8_t*>() - BlockStartHostCode)});
}
// Make sure last branch is generated. It certainly can't be eliminated here.
if (PendingTargetLabel) {
if (PendingTargetLabel->Backward.Location) {
EmitSuspendInterruptCheck();
}
b_OrRestart(PendingTargetLabel);
b(PendingTargetLabel);
}
PendingTargetLabel = nullptr;
ARMEmitter::ForwardLabel l_ExitLink;
for (auto& PendingJumpThunk : PendingJumpThunks) {
// Align as 64-bit atomics are used on the HostCode field.
Align(8);
ARMEmitter::ForwardLabel l_DoLink;
uint64_t ThunkAddress = GetCursorAddress<uint64_t>();
BindOrRestart(&PendingJumpThunk.Label);
b_OrRestart(&l_DoLink);
br(TMP1);
BindOrRestart(&l_DoLink);
ldr(TMP1, &l_ExitLink);
blr(TMP1);
// This is a ExitFunctionLinkData struct
BindOrRestart(&l_ExitLink);
dc64(0); // HostCode
dc64(PendingJumpThunk.GuestRIP); // GuestRIP
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
}
BindOrRestart(&l_ExitLink);
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
// CodeSize not including the header or tail data.
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeBegin;
// CodeSize not including the tail data.
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
// Add the JitCodeTail
Align(alignof(JITCodeTail));
@@ -1062,7 +942,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
"doesn't match up!\n");
if (auto Prev = CheckCodeBufferUpdate()) {
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
}
@@ -1081,10 +960,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Adjust host addresses
const auto Delta = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
CodeData.BlockBegin += Delta;
for (auto& EntryPoint : CodeData.EntryPoints) {
EntryPoint.second += Delta;
}
CodeBegin += Delta;
CodeData.BlockEntry += Delta;
// Copy over CodeBuffer contents
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
@@ -1095,7 +971,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
TempAllocator.DelayedDisownBuffer();
ClearICache(CodeBegin, CodeOnlySize);
ClearICache(CodeData.BlockBegin, CodeOnlySize);
#ifdef VIXL_DISASSEMBLER
if (Disassemble() & FEXCore::Config::Disassemble::STATS) {
@@ -1109,8 +985,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
}
if (Disassemble() & FEXCore::Config::Disassemble::BLOCKS) {
const auto DisasmBegin = reinterpret_cast<const vixl::aarch64::Instruction*>(CodeBegin);
const auto DisasmEnd = reinterpret_cast<const vixl::aarch64::Instruction*>(CodeBegin + CodeOnlySize);
const auto DisasmEnd = reinterpret_cast<const vixl::aarch64::Instruction*>(JITBlockTailLocation);
LogMan::Msg::IFmt("Disassemble Begin");
for (auto PCToDecode = DisasmBegin; PCToDecode < DisasmEnd; PCToDecode += 4) {
DisasmDecoder->Decode(PCToDecode);
@@ -1126,7 +1001,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
this->IR = nullptr;
return std::move(CodeData);
return CodeData;
}
void Arm64JITCore::ResetStack() {
+15 -245
View File
@@ -19,13 +19,11 @@ $end_info$
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <FEXCore/Utils/LongJump.h>
#include <CodeEmitter/Emitter.h>
#include <array>
#include <cstdint>
#include <functional>
#include <utility>
#include <variant>
@@ -33,10 +31,6 @@ namespace FEXCore::Core {
struct InternalThreadState;
}
namespace FEXCore::Context {
struct ExitFunctionLinkData;
}
namespace FEXCore::CPU {
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
public:
@@ -55,7 +49,6 @@ public:
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
const bool HostSupportsSVE128 {};
const bool HostSupportsSVE256 {};
@@ -63,46 +56,16 @@ private:
const bool HostSupportsRPRES {};
const bool HostSupportsAFP {};
struct RestartOptions {
FEXCore::LongJump::JumpBuf RestartJump;
enum class Control : uint64_t {
Incoming = 0,
EnableFarARM64Jumps = 1,
};
};
// FEXCore makes assumptions in the JIT about certain conditions being true.
// In the rare case when those assumptions are broken, FEX needs to safely restart the JIT.
RestartOptions RestartControl {};
bool RequiresFarARM64Jumps {};
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
ARMEmitter::BiDirectionalLabel* PendingCallReturnTargetLabel {};
FEXCore::Context::ContextImpl* CTX {};
const FEXCore::IR::IRListView* IR {};
uint64_t Entry {};
CPUBackend::CompiledCode CodeData {};
fextl::vector<ARMEmitter::BiDirectionalLabel> JumpTargets;
ARMEmitter::BiDirectionalLabel* JumpTarget(IR::OrderedNodeWrapper Node) {
auto Block = IR->GetOp<IR::IROp_CodeBlock>(Node);
return &JumpTargets[Block->ID];
}
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> CallReturnTargets;
struct PendingJumpThunk {
uint64_t CallerAddress;
uint64_t GuestRIP;
ARMEmitter::ForwardLabel Label;
};
fextl::vector<PendingJumpThunk> PendingJumpThunks;
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempAllocator;
static uint64_t ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
[[nodiscard]]
ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const {
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
@@ -260,10 +223,10 @@ private:
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_FU:
case FEXCore::IR::COND_VS: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FNU:
case FEXCore::IR::COND_VC: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
@@ -327,202 +290,6 @@ private:
uint32_t End;
};
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
auto& Thunk = PendingJumpThunks.back();
BindOrRestart(&Thunk.Label);
if (Call) {
bl_OrRestart(&Thunk.Label);
} else {
b_OrRestart(&Thunk.Label);
}
}
// Restart helpers
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void bl_OrRestart(T* Label) {
if (bl(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void b_OrRestart(T* Label) {
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void b_OrRestart(ARMEmitter::Condition Cond, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)b(InvertCondition(Cond), &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (b(Cond, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void cbz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)cbnz(s, rt, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (cbz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void cbnz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)cbz(s, rt, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (cbnz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void tbz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)tbnz(rt, Bit, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (tbz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void tbnz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)tbz(rt, Bit, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (tbnz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void adr_OrRestart(ARMEmitter::Register rd, T* Label) {
if (adr(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Long ADR currently unsupported!");
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void adrp_OrRestart(ARMEmitter::Register rd, T* Label) {
if (adrp(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Long ADRP currently unsupported!");
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void BindOrRestart(T* Label) {
if (Bind(Label)) {
return;
}
if (RequiresFarARM64Jumps) {
// This should have been caught before this point.
ERROR_AND_DIE_FMT("Oops. Unhandled long bind.");
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
IR::RegisterAllocationPass* RAPass {};
@@ -581,9 +348,7 @@ private:
fextl::vector<FEXCore::CPU::Relocation> Relocations;
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation>);
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() override;
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
/** @} */
@@ -608,14 +373,19 @@ private:
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
void EmitTFCheck();
void EmitInterruptChecks(bool CheckTF);
void EmitSuspendInterruptCheck();
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
// Runtime selection;
// Load and store TSO memory style
OpType RT_LoadMemTSO;
OpType RT_StoreMemTSO;
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
// Dynamic Dispatcher supporting operations
DEF_OP(ParanoidLoadMemTSO);
DEF_OP(ParanoidStoreMemTSO);
///< Unhandled handler
DEF_OP(Unhandled);
+240 -104
View File
@@ -140,7 +140,7 @@ DEF_OP(LoadRegister) {
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
} else if (Op->Class == IR::FPRClass) {
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
@@ -181,7 +181,7 @@ DEF_OP(StoreRegister) {
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
mov(ARMEmitter::Size::i64Bit, GetReg(Reg), GetReg(Op->Value));
} else if (Reg.Class == IR::FPRFixedClass) {
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
const auto guest = GetVReg(Reg);
@@ -348,25 +348,6 @@ DEF_OP(StoreContextIndexed) {
}
}
DEF_OP(FormContextAddress) {
const auto Op = IROp->C<IR::IROp_FormContextAddress>();
const auto Index = GetReg(Op->Index);
const auto Dst = GetReg(Node);
switch (Op->Stride) {
case 1:
case 2:
case 4:
case 8:
case 16:
case 32: {
add(ARMEmitter::Size::i64Bit, Dst, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled FormContextAddress stride: {}", Op->Stride); break;
}
}
DEF_OP(SpillRegister) {
const auto Op = IROp->C<IR::IROp_SpillRegister>();
const auto OpSize = IROp->Size;
@@ -761,7 +742,7 @@ DEF_OP(LoadMemTSO) {
const auto Dst = GetReg(Node);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
bool IsInline = IsInlineConstant(Op->Offset, &Offset);
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
}
@@ -776,10 +757,8 @@ DEF_OP(LoadMemTSO) {
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
}
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
@@ -793,10 +772,8 @@ DEF_OP(LoadMemTSO) {
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
}
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
@@ -810,10 +787,8 @@ DEF_OP(LoadMemTSO) {
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
}
} else {
const auto Dst = GetVReg(Node);
@@ -917,7 +892,7 @@ DEF_OP(VLoadVectorMasked) {
// If the sign bit is zero then skip the load
ARMEmitter::ForwardLabel Skip {};
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
// Do the gather load for this element into the destination
switch (IROp->ElementSize) {
case IR::OpSize::i8Bit: ld1<ARMEmitter::SubRegSize::i8Bit>(TempDst.Q(), i, TempMemReg); break;
@@ -928,7 +903,7 @@ DEF_OP(VLoadVectorMasked) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
}
(void)Bind(&Skip);
Bind(&Skip);
if ((i + 1) != NumElements) {
// Handle register rename to save a move.
@@ -1018,7 +993,7 @@ DEF_OP(VStoreVectorMasked) {
// If the sign bit is zero then skip the load
ARMEmitter::ForwardLabel Skip {};
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
// Do the gather load for this element into the destination
switch (IROp->ElementSize) {
case IR::OpSize::i8Bit: st1<ARMEmitter::SubRegSize::i8Bit>(RegData.Q(), i, TempMemReg); break;
@@ -1029,7 +1004,7 @@ DEF_OP(VStoreVectorMasked) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
}
(void)Bind(&Skip);
Bind(&Skip);
if ((i + 1) != NumElements) {
// Handle register rename to save a move.
@@ -1107,7 +1082,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
PerformMove(ElementSize, WorkingReg, MaskReg, i);
// Skip if the mask's sign bit isn't set
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
// Extract Index Element
if ((IndexElement * IR::OpSizeToSize(VectorIndexSize)) >= 16) {
@@ -1145,7 +1120,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, ElementSize); FEX_UNREACHABLE;
}
(void)Bind(&Skip);
Bind(&Skip);
}
if (NeedsDestTmp) {
@@ -1775,7 +1750,7 @@ DEF_OP(StoreMemTSO) {
const auto Src = GetZeroableReg(Op->Value);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
bool IsInline = IsInlineConstant(Op->Offset, &Offset);
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
}
@@ -1783,10 +1758,8 @@ DEF_OP(StoreMemTSO) {
// 8bit load is always aligned to natural alignment
stlurb(Src, MemReg, Offset);
} else {
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
switch (OpSize) {
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
@@ -1801,10 +1774,8 @@ DEF_OP(StoreMemTSO) {
// 8bit load is always aligned to natural alignment
stlrb(Src, MemReg);
} else {
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
switch (OpSize) {
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
@@ -1880,7 +1851,7 @@ DEF_OP(MemSet) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
(void)tbnz(DirectionReg, 1, &BackwardImpl);
tbnz(DirectionReg, 1, &BackwardImpl);
}
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
@@ -1898,9 +1869,7 @@ DEF_OP(MemSet) {
// 8bit load is always aligned to natural alignment
stlrb(Value.W(), TMP2);
} else {
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
nop();
}
nop();
switch (OpSize) {
case 2: stlrh(Value.W(), TMP2); break;
case 4: stlr(Value.W(), TMP2); break;
@@ -1930,7 +1899,7 @@ DEF_OP(MemSet) {
ARMEmitter::ForwardLabel DoneInternal {};
// Early exit if zero count.
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AgainInternal256Exit {};
@@ -1947,50 +1916,50 @@ DEF_OP(MemSet) {
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
tbnz(TMP1, 63, &AgainInternal128Exit);
// Fill VTMP2 with the set pattern
dup(SubRegSize, VTMP2.Q(), Value);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
tbnz(TMP1, 63, &AgainInternal256Exit);
(void)Bind(&AgainInternal256);
Bind(&AgainInternal256);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)tbz(TMP1, 63, &AgainInternal256);
tbz(TMP1, 63, &AgainInternal256);
(void)Bind(&AgainInternal256Exit);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
(void)Bind(&AgainInternal128);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbz(TMP1, 63, &AgainInternal128);
tbz(TMP1, 63, &AgainInternal128);
(void)Bind(&AgainInternal128Exit);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
}
}
(void)Bind(&AgainInternal);
Bind(&AgainInternal);
if (IsAtomic) {
MemStoreTSO(Value, OpSize, SizeDirection);
} else {
MemStore(Value, OpSize, SizeDirection);
}
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
(void)Bind(&DoneInternal);
Bind(&DoneInternal);
if (SizeDirection >= 0) {
switch (OpSize) {
@@ -2020,12 +1989,12 @@ DEF_OP(MemSet) {
EmitMemset(Direction);
if (Direction == 1) {
(void)b(&Done);
(void)Bind(&BackwardImpl);
b(&Done);
Bind(&BackwardImpl);
}
}
(void)Bind(&Done);
Bind(&Done);
// Destination already set to the final pointer.
}
}
@@ -2075,7 +2044,7 @@ DEF_OP(MemCpy) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
(void)tbnz(DirectionReg, 1, &BackwardImpl);
tbnz(DirectionReg, 1, &BackwardImpl);
}
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
@@ -2118,11 +2087,9 @@ DEF_OP(MemCpy) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Placeholders for backpatching barriers (one per load/store)
nop();
nop();
}
// Placeholders for backpatching barriers (one per load/store)
nop();
nop();
switch (OpSize) {
case 2: stlrh(TMP4.W(), TMP2); break;
@@ -2144,11 +2111,9 @@ DEF_OP(MemCpy) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Placeholders for backpatching barriers (one per load/store)
nop();
nop();
}
// Placeholders for backpatching barriers (one per load/store)
nop();
nop();
switch (OpSize) {
case 2: stlrh(TMP4.W(), TMP2); break;
@@ -2176,7 +2141,7 @@ DEF_OP(MemCpy) {
ARMEmitter::ForwardLabel DoneInternal {};
// Early exit if zero count.
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AbsPos {};
@@ -2186,11 +2151,11 @@ DEF_OP(MemCpy) {
ARMEmitter::BackwardLabel AgainInternal256 {};
sub(ARMEmitter::Size::i64Bit, TMP4, TMP2, TMP3);
(void)tbz(TMP4, 63, &AbsPos);
tbz(TMP4, 63, &AbsPos);
neg(ARMEmitter::Size::i64Bit, TMP4, TMP4);
(void)Bind(&AbsPos);
Bind(&AbsPos);
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
(void)tbnz(TMP4, 63, &AgainInternal);
tbnz(TMP4, 63, &AgainInternal);
if (Direction == -1) {
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
@@ -2202,30 +2167,30 @@ DEF_OP(MemCpy) {
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
tbnz(TMP1, 63, &AgainInternal128Exit);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
tbnz(TMP1, 63, &AgainInternal256Exit);
(void)Bind(&AgainInternal256);
Bind(&AgainInternal256);
MemCpy(32, 32 * Direction);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)tbz(TMP1, 63, &AgainInternal256);
tbz(TMP1, 63, &AgainInternal256);
(void)Bind(&AgainInternal256Exit);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
(void)Bind(&AgainInternal128);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbz(TMP1, 63, &AgainInternal128);
tbz(TMP1, 63, &AgainInternal128);
(void)Bind(&AgainInternal128Exit);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
@@ -2233,16 +2198,16 @@ DEF_OP(MemCpy) {
}
}
(void)Bind(&AgainInternal);
Bind(&AgainInternal);
if (IsAtomic) {
MemCpyTSO(OpSize, SizeDirection);
} else {
MemCpy(OpSize, SizeDirection);
}
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
(void)Bind(&DoneInternal);
Bind(&DoneInternal);
// Needs to use temporaries just in case of overwrite
mov(TMP1, MemRegDest.X());
@@ -2300,15 +2265,186 @@ DEF_OP(MemCpy) {
for (int32_t Direction : {1, -1}) {
EmitMemcpy(Direction);
if (Direction == 1) {
(void)b(&Done);
(void)Bind(&BackwardImpl);
b(&Done);
Bind(&BackwardImpl);
}
}
(void)Bind(&Done);
Bind(&Done);
// Destination already set to the final pointer.
}
}
DEF_OP(ParanoidLoadMemTSO) {
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
const auto OpSize = IROp->Size;
auto MemReg = GetReg(Op->Addr);
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
if (!IsInlineConstant(Op->Offset, &Offset)) {
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
}
}
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
const auto Dst = GetReg(Node);
ldapurb(Dst, MemReg, Offset);
} else {
switch (OpSize) {
case IR::OpSize::i16Bit: ldapurh(Dst, MemReg, Offset); break;
case IR::OpSize::i32Bit: ldapur(Dst.W(), MemReg, Offset); break;
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
ldaprb(Dst.W(), MemReg);
} else {
switch (OpSize) {
case IR::OpSize::i16Bit: ldaprh(Dst.W(), MemReg); break;
case IR::OpSize::i32Bit: ldapr(Dst.W(), MemReg); break;
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit: ldarb(Dst, MemReg); break;
case IR::OpSize::i16Bit: ldarh(Dst, MemReg); break;
case IR::OpSize::i32Bit: ldar(Dst.W(), MemReg); break;
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
} else {
const auto Dst = GetVReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit:
ldarb(TMP1, MemReg);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
break;
case IR::OpSize::i16Bit:
ldarh(TMP1, MemReg);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
break;
case IR::OpSize::i32Bit:
ldar(TMP1.W(), MemReg);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
break;
case IR::OpSize::i64Bit:
ldar(TMP1, MemReg);
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
break;
case IR::OpSize::i128Bit:
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
clrex();
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
break;
case IR::OpSize::i256Bit:
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
dmb(ARMEmitter::BarrierScope::ISH);
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
dmb(ARMEmitter::BarrierScope::ISH);
break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
}
DEF_OP(ParanoidStoreMemTSO) {
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
const auto OpSize = IROp->Size;
auto MemReg = GetReg(Op->Addr);
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
if (!IsInlineConstant(Op->Offset, &Offset)) {
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
}
}
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
stlurb(Src, MemReg, Offset);
} else {
switch (OpSize) {
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
case IR::OpSize::i64Bit: stlur(Src.X(), MemReg, Offset); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
case IR::OpSize::i64Bit: stlr(Src.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
} else {
const auto Src = GetVReg(Op->Value);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit:
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
stlrb(TMP1, MemReg);
break;
case IR::OpSize::i16Bit:
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
stlrh(TMP1, MemReg);
break;
case IR::OpSize::i32Bit:
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
stlr(TMP1.W(), MemReg);
break;
case IR::OpSize::i64Bit:
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
stlr(TMP1, MemReg);
break;
case IR::OpSize::i128Bit: {
// Move vector to GPRs
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
umov<ARMEmitter::SubRegSize::i64Bit>(TMP2, Src, 1);
ARMEmitter::BackwardLabel B;
Bind(&B);
// ldaxp must not have both the destination registers be the same
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
break;
}
case IR::OpSize::i256Bit: {
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
dmb(ARMEmitter::BarrierScope::ISH);
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
dmb(ARMEmitter::BarrierScope::ISH);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
}
}
DEF_OP(CacheLineClear) {
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
dmb(ARMEmitter::BarrierScope::SY);
@@ -2450,7 +2586,7 @@ DEF_OP(VStoreNonTemporalPair) {
const auto Op = IROp->C<IR::IROp_VStoreNonTemporalPair>();
const auto OpSize = IROp->Size;
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
[[maybe_unused]] const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
LOGMAN_THROW_A_FMT(Is128Bit, "This IR operation only operates at 128-bit wide");
const auto ValueLow = GetVReg(Op->ValueLow);
+1 -61
View File
@@ -10,34 +10,13 @@ $end_info$
#endif
#include "Interface/Context/Context.h"
#include "Interface/Core/JIT/DebugData.h"
#include "Interface/Core/JIT/JITClass.h"
#include "FEXCore/Debug/InternalThreadState.h"
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
namespace FEXCore::CPU {
DEF_OP(WFET) {
auto Op = IROp->C<IR::IROp_WFET>();
const auto Lower = GetReg(Op->Lower);
const auto Upper = GetReg(Op->Upper);
// Combine registers.
mov(ARMEmitter::Size::i64Bit, TMP1, Lower);
bfi(ARMEmitter::Size::i64Bit, TMP1, Upper, 32, 32);
if (CTX->Config.TSCScale) {
// Scale back to ARM64 TSC scale if necessary
lsr(ARMEmitter::Size::i64Bit, TMP1, TMP1, CTX->Config.TSCScale);
}
// Clear the exclusive monitor so it can't spuriously wake up with that event.
clrex();
// Execute wfet to wait until the TSC.
wfet(TMP1);
}
DEF_OP(GuestOpcode) {
auto Op = IROp->C<IR::IROp_GuestOpcode>();
// metadata
@@ -287,43 +266,4 @@ DEF_OP(Yield) {
yield();
}
DEF_OP(MonoBackpatcherWrite) {
auto Op = IROp->C<IR::IROp_MonoBackpatcherWrite>();
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Addr));
mov(ARMEmitter::Size::i64Bit, TMP4, GetReg(Op->Value));
PushDynamicRegs(TMP1);
SpillStaticRegs(TMP1);
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, IR::OpSizeToSize(Op->Size));
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, TMP3);
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, TMP4);
}
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
#endif
ldr(ARMEmitter::XReg::x4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.MonoBackpatcherWrite));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, uint8_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
} else {
blr(ARMEmitter::Reg::r4);
}
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
strb(ARMEmitter::WReg::zr, TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
#endif
FillStaticRegs();
PopDynamicRegs();
}
} // namespace FEXCore::CPU
+32 -32
View File
@@ -193,29 +193,29 @@ namespace FEXCore::CPU {
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2); \
}
#define DEF_FMAOP_SCALAR_INSERT(FEXOp, ARMOp) \
DEF_OP(FEXOp) { \
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
const auto ElementSize = Op->Header.ElementSize; \
\
auto ScalarEmit = [this, ElementSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2, \
ARMEmitter::VRegister Src3) { \
if (ElementSize == IR::OpSize::i16Bit) { \
ARMOp(Dst.H(), Src1.H(), Src2.H(), Src3.H()); \
} else if (ElementSize == IR::OpSize::i32Bit) { \
ARMOp(Dst.S(), Src1.S(), Src2.S(), Src3.S()); \
} else if (ElementSize == IR::OpSize::i64Bit) { \
ARMOp(Dst.D(), Src1.D(), Src2.D(), Src3.D()); \
} \
}; \
\
const auto Dst = GetVReg(Node); \
const auto Upper = GetVReg(Op->Upper); \
const auto Vector1 = GetVReg(Op->Vector1); \
const auto Vector2 = GetVReg(Op->Vector2); \
const auto Addend = GetVReg(Op->Addend); \
\
VFScalarFMAOperation(IROp->Size, ElementSize, ScalarEmit, Dst, Upper, Vector1, Vector2, Addend); \
#define DEF_FMAOP_SCALAR_INSERT(FEXOp, ARMOp) \
DEF_OP(FEXOp) { \
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
const auto ElementSize = Op->Header.ElementSize; \
\
auto ScalarEmit = \
[this, ElementSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2, ARMEmitter::VRegister Src3) { \
if (ElementSize == IR::OpSize::i16Bit) { \
ARMOp(Dst.H(), Src1.H(), Src2.H(), Src3.H()); \
} else if (ElementSize == IR::OpSize::i32Bit) { \
ARMOp(Dst.S(), Src1.S(), Src2.S(), Src3.S()); \
} else if (ElementSize == IR::OpSize::i64Bit) { \
ARMOp(Dst.D(), Src1.D(), Src2.D(), Src3.D()); \
} \
}; \
\
const auto Dst = GetVReg(Node); \
const auto Upper = GetVReg(Op->Upper); \
const auto Vector1 = GetVReg(Op->Vector1); \
const auto Vector2 = GetVReg(Op->Vector2); \
const auto Addend = GetVReg(Op->Addend); \
\
VFScalarFMAOperation(IROp->Size, ElementSize, ScalarEmit, Dst, Upper, Vector1, Vector2, Addend); \
}
DEF_UNOP(VAbs, abs, true)
@@ -803,8 +803,8 @@ DEF_OP(VFCMPScalarInsert) {
default: break;
}
};
auto ScalarEmitUNO = [this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1,
ARMEmitter::VRegister Src2) {
auto ScalarEmitUNO =
[this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
switch (SubRegSize.Scalar) {
case ARMEmitter::ScalarRegSize::i16Bit: {
fcmge(VTMP1.H(), Src1.H(), Src2.H());
@@ -838,8 +838,8 @@ DEF_OP(VFCMPScalarInsert) {
}
}
};
auto ScalarEmitNEQ = [this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1,
ARMEmitter::VRegister Src2) {
auto ScalarEmitNEQ =
[this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
switch (SubRegSize.Scalar) {
case ARMEmitter::ScalarRegSize::i16Bit: {
fcmeq(VTMP1.H(), Src2.H(), Src1.H());
@@ -868,8 +868,8 @@ DEF_OP(VFCMPScalarInsert) {
}
}
};
auto ScalarEmitORD = [this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1,
ARMEmitter::VRegister Src2) {
auto ScalarEmitORD =
[this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
switch (SubRegSize.Scalar) {
case ARMEmitter::ScalarRegSize::i16Bit: {
fcmge(VTMP1.H(), Src1.H(), Src2.H());
@@ -1115,7 +1115,7 @@ DEF_OP(VAddP) {
}
DEF_OP(VFAddV) {
const auto Op = IROp->C<IR::IROp_VFAddV>();
const auto Op = IROp->C<IR::IROp_VAddV>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
@@ -1352,7 +1352,7 @@ DEF_OP(VFMin) {
const auto ElementSize = Op->Header.ElementSize;
const auto SubRegSize = ConvertSubRegSize248(IROp);
const auto IsScalar = ElementSize == OpSize;
[[maybe_unused]] const auto IsScalar = ElementSize == OpSize;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
@@ -1425,7 +1425,7 @@ DEF_OP(VFMax) {
const auto ElementSize = Op->Header.ElementSize;
const auto SubRegSize = ConvertSubRegSize248(IROp);
const auto IsScalar = ElementSize == OpSize;
[[maybe_unused]] const auto IsScalar = ElementSize == OpSize;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
+19 -24
View File
@@ -56,8 +56,6 @@ struct GuestToHostMap {
fextl::robin_map<uint64_t, uint64_t> BlockList;
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
GuestToHostMap();
// Adds to Guest -> Host code mapping
@@ -78,7 +76,7 @@ struct GuestToHostMap {
return HostCode->second;
}
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LockToken&) {
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LockToken&) {
// Sever any links to this block
auto lower = BlockLinks->lower_bound({Address, nullptr});
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
@@ -87,7 +85,7 @@ struct GuestToHostMap {
}
// Remove from BlockList
return BlockList.erase(Address) != 0;
BlockList.erase(Address);
}
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
@@ -95,18 +93,6 @@ struct GuestToHostMap {
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
}
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LockToken&) {
bool rv = false;
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
auto& CodePage = CodePages[CurrentPage];
rv |= CodePage.empty();
CodePage.insert(CodePage.end(), Addresses.begin(), Addresses.end());
}
return rv;
}
void ClearCache(const LockToken&);
};
@@ -169,11 +155,22 @@ public:
GuestToHostMap* Shared = nullptr;
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
// Appends Block {Address} to CodePages [Start, Start + Length)
// Returns true if new pages are marked as containing code
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
auto lk = Shared->AcquireLock();
return Shared->AddBlockExecutableRange(Addresses, Start, Length, lk);
bool rv = false;
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
auto& CodePage = CodePages[CurrentPage];
rv |= CodePage.empty();
CodePage.push_back(Address);
}
return rv;
}
// Adds to Guest -> Host code mapping
@@ -192,16 +189,15 @@ public:
// NOTE: It's the caller's responsibility to call Erase() for all other
// GuestToHostMaps that share the same LookupCache. Otherwise, the
// L1/L2 caches will contain stale references to deallocated memory.
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
auto lk = Shared->AcquireLock();
bool ErasedAny = Shared->Erase(Frame, Address, lk);
Shared->Erase(Frame, Address, lk);
// Do L1
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
if (L1Entry.GuestCode == Address) {
L1Entry.GuestCode = 0;
ErasedAny = true;
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
// This is a soft guarantee for cross thread invalidation, as atomics are not used
// and it hasn't been thoroughly tested
@@ -216,14 +212,13 @@ public:
uint64_t LocalPagePointer = Pointers[Address];
if (!LocalPagePointer) {
// Page for this code didn't even exist, nothing to do
return ErasedAny;
return;
}
// Page exists, just set the offset to zero
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
BlockPointers[PageOffset].GuestCode = 0;
BlockPointers[PageOffset].HostCode = 0;
return true;
}
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
@@ -0,0 +1,86 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/CompilerDefs.h>
#include <cstdint>
namespace FEXCore::CodeSerialize {
// If any of the config options mismatch on load then the cache won't be used
// Any of these will result in codegen changes
struct FEX_PACKED CodeObjectSerializationConfig {
// Cookie in the header of the file, isn't part of the config hash
uint64_t Cookie {};
// Instructions per block configuration
int32_t MaxInstPerBlock {};
// Follows CPUID 4000_0001_EAX[3:0]
unsigned Arch : 4;
// Multiblock enabled
unsigned MultiBlock : 1;
// Hardware TSO enabled
unsigned HardwareTSOEnabled : 1;
// TSO enabled
unsigned TSOEnabled : 1;
// ABI local flag unsafe optimization
unsigned ABILocalFlags : 1;
// Paranoid TSO mode enabled
unsigned ParanoidTSO : 1;
// Guest code execution mode (We don't support live mode switch)
unsigned Is64BitMode : 1;
// SMC checks style
unsigned SMCChecks : 2;
// x87 reduced precision
unsigned x87ReducedPrecision : 1;
// Padding to remove uninitialized data warning from asan
// Shows remaining amount of bits available for config
unsigned _Pad : 19;
bool operator==(const CodeObjectSerializationConfig& other) const {
return Cookie == other.Cookie && MaxInstPerBlock == other.MaxInstPerBlock && Arch == other.Arch && MultiBlock == other.MultiBlock &&
HardwareTSOEnabled == other.HardwareTSOEnabled && TSOEnabled == other.TSOEnabled && ABILocalFlags == other.ABILocalFlags &&
ParanoidTSO == other.ParanoidTSO && Is64BitMode == other.Is64BitMode && SMCChecks == other.SMCChecks &&
x87ReducedPrecision == other.x87ReducedPrecision;
}
static uint64_t GetHash(const CodeObjectSerializationConfig& other) {
// For < 64-bits of data just pack directly
// Skip the cookie
uint64_t Hash {};
Hash <<= 32;
Hash |= other.MaxInstPerBlock;
Hash <<= 1;
Hash |= other.Arch;
Hash <<= 1;
Hash |= other.MultiBlock;
Hash <<= 1;
Hash |= other.HardwareTSOEnabled;
Hash <<= 1;
Hash |= other.TSOEnabled;
Hash <<= 1;
Hash |= other.ABILocalFlags;
Hash <<= 1;
Hash |= other.ParanoidTSO;
Hash <<= 1;
Hash |= other.Is64BitMode;
Hash <<= 2;
Hash |= other.SMCChecks;
Hash <<= 1;
Hash |= other.x87ReducedPrecision;
return Hash;
}
};
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is "
"generated!");
} // namespace FEXCore::CodeSerialize
@@ -0,0 +1,121 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/Filesystem.h>
#include <fcntl.h>
#include <xxhash.h>
namespace FEXCore::CodeSerialize {
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
#ifndef _WIN32
// This function adds a named region *JOB* to our named region handler
// This needs to be as fast as possible to keep out of the way of the JIT
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
if (!BaseFilename.empty()) {
// Create a new entry that once set up will be put in to our section object map
auto Entry = fextl::make_unique<CodeRegionEntry>(Base, Size, Offset, filename, NamedRegionHandler->DefaultCodeHeader(Base, Offset));
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
Entry->NamedJobRefCountMutex.lock();
CodeRegionMapType::iterator EntryIterator;
{
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
auto it = EntryMap.emplace(Base, std::move(Entry));
if (!it.second) {
// This happens when an application overwrites a previous region without unmapping what was there
// Lock this entry's Named job reference counter.
// Once this passes then we know that this section has been loaded.
it.first->second->NamedJobRefCountMutex.lock();
// Finalize anything the region needs to do first.
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
// munmap the file that was mapped
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
// Remove this entry from the unrelocated map as well
{
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
}
// Now overwrite the entry in the map
it = EntryMap.insert_or_assign(Base, std::move(Entry));
EntryIterator = it.first;
} else {
// No overwrite, just insert
EntryIterator = it.first;
}
}
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
// do the loading for us.
//
// Create the async work queue job now so it can load
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
#endif
}
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
#ifndef _WIN32
// Removing a named region through the job system
// We need to find the entry that we are deleting first
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
{
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
auto it = EntryMap.find(Base);
if (it != EntryMap.end()) {
// Lock the job ref counter since we are erasing it
// Once this passes it will have been loaded
it->second->NamedJobRefCountMutex.lock();
// Take the pointer from the map
EntryPointer = std::move(it->second);
// We can now unmap the file data
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
// Remove this from the entry map
EntryMap.erase(it);
// Remove this entry from the unrelocated map as well
{
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
}
} else {
// Tried to remove something that wasn't in our code object tracking
return;
}
// Create the async work queue job now so it can finalize what it needs to do
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
#endif
}
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
// XXX: Actually add serialization job
}
} // namespace FEXCore::CodeSerialize
@@ -0,0 +1,71 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
namespace FEXCore::CodeSerialize {
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx) {
DefaultSerializationConfig.Cookie = CODE_COOKIE;
// Initialize the Arch from CPUID
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
DefaultSerializationConfig.Arch = Arch;
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
}
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename,
const fextl::string& filename, bool Executable) {
// XXX: Add named region objects
// XXX: Until entry loading is complete just claim it is loaded
Entry->second->NamedJobRefCountMutex.unlock();
}
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
// XXX: Remove named region objects
// XXX: Until entry loading is complete just claim it is loaded
Entry->NamedJobRefCountMutex.unlock();
}
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
// Walk through all of our jobs sequentially until the work queue is empty
while (NamedWorkQueueJobs.load()) {
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
{
// Lock the work queue mutex for a short moment and grab an item from the list
std::unique_lock lk {NamedWorkQueueMutex};
size_t WorkItems = WorkQueue.size();
if (WorkItems != 0) {
WorkItem = std::move(WorkQueue.front());
WorkQueue.pop();
}
// Atomically update the number of jobs
--NamedWorkQueueJobs;
}
if (WorkItem) {
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion*>(WorkItem.get());
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
}
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion*>(WorkItem.get());
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
}
}
}
}
} // namespace FEXCore::CodeSerialize
@@ -0,0 +1,85 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/Utils/Threads.h>
namespace {
static void* ThreadHandler(void* Arg) {
FEXCore::CodeSerialize::CodeObjectSerializeService* This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
This->ExecutionThread();
return nullptr;
}
} // namespace
namespace FEXCore::CodeSerialize {
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx)
: CTX {ctx}
, AsyncHandler {&NamedRegionHandler, this}
, NamedRegionHandler {ctx} {
Initialize();
}
void CodeObjectSerializeService::Shutdown() {
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
return;
}
WorkerThreadShuttingDown = true;
// Kick the working thread
WorkAvailable.NotifyAll();
if (WorkerThread->joinable()) {
// Wait for worker thread to close down
WorkerThread->join(nullptr);
}
}
void CodeObjectSerializeService::Initialize() {
// Add a canary so we don't crash on empty map iterator handling
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
FEXCore::Threads::SetSignalMask(OldMask);
}
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it) {
if (Base == ~0ULL) {
// Don't do closure on canary
return;
}
// XXX: Do code region closure
}
const CodeObjectFileSection* CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
// XXX: Actually fetch code objects from cache
return nullptr;
}
void CodeObjectSerializeService::ExecutionThread() {
// Set our thread name so we can see its relation
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
while (WorkerThreadShuttingDown.load() != true) {
// Wait for work
WorkAvailable.Wait();
// Handle named region async jobs first. Highest priority
NamedRegionHandler.HandleNamedRegionObjectJobs();
// XXX: Handle code serialization jobs second.
}
// Do final code region closures on thread shutdown
for (auto& it : AddressToEntryMap) {
DoCodeRegionClosure(it.first, it.second.get());
}
// Safely clear our maps now
AddressToEntryMap.clear();
UnrelocatedAddressToEntryMap.clear();
}
} // namespace FEXCore::CodeSerialize
@@ -0,0 +1,457 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Interface/Context/Context.h"
#include "Interface/Core/ObjectCache/Relocations.h"
#include "Interface/Core/ObjectCache/CodeObjectSerializationConfig.h"
#include "Interface/IR/AOTIR.h"
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/Threads.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/queue.h>
#include <FEXCore/fextl/robin_map.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <shared_mutex>
namespace FEXCore::CodeSerialize {
// XXX: Does this need to be signal safe?
using CodeSerializationMutex = std::shared_mutex;
struct CodeSerializationData {};
struct CodeObjectFileSection {
bool Serialized;
bool Invalid;
const CodeSerializationData* Data;
const char* HostCode;
uint64_t NumRelocations;
const char* Relocations;
};
/**
* @brief This is the file header that lives at the start of an object cache file
*
* This header is updated from multiple processes!
* Care must be taken to use OS locks when updating the file backing including this header
*/
struct CodeObjectSerializationHeader {
// The configuration that this file has
CodeObjectSerializationConfig Config;
// The original RIP that this object section was mapped at
uint64_t OriginalBase {};
// The original offset in to the file that this object section was loaded from
uint64_t OriginalOffset {};
// Total amount of code that should be in this file
uint64_t TotalCodeSize {};
// Used to reserve the TSL map
uint64_t NumCodeEntries {};
// The number of relocations that point to this section
uint64_t NumRelocationsTo {};
// Total relocations in this file
uint64_t TotalRelocationsCount {};
};
struct CodeRegionEntry {
/**
* @name Threaded initialization objects for the initial object creation
* @{ */
// Base address in memory where the code region is at
uint64_t Base {};
// Size of this code entry
uint64_t Size {};
// The offset inside the file that is mapped to Base
uint64_t Offset {};
// Filename of the object
fextl::string Filename {};
CodeObjectSerializationHeader EntryHeader {};
/** @} */
// The filename of the object cache for this entry
fextl::string ObjectEntrySourceFilename {};
// In the case of file corruption that we can detect, we can disable serialization early for an entry
// We should be resiliant to corruption but things happen
bool StillSerializing {true};
// Long lived FD for serialization if we have multiple jobs to serialize
// Bursts of code entries are common and this reduces file lock overhead
//
// Especially useful over network mounts where file locks are very slow
int CurrentSerializedFD {-1};
/**
* @name Objects required to sync objects between threads
* @{ */
// Refcount for the number of outstanding code entries waiting to be written for this object section
CodeSerializationMutex ObjectJobRefCountMutex;
// Refcount for outstanding named object region entry loading itself
// Will block JIT code cache look up when this has a unique_lock held
CodeSerializationMutex NamedJobRefCountMutex;
/** @} */
/**
* @name Object Entry data management
* @{ */
/**
* @name This is the raw file data that we loaded from the code region entry file
* @{ */
char* CodeData {};
size_t FileSize {};
fextl::vector<CodeObjectFileSection> FileCodeSections;
/** @} */
// This per section map takes the most time to load and needs to be quick
// This is the map of all code segments for this entry
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap {};
/** @} */
// Default initialization
CodeRegionEntry() = default;
// Initializer specifically for threaded loading
CodeRegionEntry(uint64_t Base, uint64_t Size, uint64_t Offset, const fextl::string& Filename, const CodeObjectSerializationHeader& DefaultHeader)
: Base {Base}
, Size {Size}
, Offset {Offset}
, Filename {Filename}
, EntryHeader {DefaultHeader} {}
};
// Map type must use an interator that isn't invalidation on erase/insert
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
class NamedRegionObjectHandler;
class CodeObjectSerializeService;
class AsyncJobHandler final {
public:
/**
* @brief Structure containing all the data required to async serialize code objects
*/
struct SerializationJobData {
uint64_t GuestRIP; ///< The RIP for the guest
// XXX: Support multiblock
uint64_t GuestCodeLength; ///< The Guest's code length
uint64_t GuestCodeHash; ///< Hash of the guest code
void* HostCodeBegin; ///< Host JIT code starting memory address
size_t HostCodeLength; ///< Host JIT code length
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
// This is the thread specific ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
// This way it will wait until the async job handler is complete with it.
CodeSerializationMutex* ThreadJobRefCount;
// These are the reolocations for this serialization job
// Relatively small number of entries most of the time
fextl::vector<FEXCore::CPU::Relocation> Relocations;
/**
* @name Objects filled in from the Code Object Serialization service when a job is added
* @{ */
// This is the code region's ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
CodeSerializationMutex* ObjectJobRefCountMutexPtr;
// This is the code region iterator to reduce the number of map lookups
// This will remain valid while jobs are outstanding for this region
CodeRegionMapType::iterator CodeRegionIterator;
/** @} */
};
AsyncJobHandler(NamedRegionObjectHandler* NamedRegionHandler, CodeObjectSerializeService* CodeObjectCacheService)
: NamedRegionHandler {NamedRegionHandler}
, CodeObjectCacheService {CodeObjectCacheService} {}
protected:
friend class CodeObjectSerializeService;
friend class NamedRegionObjectHandler;
/**
* @name Async job submission functions
* @{ */
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename);
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
/** @} */
/**
* @name Async named region handling
* @{ */
/**
* @brief The async named region jobs to handle.
*
* Only two, Code serialization goes in to a different queue.
*/
enum class NamedRegionJobType {
JOB_ADD_NAMED_REGION,
JOB_REMOVE_NAMED_REGION,
};
class NamedRegionWorkItem {
public:
NamedRegionJobType GetType() const {
return Type;
}
protected:
friend class WorkItemAddNamedRegion;
NamedRegionWorkItem(NamedRegionJobType type)
: Type {type} {}
private:
NamedRegionJobType Type;
};
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
public:
WorkItemAddNamedRegion(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
, BaseFilename {base}
, Filename {filename}
, Executable {executable}
, Entry {entry} {}
const fextl::string BaseFilename;
const fextl::string Filename;
bool Executable;
CodeRegionMapType::iterator Entry;
};
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
public:
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
, Base {base}
, Size {size}
, Entry {std::move(entry)} {}
uint64_t Base;
uint64_t Size;
fextl::unique_ptr<CodeRegionEntry> Entry;
};
/** @} */
private:
NamedRegionObjectHandler* NamedRegionHandler;
CodeObjectSerializeService* CodeObjectCacheService;
};
class NamedRegionObjectHandler final {
public:
NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx);
void HandleNamedRegionObjectJobs();
const CodeObjectSerializationConfig& GetDefaultSerializationConfig() const {
return DefaultSerializationConfig;
}
protected:
friend class AsyncJobHandler;
// Return a default code header based off the default serialization config
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
return CodeObjectSerializationHeader {
.Config = DefaultSerializationConfig,
.OriginalBase = Base,
.OriginalOffset = Offset,
.NumCodeEntries = 0,
.NumRelocationsTo = 0,
.TotalRelocationsCount = 0,
};
}
/**
* @brief Adds an asynchronous add named region work item to the object queue
*
* This adds the job that will do the loading of file resources and data tracking.
*/
void AsyncAddNamedRegionWorkItem(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion>(base, filename, executable, entry));
++NamedWorkQueueJobs;
}
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion>(Base, Size, std::move(Entry)));
++NamedWorkQueueJobs;
}
private:
// Code version. If the code emission changes then this needs to increment
constexpr static uint32_t CODE_VERSION = 0x0;
// Default cookie header for the file header
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
// Code serialization config for our current process configuration
CodeObjectSerializationConfig DefaultSerializationConfig;
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
std::atomic<uint64_t> NamedWorkQueueJobs {};
// Mutex for ading new jobs to the work queue
std::mutex NamedWorkQueueMutex {};
// The job queue itself
// Jobs get consumed as a FIFO
// Jobs always get appended to the end
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue {};
/**
* @name Named Region object handling
* @{ */
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename, const fextl::string& filename, bool Executable);
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
/** @} */
};
/**
* @brief Context specific code object serialization class
*
* Contains everything required for FEXCore to serialize code objects
*/
class CodeObjectSerializeService final {
public:
CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx);
/**
* @brief Initialize the internal interface
*
* Is a public interface to allow the service to reinitialize after forking
*/
void Initialize();
/**
* @brief Safely shut down the Code Object serialization service.
*
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
*/
void Shutdown();
/**
* @name Async interface
* @{ */
/**
* @brief Loads a named region in to the code serialization service. As async as possible.
*
* @param Base - Virtual address that this named region is loaded
* @param Size - The size of the region
* @param Offset - The offset from the file
* @param filename - The filename itself
*/
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
}
/**
* @brief Unloads a named region from the code serialization service. As async as possible.
*
* @param Base - Virtual address of the named region
* @param Size - The size of the region
*/
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
}
/**
* @brief Adds a code object serialization job. As async as possible.
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
*
* @param Data - A fully filled out struct containing all the code serialization
*/
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
}
/** @} */
/**
* @name Synchronous interface
* @{ */
/**
* @brief Synchronously waits for this thread's job queue to become empty.
*
* This is necessary for when a thread is shutting down
*
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
*/
static void WaitForEmptyJobQueue(CodeSerializationMutex* ThreadJobRefCount) {
// Once the shared mutex is empty this unique lock will be gained
std::unique_lock lk {*ThreadJobRefCount};
}
/**
* @brief Fetches object code from the Code Object Cache for JIT.
*
* @param GuestRIP - Which GuestRIP to search the cache for
*
* @return Data required for the JIT to relocate the Object code.
*/
const CodeObjectFileSection* FetchCodeObjectFromCache(uint64_t GuestRIP);
/** @} */
// Public for threading
void ExecutionThread();
protected:
friend class AsyncJobHandler;
/**
* @brief Safely closes out code object regions from the map
*
* @param it - iterator to do a closure on
*/
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it);
CodeSerializationMutex& GetEntryMapMutex() {
return EntryMapMutex;
}
CodeSerializationMutex& GetUnrelocatedEntryMapMutex() {
return EntryMapMutex;
}
CodeRegionMapType& GetEntryMap() {
return AddressToEntryMap;
}
CodeRegionPtrMapType& GetUnrelocatedEntryMap() {
return UnrelocatedAddressToEntryMap;
}
/**
* @brief Notify the async thread that it has work to do
*/
void NotifyWork() {
WorkAvailable.NotifyOne();
}
private:
FEXCore::Context::ContextImpl* CTX;
Event WorkAvailable {};
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
std::atomic_bool WorkerThreadShuttingDown {false};
AsyncJobHandler AsyncHandler;
NamedRegionObjectHandler NamedRegionHandler;
// Mutex to hold when modifying the entry maps
CodeSerializationMutex EntryMapMutex;
CodeSerializationMutex UnrelocatedEntryMapMutex;
// Entry maps
CodeRegionMapType AddressToEntryMap;
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
};
} // namespace FEXCore::CodeSerialize
File diff suppressed because it is too large. Load diff
+166 -210
View File
@@ -27,6 +27,8 @@
#include <xxhash.h>
namespace FEXCore::IR {
class Pass;
class PassManager;
enum class MemoryAccessType {
// Choose TSO or Non-TSO depending on access type
@@ -78,13 +80,10 @@ struct LoadSourceOptions {
bool AllowUpperGarbage = false;
};
struct DispatchTableEntry {
uint16_t Op;
uint8_t Count;
X86Tables::OpDispatchPtr Ptr;
};
class OpDispatchBuilder final : public IREmitter {
friend class FEXCore::IR::Pass;
friend class FEXCore::IR::PassManager;
public:
Ref GetNewJumpBlock(uint64_t RIP) {
auto it = JumpTargets.find(RIP);
@@ -161,13 +160,9 @@ public:
auto InlineConst = _InlineConstant(Bit);
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, {Set ? COND_TSTNZ : COND_TSTZ}, OpSize::iInvalid, false);
}
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint = BranchHint::None) {
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP) {
FlushRegisterCache();
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, InvalidNode, InvalidNode);
}
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
FlushRegisterCache();
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, CallReturnAddress, CallReturnBlock);
return _ExitFunction(GetOpSize(NewRIP), NewRIP);
}
IRPair<IROp_Break> Break(BreakDefinition Reason) {
FlushRegisterCache();
@@ -193,10 +188,11 @@ public:
auto it = JumpTargets.find(NextRIP);
if (it == JumpTargets.end()) {
const auto GPRSize = GetGPROpSize();
const auto GPRSize = CTX->GetGPROpSize();
// If we don't have a jump target to a new block then we have to leave
// Set the RIP to the next instruction and leave
ExitFunction(_InlineEntrypointOffset(GPRSize, NextRIP - Entry));
auto RelocatedNextRIP = _EntrypointOffset(GPRSize, NextRIP - Entry);
ExitFunction(RelocatedNextRIP);
} else if (it != JumpTargets.end()) {
Jump(it->second.BlockEntry);
return true;
@@ -209,17 +205,10 @@ public:
}
static bool CanHaveSideEffects(const FEXCore::X86Tables::X86InstInfo* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
if (TableInfo) {
if (TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
// If it is marked as having memory access then always say it has a side-effect.
// Not always true but better to be safe.
return true;
}
if (TableInfo->Flags & (X86Tables::InstFlags::FLAGS_SETS_RIP | X86Tables::InstFlags::FLAGS_BLOCK_END)) {
// Cooperative suspend interrupts can be triggered at any back-edge, the RIP must be reconstructed correctly in such cases
return true;
}
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
// If it is marked as having memory access then always say it has a side-effect.
// Not always true but better to be safe.
return true;
}
auto CanHaveSideEffects = false;
@@ -243,7 +232,7 @@ public:
template<typename F>
void ForeachDirection(F&& Routine) {
// Otherwise, prepare to branch.
auto Zero = Constant(0);
auto Zero = _Constant(0);
// If the shift is zero, do not touch the flags.
auto ForwardBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
@@ -267,6 +256,7 @@ public:
}
OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx);
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator& Allocator);
void ResetWorkingList();
void ResetDecodeFailure() {
@@ -300,8 +290,7 @@ public:
return ShouldDump;
}
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions,
bool Is64BitMode, bool MonoBackpatcherBlock);
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions);
void Finalize();
// Dispatch builder functions
@@ -318,7 +307,6 @@ public:
void UnhandledOp(OpcodeArgs);
void MOVGPROp(OpcodeArgs, uint32_t SrcIndex);
void MOVGPRImmediate(OpcodeArgs);
void MOVGPRNTOp(OpcodeArgs);
void MOVVectorAlignedOp(OpcodeArgs);
void MOVVectorUnalignedOp(OpcodeArgs);
@@ -352,9 +340,6 @@ public:
void LoopOp(OpcodeArgs);
void JUMPOp(OpcodeArgs);
void JUMPAbsoluteOp(OpcodeArgs);
void JUMPFARIndirectOp(OpcodeArgs);
void CALLFARIndirectOp(OpcodeArgs);
void RETFARIndirectOp(OpcodeArgs);
void TESTOp(OpcodeArgs, uint32_t SrcIndex);
void MOVSXDOp(OpcodeArgs);
void MOVSXOp(OpcodeArgs);
@@ -717,7 +702,7 @@ public:
Ref ReconstructX87StateFromFSW_Helper(Ref FSW);
void FLD(OpcodeArgs, IR::OpSize Width);
void FLDFromStack(OpcodeArgs);
void FLD_Const(OpcodeArgs, NamedVectorConstant K);
void FLD_Const(OpcodeArgs, NamedVectorConstant Constant);
void FBLD(OpcodeArgs);
void FBSTP(OpcodeArgs);
@@ -740,7 +725,6 @@ public:
void FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
void FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
void FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
void FNCLEX(OpcodeArgs);
void FNINIT(OpcodeArgs);
void FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
void FTST(OpcodeArgs);
@@ -840,13 +824,11 @@ public:
void PHADDS(OpcodeArgs);
void PHSUBS(OpcodeArgs);
void CLWBOrTPause(OpcodeArgs);
void CLWB(OpcodeArgs);
void CLFLUSHOPT(OpcodeArgs);
void LoadFenceOrXRSTOR(OpcodeArgs);
void MemFenceOrXSAVEOPT(OpcodeArgs);
void StoreFenceOrCLFlush(OpcodeArgs);
void UMonitorOrCLRSSBSY(OpcodeArgs);
void UMWaitOp(OpcodeArgs);
void CLZeroOp(OpcodeArgs);
void RDTSCPOp(OpcodeArgs);
void RDPIDOp(OpcodeArgs);
@@ -855,6 +837,8 @@ public:
void PSADBW(OpcodeArgs);
Ref BitwiseAtLeastTwo(Ref A, Ref B, Ref C);
void SHA1NEXTEOp(OpcodeArgs);
void SHA1MSG1Op(OpcodeArgs);
void SHA1MSG2Op(OpcodeArgs);
@@ -944,6 +928,7 @@ public:
RefVSIB AVX128_LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags, bool NeedsHigh);
void AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand, const RefPair Src,
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
void InstallAVX128Handlers();
void AVX128_VMOVScalarImpl(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_VectorALU(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
void AVX128_VectorUnary(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
@@ -972,26 +957,39 @@ public:
void AVX128_VMOVDDUP(OpcodeArgs);
void AVX128_VMOVSLDUP(OpcodeArgs);
void AVX128_VMOVSHDUP(OpcodeArgs);
void AVX128_VBROADCAST(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_VPUNPCKL(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_VPUNPCKH(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VBROADCAST(OpcodeArgs);
template<IR::OpSize ElementSize>
void AVX128_VPUNPCKL(OpcodeArgs);
template<IR::OpSize ElementSize>
void AVX128_VPUNPCKH(OpcodeArgs);
void AVX128_MOVVectorUnaligned(OpcodeArgs);
void AVX128_InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstElementSize);
void AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode);
template<IR::OpSize DstElementSize>
void AVX128_InsertCVTGPR_To_FPR(OpcodeArgs);
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
void AVX128_CVTFPR_To_GPR(OpcodeArgs);
void AVX128_VANDN(OpcodeArgs);
void AVX128_VPACKSS(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_VPACKUS(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VPACKSS(OpcodeArgs);
template<IR::OpSize ElementSize>
void AVX128_VPACKUS(OpcodeArgs);
Ref AVX128_PSIGNImpl(IR::OpSize ElementSize, Ref Src1, Ref Src2);
void AVX128_VPSIGN(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VPSIGN(OpcodeArgs);
template<IR::OpSize ElementSize>
void AVX128_UCOMISx(OpcodeArgs);
void AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IROps IROp, IR::OpSize ElementSize);
Ref AVX128_VFCMPImpl(IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
void AVX128_VFCMP(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VFCMP(OpcodeArgs);
template<IR::OpSize ElementSize>
void AVX128_InsertScalarFCMP(OpcodeArgs);
void AVX128_MOVBetweenGPR_FPR(OpcodeArgs);
void AVX128_PExtr(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_PExtr(OpcodeArgs);
void AVX128_ExtendVectorElements(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DstElementSize, bool Signed);
void AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_MOVMSK(OpcodeArgs);
void AVX128_MOVMSKB(OpcodeArgs);
void AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op, const X86Tables::DecodedOperand& Imm);
@@ -1005,25 +1003,33 @@ public:
void AVX128_VINSERTPS(OpcodeArgs);
Ref AVX128_PHSUBImpl(Ref Src1, Ref Src2, size_t ElementSize);
void AVX128_VPHSUB(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VPHSUB(OpcodeArgs);
void AVX128_VPHSUBSW(OpcodeArgs);
void AVX128_VADDSUBP(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VADDSUBP(OpcodeArgs);
void AVX128_VPMULL(OpcodeArgs, IR::OpSize ElementSize, bool Signed);
template<IR::OpSize ElementSize, bool Signed>
void AVX128_VPMULL(OpcodeArgs);
void AVX128_VPMULHRSW(OpcodeArgs);
void AVX128_VPMULHW(OpcodeArgs, bool Signed);
template<bool Signed>
void AVX128_VPMULHW(OpcodeArgs);
void AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize);
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
void AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs);
void AVX128_Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize);
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
void AVX128_Vector_CVT_Float_To_Float(OpcodeArgs);
void AVX128_Vector_CVT_Float_To_Int(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode);
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
void AVX128_Vector_CVT_Float_To_Int(OpcodeArgs);
void AVX128_Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen);
template<IR::OpSize SrcElementSize, bool Widen>
void AVX128_Vector_CVT_Int_To_Float(OpcodeArgs);
void AVX128_VEXTRACT128(OpcodeArgs);
void AVX128_VAESImc(OpcodeArgs);
@@ -1040,28 +1046,36 @@ public:
void AVX128_PHMINPOSUW(OpcodeArgs);
void AVX128_VectorRound(OpcodeArgs, IR::OpSize ElementSize);
void AVX128_InsertScalarRound(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VectorRound(OpcodeArgs);
template<IR::OpSize ElementSize>
void AVX128_InsertScalarRound(OpcodeArgs);
void AVX128_VDPP(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VDPP(OpcodeArgs);
void AVX128_VPERMQ(OpcodeArgs);
void AVX128_VPSHUFW(OpcodeArgs, bool Low);
void AVX128_VSHUF(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VSHUF(OpcodeArgs);
void AVX128_VPERMILImm(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VPERMILImm(OpcodeArgs);
void AVX128_VHADDP(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
template<IROps IROp, IR::OpSize ElementSize>
void AVX128_VHADDP(OpcodeArgs);
void AVX128_VPHADDSW(OpcodeArgs);
void AVX128_VPMADDUBSW(OpcodeArgs);
void AVX128_VPMADDWD(OpcodeArgs);
void AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VBLEND(OpcodeArgs);
void AVX128_VHSUBP(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VHSUBP(OpcodeArgs);
void AVX128_VPSHUFB(OpcodeArgs);
void AVX128_VPSADBW(OpcodeArgs);
@@ -1072,23 +1086,28 @@ public:
void AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DstSize, bool IsStore, const X86Tables::DecodedOperand& MaskOp,
const X86Tables::DecodedOperand& DataOp);
void AVX128_VPMASKMOV(OpcodeArgs, bool IsStore);
template<bool IsStore>
void AVX128_VPMASKMOV(OpcodeArgs);
void AVX128_VMASKMOV(OpcodeArgs, IR::OpSize ElementSize, bool IsStore);
template<IR::OpSize ElementSize, bool IsStore>
void AVX128_VMASKMOV(OpcodeArgs);
void AVX128_MASKMOV(OpcodeArgs);
void AVX128_VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VectorVariableBlend(OpcodeArgs);
void AVX128_SaveAVXState(Ref MemBase);
void AVX128_RestoreAVXState(Ref MemBase);
void AVX128_DefaultAVXState();
void AVX128_VPERM2(OpcodeArgs);
void AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VTESTP(OpcodeArgs);
void AVX128_PTest(OpcodeArgs);
void AVX128_VPERMILReg(OpcodeArgs, IR::OpSize ElementSize);
template<IR::OpSize ElementSize>
void AVX128_VPERMILReg(OpcodeArgs);
void AVX128_VPERMD(OpcodeArgs);
@@ -1101,7 +1120,8 @@ public:
RefPair AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB);
RefPair AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
void AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize);
template<OpSize AddrElementSize>
void AVX128_VPGATHER(OpcodeArgs);
void AVX128_VCVTPH2PS(OpcodeArgs);
void AVX128_VCVTPS2PH(OpcodeArgs);
@@ -1135,14 +1155,9 @@ public:
StoreXMMRegister(XMM, Value);
}
void AVXVectorALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
void AVXVectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
void AVXVectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize);
// End of AVX 256-bit implementation
void InvalidOp(OpcodeArgs);
void NoExecOp(OpcodeArgs);
void SetPackedRFLAG(bool Lower8, Ref Src);
Ref GetPackedRFLAG(uint32_t FlagsMask = ~0U);
@@ -1161,44 +1176,15 @@ public:
}
}
void StoreContextHelper(IR::OpSize Size, RegisterClassType Class, Ref Value, uint32_t Offset) {
// For i128Bit, we won't see a normal Constant to inline, but as a special
// case we can replace with a 2x64-bit store which can use inline zeroes.
if (Size == OpSize::i128Bit) {
auto Header = GetOpHeader(WrapNode(Value));
const auto MAX_STP_OFFSET = (252 * 4);
if (Offset <= MAX_STP_OFFSET && Header->Op == OP_LOADNAMEDVECTORCONSTANT) {
auto Const = Header->C<IR::IROp_LoadNamedVectorConstant>();
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
Ref Zero = _Constant(0);
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Offset);
// XXX: This works around InlineConstant not having an associated
// register class, else we'd just do InlineConstant above.
Ref InlineZero = _InlineConstant(0);
ReplaceNodeArgument(STP, 0, InlineZero);
ReplaceNodeArgument(STP, 1, InlineZero);
return;
}
}
}
_StoreContext(Size, Class, Value, Offset);
}
void FlushRegisterCache(bool SRAOnly = false, bool MMXOnly = false) {
void FlushRegisterCache(bool SRAOnly = false) {
// At block boundaries, fix up the carry flag.
if (!SRAOnly) {
RectifyCarryInvert(CFInvertedABI);
}
if (!MMXOnly) {
CalculateDeferredFlags();
}
CalculateDeferredFlags();
const auto GPRSize = GetGPROpSize();
const auto GPRSize = CTX->GetGPROpSize();
const auto VectorSize = GetGuestVectorLength();
// Write backwards. This is a heuristic to improve coalescing, since we
@@ -1218,11 +1204,6 @@ public:
Bits &= Mask;
}
if (MMXOnly) {
Mask &= ((1ull << (MM7Index - MM0Index + 1)) - 1) << MM0Index;
Bits &= Mask;
}
while (Bits != 0) {
uint32_t Index = 63 - std::countl_zero(Bits);
Ref Value = RegCache.Value[Index];
@@ -1260,10 +1241,10 @@ public:
_StoreContextPair(Size, Class, ValueNext, Value, Offset - SizeInt);
Bits &= ~NextBit;
} else {
StoreContextHelper(Size, Class, Value, Offset);
_StoreContext(Size, Class, Value, Offset);
// If Partial and MMX register, then we need to store all 1s in bits 64-80
if (Partial && Index >= MM0Index && Index <= MM7Index) {
_StoreContext(OpSize::i16Bit, IR::GPRClass, Constant(0xFFFF), Offset + 8);
_StoreContext(OpSize::i16Bit, IR::GPRClass, _Constant(0xFFFF), Offset + 8);
}
}
}
@@ -1276,10 +1257,6 @@ public:
RegCache.Partial &= ~Mask;
}
IR::OpSize GetGPROpSize() const {
return Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
}
protected:
void RecordX87Use() override {
CurrentHeader->HasX87 = true;
@@ -1333,7 +1310,6 @@ private:
struct JumpTargetInfo {
Ref BlockEntry;
bool HaveEmitted;
bool IsEntryPoint;
};
FEXCore::Context::ContextImpl* CTX {};
@@ -1382,6 +1358,11 @@ private:
Ref ADDSUBPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1, Ref Src2);
void AVXVectorALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
void AVXVectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
void AVXVectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize);
void AVXVariableShiftImpl(OpcodeArgs, IROps IROp);
Ref AESKeyGenAssistImpl(OpcodeArgs);
@@ -1514,7 +1495,7 @@ private:
#undef OpcodeArgs
Ref AppendSegmentOffset(Ref Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
Ref GetSegment(uint32_t Flags, uint32_t DefaultPrefix = FEXCore::X86Tables::DecodeFlags::FLAG_NO_PREFIX, bool Override = false);
Ref GetSegment(uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
void UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg);
@@ -1522,32 +1503,14 @@ private:
void StoreGPRRegister(uint32_t GPR, const Ref Src, IR::OpSize Size = OpSize::iInvalid, uint8_t Offset = 0);
void StoreXMMRegister(uint32_t XMM, const Ref Src);
Ref _GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, bool Inline) {
const auto GPRSize = GetGPROpSize();
const auto Offs = Op->PC + Op->InstSize + Offset - Entry;
return Inline ? _InlineEntrypointOffset(GPRSize, Offs) : _EntrypointOffset(GPRSize, Offs);
}
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0);
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
return _GetRelocatedPC(Op, Offset, false);
}
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */));
}
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock);
}
[[nodiscard]]
static bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
// Literals are immediates as sources but memory addresses as destinations.
return !(Load && Operand.IsLiteral()) && !Operand.IsGPR();
}
[[nodiscard]]
static bool IsNonTSOReg(MemoryAccessType Access, uint8_t Reg) {
bool IsNonTSOReg(MemoryAccessType Access, uint8_t Reg) {
return Access == MemoryAccessType::DEFAULT && Reg == X86State::REG_RSP;
}
@@ -1654,7 +1617,7 @@ private:
}
void ZeroNZCV() {
CachedNZCV = Constant(0);
CachedNZCV = _Constant(0);
NZCVDirty = true;
}
@@ -1667,9 +1630,9 @@ private:
// This is currently worse for 8/16-bit, but that should be optimized. TODO
if (SrcSize >= OpSize::i32Bit) {
if (SetPF) {
CalculatePF(SubWithFlags(SrcSize, Res, (uint64_t)0));
CalculatePF(_SubWithFlags(SrcSize, Res, _Constant(0)));
} else {
_SubNZCV(SrcSize, Res, Constant(0));
_SubNZCV(SrcSize, Res, _Constant(0));
}
CFInverted = true;
@@ -1740,7 +1703,7 @@ private:
} else {
// Invert as a GPR
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), Constant(1u << Bit)));
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), _Constant(1u << Bit)));
CalculateDeferredFlags();
}
@@ -1778,7 +1741,7 @@ private:
}
HandleNZCVWrite();
_SubNZCV(OpSize::i32Bit, Constant(0), Value);
_SubNZCV(OpSize::i32Bit, _Constant(0), Value);
CFInverted = true;
}
@@ -1803,25 +1766,25 @@ private:
StoreRegister(Core::CPUState::AF_AS_GREG, false, Value);
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
// For DF, we need to transform 0/1 into 1/-1
StoreDF(_SubShift(OpSize::i64Bit, Constant(1), Value, ShiftType::LSL, 1));
StoreDF(_SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1));
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
auto PackedTF = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
// unblocked bit before raising the exception.
auto NewPackedTF = _Select(FEXCore::IR::COND_EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
auto NewPackedTF = _Select(FEXCore::IR::COND_EQ, Value, _Constant(0), _And(OpSize::i32Bit, PackedTF, _Constant(~1)), _Constant(1));
_StoreContext(OpSize::i8Bit, GPRClass, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
} else {
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
}
}
void SetAF(unsigned K) {
void SetAF(unsigned Constant) {
// AF is stored in bit 4 of the AF flag byte, with garbage in the other
// bits. This allows us to defer the extract in the usual case. When it is
// read, bit 4 is extracted. In order to write a constant value of AF, that
// means we need to left-shift here to compensate.
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(K << 4));
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(Constant << 4));
}
void ZeroPF_AF();
@@ -1837,8 +1800,7 @@ private:
InvalidateReg(Core::CPUState::AF_AS_GREG);
}
[[nodiscard]]
static CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
switch (BitOffset) {
case X86State::RFLAG_SF_RAW_LOC: return {Invert ? COND_PL : COND_MI};
case X86State::RFLAG_ZF_RAW_LOC: return {Invert ? COND_NEQ : COND_EQ};
@@ -1854,27 +1816,26 @@ private:
static const int PFIndex = 16;
static const int AFIndex = 17;
/* Gap 18..19 */
/* Note this range is only valid if MMXState = MMXState_MMX */
static const int MM0Index = 20;
static const int MM7Index = 27;
/* Gap 28..30 */
static const int AbridgedFTWIndex = 28;
/* Gap 29..30 */
static const int DFIndex = 31;
static const int FPR0Index = 32;
static const int FPR15Index = 47;
static const int AVXHigh0Index = 48;
static const int AVXHigh15Index = 63;
[[nodiscard]]
static uint32_t CacheIndexToContextOffset(int Index) {
uint32_t CacheIndexToContextOffset(int Index) {
switch (Index) {
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
default: return ~0U;
}
}
[[nodiscard]]
static RegisterClassType CacheIndexClass(int Index) {
RegisterClassType CacheIndexClass(int Index) {
if ((Index >= MM0Index && Index <= MM7Index) || Index >= FPR0Index) {
return FPRClass;
} else {
@@ -1882,8 +1843,7 @@ private:
}
}
[[nodiscard]]
static IR::OpSize CacheIndexToOpSize(int Index) {
IR::OpSize CacheIndexToOpSize(int Index) {
// MMX registers are rounded up to 128-bit since they are shared with 80-bit
// x87 registers, even though MMX is logically only 64-bit.
if (Index >= AVXHigh0Index || ((Index >= MM0Index && Index <= MM7Index))) {
@@ -1930,7 +1890,7 @@ private:
if (!(RegCache.Cached & Bit)) {
if (Index == DFIndex) {
RegCache.Value[Index] = _LoadDF();
} else if ((Index >= MM0Index && Index <= MM7Index) || Index >= AVXHigh0Index) {
} else if ((Index >= MM0Index && Index <= AbridgedFTWIndex) || Index >= AVXHigh0Index) {
RegCache.Value[Index] = _LoadContext(Size, RegClass, Offset);
// We may have done a partial load, this requires special handling.
@@ -1991,7 +1951,7 @@ private:
}
Ref LoadGPR(uint8_t Reg) {
return LoadRegCache(Reg, GPR0Index + Reg, GPRClass, GetGPROpSize());
return LoadRegCache(Reg, GPR0Index + Reg, GPRClass, CTX->GetGPROpSize());
}
Ref LoadContext(IR::OpSize Size, uint8_t Index) {
@@ -2045,14 +2005,14 @@ private:
auto Value = _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV());
if (Invert) {
return _Xor(OpSize::i32Bit, Value, Constant(1));
return _Xor(OpSize::i32Bit, Value, _Constant(1));
} else {
return Value;
}
} else {
// Because we explicitly inverted for CF above, we use the unsafe
// _NZCVSelect rather than the safe CF-aware version.
return _NZCVSelect01(CondForNZCVBit(BitOffset, Invert));
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert), _Constant(1), _Constant(0));
}
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
return LoadGPR(Core::CPUState::PF_AS_GREG);
@@ -2060,7 +2020,7 @@ private:
return LoadGPR(Core::CPUState::AF_AS_GREG);
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
// Recover the sign bit, it is the logical DF value
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63));
return _Lshr(OpSize::i64Bit, LoadDF(), _Constant(63));
} else {
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(Core::CPUState, flags[BitOffset]));
}
@@ -2132,7 +2092,7 @@ private:
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
// byte to zero will indeed zero AF as intended.
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(0));
}
// Convert NZCV from the Arm representation to an eXternal representation
@@ -2149,7 +2109,7 @@ private:
void ConvertNZCVToX87() {
LOGMAN_THROW_A_FMT(NZCVDirty && CachedNZCV, "NZCV must be saved");
Ref V = _NZCVSelect01(CondForNZCVBit(FEXCore::X86State::RFLAG_OF_RAW_LOC, false));
Ref V = _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(FEXCore::X86State::RFLAG_OF_RAW_LOC, false), _Constant(1), _Constant(0));
if (CTX->HostFeatures.SupportsFlagM2) {
// Convert to x86 flags, saves us from or'ing after.
@@ -2157,8 +2117,8 @@ private:
}
// CF is inverted after FCMP
Ref C = _NZCVSelect01(CondForNZCVBit(FEXCore::X86State::RFLAG_CF_RAW_LOC, true));
Ref Z = _NZCVSelect01(CondForNZCVBit(FEXCore::X86State::RFLAG_ZF_RAW_LOC, false));
Ref C = _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(FEXCore::X86State::RFLAG_CF_RAW_LOC, true), _Constant(1), _Constant(0));
Ref Z = _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(FEXCore::X86State::RFLAG_ZF_RAW_LOC, false), _Constant(1), _Constant(0));
if (!CTX->HostFeatures.SupportsFlagM2) {
C = _Or(OpSize::i32Bit, C, V);
@@ -2166,7 +2126,7 @@ private:
}
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(C);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(V);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Z);
}
@@ -2216,9 +2176,9 @@ private:
return CachedNamedVectorConstants[NamedConstant][log2_size_bytes];
}
auto K = _LoadNamedVectorConstant(Size, NamedConstant);
CachedNamedVectorConstants[NamedConstant][log2_size_bytes] = K;
return K;
auto Constant = _LoadNamedVectorConstant(Size, NamedConstant);
CachedNamedVectorConstants[NamedConstant][log2_size_bytes] = Constant;
return Constant;
}
Ref LoadAndCacheIndexedNamedVectorConstant(IR::OpSize Size, FEXCore::IR::IndexNamedVectorConstant NamedIndexedConstant, uint32_t Index) {
IndexNamedVectorMapKey Key {
@@ -2232,9 +2192,9 @@ private:
return it->second;
}
auto K = _LoadNamedVectorIndexedConstant(Size, NamedIndexedConstant, Index);
CachedIndexedNamedVectorConstants.insert_or_assign(Key, K);
return K;
auto Constant = _LoadNamedVectorIndexedConstant(Size, NamedIndexedConstant, Index);
CachedIndexedNamedVectorConstants.insert_or_assign(Key, Constant);
return Constant;
}
Ref LoadUncachedZeroVector(IR::OpSize Size) {
@@ -2252,9 +2212,9 @@ private:
CachedIndexedNamedVectorConstants.clear();
}
std::optional<CondClassType> DecodeNZCVCondition(uint8_t OP);
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP);
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
Ref SelectCC0All1(uint8_t OP);
Ref SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
/**
* @brief Flushes NZCV. Mostly vestigial.
@@ -2289,7 +2249,7 @@ private:
}
// Otherwise, prepare to branch.
auto Zero = Constant(0);
auto Zero = _Constant(0);
// If the shift is zero, do not touch the flags.
auto SetBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
@@ -2329,6 +2289,7 @@ private:
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
* @{ */
Ref LoadPFRaw(bool Mask, bool Invert);
Ref SelectPF(bool Invert, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
Ref LoadAF();
void FixupAF();
void SetAFAndFixup(Ref AF);
@@ -2364,19 +2325,17 @@ private:
void ChgStateX87_MMX() override {
LOGMAN_THROW_A_FMT(MMXState == MMXState_X87, "Expected state to be x87");
_StackForceSlow();
SetX87Top(Constant(0)); // top reset to zero
_StoreContext(OpSize::i8Bit, GPRClass, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
SetX87Top(_Constant(0)); // top reset to zero
StoreContext(AbridgedFTWIndex, _Constant(0xFFFFUL)); // all valid
MMXState = MMXState_MMX;
}
void ChgStateMMX_X87() override {
LOGMAN_THROW_A_FMT(MMXState == MMXState_MMX, "Expected state to be MMX");
// The opcode dispatcher register cache is used for MMX, but the x87 pass register cache is used for x87, spill to
// context to ensure coherence.
FlushRegisterCache(false, true);
// We explicitly initialize to x87 state in StartNewBlock.
// So if we ever change this to do something else, we need to
// make sure that we consider if we need to explicitly set it there.
FlushRegisterCache();
MMXState = MMXState_X87;
}
@@ -2392,17 +2351,10 @@ private:
bool BlockSetRIP {false};
bool Multiblock {};
bool Is64BitMode {};
uint64_t Entry {};
// Set if mono hacks are enabled and the current block is the mono callsite backpatcher, in which case the
// XCHG ops that would patch code are replaced with a hook that performs the write and manually invalidates
// the target address.
bool IsMonoBackpatcherBlock {false};
IROp_IRHeader* CurrentHeader {};
[[nodiscard]]
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) const {
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) {
if (ForceTSO == ForceTSOMode::ForceEnabled) {
return true;
} else if (ForceTSO == ForceTSOMode::ForceDisabled) {
@@ -2432,7 +2384,7 @@ private:
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
if (AtomicTSO) {
return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
@@ -2452,7 +2404,7 @@ private:
A.Offset = 0;
}
Out.Base = LoadEffectiveAddress(this, A, GetGPROpSize(), true, false);
Out.Base = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true, false);
return Out;
}
@@ -2483,7 +2435,7 @@ private:
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
if (AtomicTSO) {
return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
@@ -2534,7 +2486,7 @@ private:
void Push(IR::OpSize Size, Ref Value) {
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
auto NewSP = _Push(GetGPROpSize(), Size, Value, OldSP);
auto NewSP = _Push(CTX->GetGPROpSize(), Size, Value, OldSP);
StoreGPRRegister(X86State::REG_RSP, NewSP);
FlushRegisterCache();
}
@@ -2564,11 +2516,11 @@ private:
}
ArithRef And(uint64_t K) {
return IsConstant ? ArithRef(E, C & K) : ArithRef(E, E->_And(OpSize::i64Bit, R, E->Constant(K)));
return IsConstant ? ArithRef(E, C & K) : ArithRef(E, E->_And(OpSize::i64Bit, R, E->_Constant(K)));
}
ArithRef Presub(uint64_t K) {
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->Sub(OpSize::i64Bit, E->Constant(K), R));
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->_Sub(OpSize::i64Bit, E->_Constant(K), R));
}
ArithRef Lshl(uint64_t Shift) {
@@ -2577,7 +2529,7 @@ private:
} else if (IsConstant) {
return ArithRef(E, C << Shift);
} else {
return ArithRef(E, E->_Lshl(OpSize::i64Bit, R, E->Constant(Shift)));
return ArithRef(E, E->_Lshl(OpSize::i64Bit, R, E->_Constant(Shift)));
}
}
@@ -2617,7 +2569,7 @@ private:
}
if (IsConstant) {
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, E->Constant(C));
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, E->_Constant(C));
} else {
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, R);
}
@@ -2633,15 +2585,15 @@ private:
return ArithRef(E, Result);
} else {
return ArithRef(E, E->_Lshl(Size, E->Constant(1), R));
return ArithRef(E, E->_Lshl(Size, E->_Constant(1), R));
}
}
Ref Ref() {
return IsConstant ? E->Constant(C) : R;
return IsConstant ? E->_Constant(C) : R;
}
bool IsDefinitelyZero() const {
bool IsDefinitelyZero() {
return IsConstant && C == 0;
}
};
@@ -2660,6 +2612,8 @@ private:
return ArithRef(this, K);
}
void InstallHostSpecificOpcodeHandlers();
///< Segment telemetry tracking
uint32_t SegmentsNeedReadCheck {~0U};
void CheckLegacySegmentWrite(Ref NewNode, uint32_t SegmentReg);
@@ -2668,19 +2622,21 @@ private:
constexpr inline void InstallToTable(auto& FinalTable, const auto& LocalTable) {
for (const auto& Op : LocalTable) {
auto OpNum = Op.Op;
auto Dispatcher = Op.Ptr;
for (uint8_t i = 0; i < Op.Count; ++i) {
auto OpNum = std::get<0>(Op);
auto Dispatcher = std::get<2>(Op);
for (uint8_t i = 0; i < std::get<1>(Op); ++i) {
auto& TableOp = FinalTable[OpNum + i];
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
if (TableOp.OpcodeDispatcher.OpDispatch) {
if (TableOp.OpcodeDispatcher) {
ERROR_AND_DIE_FMT("Duplicate Entry {}", TableOp.Name);
}
#endif
TableOp.OpcodeDispatcher.OpDispatch = Dispatcher;
TableOp.OpcodeDispatcher = Dispatcher;
}
}
}
void InstallOpcodeHandlers(Context::OperatingMode Mode);
} // namespace FEXCore::IR
File diff suppressed because it is too large. Load diff
@@ -3,7 +3,7 @@
#include "Interface/Core/OpcodeDispatcher.h"
namespace FEXCore::IR {
constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
constexpr inline std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_BaseOpTable[] = {
// Instructions
{0x00, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0>},
@@ -53,11 +53,10 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
{0xAA, 2, &OpDispatchBuilder::STOSOp},
{0xAC, 2, &OpDispatchBuilder::LODSOp},
{0xAE, 2, &OpDispatchBuilder::SCASOp},
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPRImmediate>},
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
{0xC2, 2, &OpDispatchBuilder::RETOp},
{0xC8, 1, &OpDispatchBuilder::EnterOp},
{0xC9, 1, &OpDispatchBuilder::LEAVEOp},
{0xCA, 2, &OpDispatchBuilder::RETFARIndirectOp},
{0xCC, 2, &OpDispatchBuilder::INTOp},
{0xCF, 1, &OpDispatchBuilder::IRETOp},
{0xD7, 2, &OpDispatchBuilder::XLATOp},
@@ -76,4 +75,33 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
{0xFA, 2, &OpDispatchBuilder::PermissionRestrictedOp},
{0xFC, 2, &OpDispatchBuilder::FLAGControlOp},
};
constexpr inline std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_BaseOpTable_64[] = {
{0x63, 1, &OpDispatchBuilder::MOVSXDOp},
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
};
constexpr inline std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> OpDispatch_BaseOpTable_32[] = {
{0x06, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
{0x07, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
{0x0E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX>},
{0x16, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
{0x17, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
{0x1E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
{0x1F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
{0x27, 1, &OpDispatchBuilder::DAAOp},
{0x2F, 1, &OpDispatchBuilder::DASOp},
{0x37, 1, &OpDispatchBuilder::AAAOp},
{0x3F, 1, &OpDispatchBuilder::AASOp},
{0x40, 8, &OpDispatchBuilder::INCOp},
{0x48, 8, &OpDispatchBuilder::DECOp},
{0x60, 1, &OpDispatchBuilder::PUSHAOp},
{0x61, 1, &OpDispatchBuilder::POPAOp},
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
{0xCE, 1, &OpDispatchBuilder::INTOp},
{0xD4, 1, &OpDispatchBuilder::AAMOp},
{0xD5, 1, &OpDispatchBuilder::AADOp},
{0xD6, 1, &OpDispatchBuilder::SALCOp},
};
} // namespace FEXCore::IR
@@ -11,7 +11,10 @@ $end_info$
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Core/OpcodeDispatcher.h"
#include <array>
#include <cstdint>
#include <tuple>
#include <utility>
namespace FEXCore::IR {
class OrderedNode;
@@ -19,20 +22,24 @@ class OrderedNode;
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsSHA) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
auto RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
Ref RotatedNode {};
if (CTX->HostFeatures.SupportsSHA) {
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
} else {
// SHA1 extension missing, manually rotate.
// Emulate rotate.
auto ShiftLeft = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Dest, 30);
RotatedNode = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeft, Dest, 2);
}
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
@@ -40,10 +47,6 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
}
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsSHA) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
@@ -56,145 +59,334 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
}
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsSHA) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
auto Src1 = SHADataShuffle(Dest);
auto Src2 = SHADataShuffle(Src);
Ref Result;
if (CTX->HostFeatures.SupportsSHA) {
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
auto Src1 = SHADataShuffle(Dest);
auto Src2 = SHADataShuffle(Src);
// The result is swizzled differently than expected
Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
} else {
// Shift the incoming source left by a 32-bit element, inserting Zeros.
// This could be slightly improved to use a VInsGPR with the zero register.
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
// Emulate rotate.
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
// Element0 didn't get XOR'd with anything, so do it now.
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
// Emulate rotate.
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
}
// The result is swizzled differently than expected
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsSHA) {
UnimplementedOp(Op);
return;
}
using FnType = Ref (*)(OpDispatchBuilder&, Ref, Ref, Ref);
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1c?
return Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._Andn(OpSize::i32Bit, D, B));
};
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p with different key
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
};
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1m
return Self.BitwiseAtLeastTwo(B, C, D);
};
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
};
constexpr std::array<uint32_t, 4> k_array {
0x5A827999U,
0x6ED9EBA1U,
0x8F1BBCDCU,
0xCA62C1D6U,
};
constexpr std::array<FnType, 4> fn_array {
f0,
f1,
f2,
f3,
};
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result {};
Ref ConstantVector {};
switch (Imm8) {
case 0:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0);
break;
case 1:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1);
break;
case 2:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2);
break;
case 3:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3);
break;
}
if (CTX->HostFeatures.SupportsSHA) {
Ref ConstantVector {};
switch (Imm8) {
case 0:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0);
break;
case 1:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1);
break;
case 2:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2);
break;
case 3:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3);
break;
}
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
Ref Src1 = SHADataShuffle(Dest);
Ref Src2 = SHADataShuffle(Src);
Src2 = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src2, ConstantVector);
Ref Src1 = SHADataShuffle(Dest);
Ref Src2 = SHADataShuffle(Src);
Src2 = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src2, ConstantVector);
switch (Imm8) {
case 0: Result = SHADataShuffle(_VSha1C(Src1, ZeroRegister, Src2)); break;
case 2: Result = SHADataShuffle(_VSha1M(Src1, ZeroRegister, Src2)); break;
case 1:
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
switch (Imm8) {
case 0: Result = SHADataShuffle(_VSha1C(Src1, ZeroRegister, Src2)); break;
case 2: Result = SHADataShuffle(_VSha1M(Src1, ZeroRegister, Src2)); break;
case 1:
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
}
} else {
const FnType Fn = fn_array[Imm8];
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
const auto Round0 = [&]() -> RoundResult {
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
auto A1 =
_Add(OpSize::i32Bit,
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
auto B1 = A;
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
auto D1 = C;
auto E1 = D;
return {A1, B1, C1, D1, E1};
};
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
// Kill W and E at the beginning
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
auto Q = _Add(OpSize::i32Bit, W, E);
auto ANext =
_Add(OpSize::i32Bit,
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
auto BNext = A;
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
auto DNext = C;
auto ENext = D;
return {ANext, BNext, CNext, DNext, ENext};
};
auto [A1, B1, C1, D1, E1] = Round0();
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
}
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsSHA) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto Result = _VSha256U0(Dest, Src);
Ref Result {};
if (CTX->HostFeatures.SupportsSHA) {
Result = _VSha256U0(Dest, Src);
} else {
const auto Sigma0 = [this](Ref W) -> Ref {
return _Xor(
OpSize::i32Bit,
_Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 7)), _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 18))),
_Lshr(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 3)));
};
auto W4 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
auto W3 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto W2 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
auto W1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
auto W0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, Sig3);
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, Sig2);
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, Sig1);
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, Sig0);
}
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsSHA) {
UnimplementedOp(Op);
return;
}
const auto Sigma1 = [this](Ref W) -> Ref {
return _Xor(
OpSize::i32Bit,
_Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 17)), _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 19))),
_Lshr(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 10)));
};
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Src2 = _VZip2(OpSize::i128Bit, OpSize::i64Bit, DupDst, Src);
Ref Result;
if (CTX->HostFeatures.SupportsSHA) {
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Src2 = _VZip2(OpSize::i128Bit, OpSize::i64Bit, DupDst, Src);
auto Result = _VSha256U1(Src1, Src2);
Result = _VSha256U1(Src1, Src2);
} else {
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
}
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
// Returns whether at least 2/3 of A/B/C is true.
// Expressed as (A & (B | C)) | (B & C)
//
// Equivalent to expression in SHA calculations: (A & B) ^ (A & C) ^ (B & C)
auto And = _And(OpSize::i32Bit, B, C);
auto Or = _Or(OpSize::i32Bit, B, C);
return _Or(OpSize::i32Bit, _And(OpSize::i32Bit, A, Or), And);
}
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsSHA) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// Hardcoded to XMM0
auto XMM0 = LoadXMMRegister(0);
auto shuffle_abcd = [this](Ref Src1, Ref Src2) -> Ref {
// Generates a suitable SHA256 `abcd` configuration from x86 format.
auto Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
};
Ref Result;
if (CTX->HostFeatures.SupportsSHA) {
auto shuffle_abcd = [this](Ref Src1, Ref Src2) -> Ref {
// Generates a suitable SHA256 `abcd` configuration from x86 format.
auto Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
};
auto shuffle_efgh = [this](Ref Src1, Ref Src2) -> Ref {
// Generates a suitable SHA256 `efgh` configuration from x86 format.
auto Tmp = _VZip(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
};
auto shuffle_efgh = [this](Ref Src1, Ref Src2) -> Ref {
// Generates a suitable SHA256 `efgh` configuration from x86 format.
auto Tmp = _VZip(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
};
auto ABCD = shuffle_abcd(Dest, Src);
auto EFGH = shuffle_efgh(Dest, Src);
auto ABCD = shuffle_abcd(Dest, Src);
auto EFGH = shuffle_efgh(Dest, Src);
// x86 uses only the bottom 64-bits of the key, so duplicate to match ARM64 semantics.
auto Key = _VDupElement(OpSize::i128Bit, OpSize::i64Bit, XMM0, 0);
// x86 uses only the bottom 64-bits of the key, so duplicate to match ARM64 semantics.
auto Key = _VDupElement(OpSize::i128Bit, OpSize::i64Bit, XMM0, 0);
auto A = _VSha256H(ABCD, EFGH, Key);
auto B = _VSha256H2(EFGH, ABCD, Key);
auto Result = shuffle_abcd(A, B);
auto A = _VSha256H(ABCD, EFGH, Key);
auto B = _VSha256H2(EFGH, ABCD, Key);
Result = shuffle_abcd(A, B);
} else {
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
};
const auto Sigma0 = [this](Ref A) -> Ref {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
A, ShiftType::ROR, 22);
};
const auto Sigma1 = [this](Ref E) -> Ref {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
E, ShiftType::ROR, 25);
};
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
Q0 = _Add(OpSize::i32Bit, Q0, H0);
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
Q1 = _Add(OpSize::i32Bit, Q1, G0);
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
// Rematerialize C0. As with G0.
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
}
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESImc(Src);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
@@ -203,7 +395,7 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is128Bit = DstSize == OpSize::i128Bit;
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESENC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
@@ -216,10 +408,6 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
}
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
@@ -228,7 +416,7 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is128Bit = DstSize == OpSize::i128Bit;
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESENCLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
@@ -241,10 +429,6 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
}
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
@@ -253,7 +437,7 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is128Bit = DstSize == OpSize::i128Bit;
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESDEC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
@@ -266,10 +450,6 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
}
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
@@ -278,7 +458,7 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is128Bit = DstSize == OpSize::i128Bit;
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESDECLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
@@ -299,20 +479,11 @@ Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
}
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Result = AESKeyGenAssistImpl(Op);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsPMULL_128Bit) {
UnimplementedOp(Op);
return;
}
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
@@ -322,10 +493,6 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
}
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
if (!CTX->HostFeatures.SupportsPMULL_128Bit) {
UnimplementedOp(Op);
return;
}
const auto DstSize = OpSizeFromDst(Op);
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
@@ -3,7 +3,7 @@
#include "Interface/Core/OpcodeDispatcher.h"
namespace FEXCore::IR {
constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_DDDTable[] = {
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
{0x0D, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
@@ -28,7 +28,7 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
void OpDispatchBuilder::ZeroPF_AF() {
// PF is stored inverted, so invert it when we zero.
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Constant(1));
SetAF(0);
}
@@ -201,7 +201,7 @@ void OpDispatchBuilder::FixupAF() {
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
// Again 64-bit as masking is more expensive.
// Again 64-bit as masking is more expensive given our ConstProp design.
Ref XorRes = _Xor(OpSize::i64Bit, AFRaw, PFRaw);
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
}
@@ -247,7 +247,7 @@ void OpDispatchBuilder::CalculateAF(Ref Src1, Ref Src2) {
// We store the XOR of the arguments. At read time, we XOR with the
// appropriate bit of the result (available as the PF flag) and extract the
// appropriate bit. Again 64-bit to avoid masking.
Ref XorRes = Src1 == Src2 ? Constant(0) : _Xor(OpSize::i64Bit, Src1, Src2);
Ref XorRes = Src1 == Src2 ? _Constant(0) : _Xor(OpSize::i64Bit, Src1, Src2);
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
}
@@ -267,6 +267,8 @@ Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
}
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
auto Zero = _InlineConstant(0);
auto One = _InlineConstant(1);
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
Ref Res;
@@ -286,11 +288,11 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
Ref Src2PlusCF = IncrementByCarry(OpSize, Src2);
// Need to zero-extend for the comparison.
Res = Add(OpSize, Src1, Src2PlusCF);
Res = _Add(OpSize, Src1, Src2PlusCF);
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
// TODO: We can fold that second Bfe in (cmp uxth).
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Res, Src2PlusCF);
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Res, Src2PlusCF, One, Zero);
SetNZ_ZeroCV(SrcSize, Res);
SetCFInverted(SelectCFInv);
@@ -302,6 +304,8 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
}
Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
auto Zero = _InlineConstant(0);
auto One = _InlineConstant(1);
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
CalculateAF(Src1, Src2);
@@ -321,10 +325,10 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
Res = Sub(OpSize, Src1, Src2PlusCF);
Res = _Sub(OpSize, Src1, Src2PlusCF);
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Src1, Src2PlusCF);
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Src1, Src2PlusCF, One, Zero);
SetNZ_ZeroCV(SrcSize, Res);
SetCFInverted(SelectCFInv);
@@ -345,10 +349,10 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(IR::OpSize SrcSize, Ref Src1, Ref Src2
Ref Res;
if (SrcSize >= OpSize::i32Bit) {
Res = SubWithFlags(SrcSize, Src1, Src2);
Res = _SubWithFlags(SrcSize, Src1, Src2);
} else {
_SubNZCV(SrcSize, Src1, Src2);
Res = Sub(OpSize::i32Bit, Src1, Src2);
Res = _Sub(OpSize::i32Bit, Src1, Src2);
}
CalculatePF(Res);
@@ -375,10 +379,10 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(IR::OpSize SrcSize, Ref Src1, Ref Src2
Ref Res;
if (SrcSize >= OpSize::i32Bit) {
Res = AddWithFlags(SrcSize, Src1, Src2);
Res = _AddWithFlags(SrcSize, Src1, Src2);
} else {
_AddNZCV(SrcSize, Src1, Src2);
Res = Add(OpSize::i32Bit, Src1, Src2);
Res = _Add(OpSize::i32Bit, Src1, Src2);
}
CalculatePF(Res);
@@ -6,10 +6,9 @@ namespace FEXCore::IR {
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
constexpr uint16_t PF_38_NONE = 0;
constexpr uint16_t PF_38_66 = (1U << 0);
constexpr uint16_t PF_38_F2 = (1U << 1);
constexpr uint16_t PF_38_F3 = (1U << 2);
constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F38Table[] = {
{OPD(PF_38_NONE, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
{OPD(PF_38_66, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
{OPD(PF_38_NONE, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, OpSize::i16Bit>},
@@ -72,28 +71,9 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
{OPD(PF_38_66, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, OpSize::i32Bit>},
{OPD(PF_38_66, 0x41), 1, &OpDispatchBuilder::PHMINPOSUWOp},
{OPD(PF_38_NONE, 0xC8), 1, &OpDispatchBuilder::SHA1NEXTEOp},
{OPD(PF_38_NONE, 0xC9), 1, &OpDispatchBuilder::SHA1MSG1Op},
{OPD(PF_38_NONE, 0xCA), 1, &OpDispatchBuilder::SHA1MSG2Op},
{OPD(PF_38_NONE, 0xCB), 1, &OpDispatchBuilder::SHA256RNDS2Op},
{OPD(PF_38_NONE, 0xCC), 1, &OpDispatchBuilder::SHA256MSG1Op},
{OPD(PF_38_NONE, 0xCD), 1, &OpDispatchBuilder::SHA256MSG2Op},
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
{OPD(PF_38_66, 0xDC), 1, &OpDispatchBuilder::AESEncOp},
{OPD(PF_38_66, 0xDD), 1, &OpDispatchBuilder::AESEncLastOp},
{OPD(PF_38_66, 0xDE), 1, &OpDispatchBuilder::AESDecOp},
{OPD(PF_38_66, 0xDF), 1, &OpDispatchBuilder::AESDecLastOp},
{OPD(PF_38_NONE, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
{OPD(PF_38_66, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
{OPD(PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
{OPD(PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
{OPD(PF_38_66 | PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
{OPD(PF_38_66 | PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
{OPD(PF_38_66, 0xF6), 1, &OpDispatchBuilder::ADXOp},
{OPD(PF_38_F3, 0xF6), 1, &OpDispatchBuilder::ADXOp},
};
@@ -8,7 +8,7 @@ namespace FEXCore::IR {
#define PF_3A_66 1
constexpr auto OpDispatchTableGenH0F3A = []() consteval {
constexpr auto OpDispatchTableGenH0F3AREX = []<uint16_t REX>() consteval {
constexpr DispatchTableEntry Table[] = {
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> Table[] = {
{OPD(REX, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i32Bit>},
@@ -29,7 +29,6 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
{OPD(REX, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
{OPD(REX, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
@@ -37,16 +36,14 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
};
return std::to_array(Table);
};
auto REX0 = OpDispatchTableGenH0F3AREX.template operator()<0>();
auto REX1 = OpDispatchTableGenH0F3AREX.template operator()<1>();
auto concat = []<typename T, size_t N1, size_t N2>(const std::array<T, N1>& lhs,
const std::array<T, N2>& rhs) consteval -> std::array<T, N1 + N2> {
auto concat = []<typename T, size_t N1, size_t N2>(std::array<T, N1> const& lhs,
std::array<T, N2> const& rhs) consteval -> std::array<T, N1 + N2> {
std::array<T, N1 + N2> Table {};
for (size_t i = 0; i < N1; ++i) {
Table[i] = lhs[i];
@@ -63,11 +60,16 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
constexpr auto OpDispatch_H0F3ATableIgnoreREX = OpDispatchTableGenH0F3A();
constexpr DispatchTableEntry OpDispatch_H0F3ATableNeedsREX0[] = {
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATableNeedsREX0[] = {
{OPD(0, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i32Bit>},
};
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable_64[] = {
{OPD(1, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i64Bit>},
{OPD(1, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i64Bit>},
};
#undef PF_3A_NONE
#undef PF_3A_66
@@ -5,7 +5,7 @@
namespace FEXCore::IR {
using X86Tables::OpToIndex;
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_PrimaryGroupTables[] = {
// GROUP 1
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
@@ -117,9 +117,7 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 0), 1, &OpDispatchBuilder::INCOp}, // INC
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 1), 1, &OpDispatchBuilder::DECOp}, // DEC
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 2), 1, &OpDispatchBuilder::CALLAbsoluteOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 3), 1, &OpDispatchBuilder::CALLFARIndirectOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 4), 1, &OpDispatchBuilder::JUMPAbsoluteOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 5), 1, &OpDispatchBuilder::JUMPFARIndirectOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 6), 1, &OpDispatchBuilder::PUSHOp},
// GROUP 11
@@ -8,7 +8,7 @@ constexpr uint16_t PF_NONE = 0;
constexpr uint16_t PF_F3 = 1;
constexpr uint16_t PF_66 = 2;
constexpr uint16_t PF_F2 = 3;
constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryGroupTables[] = {
// GROUP 6
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_NONE, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_6, PF_F3, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
@@ -69,16 +69,10 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
// GROUP 9
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F2, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 7), 1, &OpDispatchBuilder::RDPIDOp},
// GROUP 12
@@ -119,13 +113,11 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, // SFENCE (or CLFLUSH)
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 5), 1, &OpDispatchBuilder::UnimplementedOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 6), 1, &OpDispatchBuilder::UMonitorOrCLRSSBSY},
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 6), 1, &OpDispatchBuilder::UnimplementedOp},
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_66, 6), 1, &OpDispatchBuilder::CLWBOrTPause},
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_66, 6), 1, &OpDispatchBuilder::CLWB},
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_66, 7), 1, &OpDispatchBuilder::CLFLUSHOPT},
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F2, 6), 1, &OpDispatchBuilder::UMWaitOp},
// GROUP 16
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, true, 1>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
@@ -162,6 +154,18 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_F2, 0), 8, &OpDispatchBuilder::NOPOp},
};
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryGroupTables_64[] = {
// GROUP 15
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 0), 1,
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::FS>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 1), 1,
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::GS>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 2), 1,
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::FS>},
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 3), 1,
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::GS>},
};
#undef OPD
} // namespace FEXCore::IR
Loaded 100 of 666 files, more files were not shown because too many files have changed in this diff. Show more