Compare commits

..
Author SHA1 Message Date
Ryan Houdek c094dc238e Docs: Update for release FEX-2509.1 2025-09-15 18:33:36 -07:00
Billy Laws ceaf38e996 Dispatcher: Fix FABI_F32_I16_F80_PTR argument size
This takes an f80 as input and returns an f32. A copy-paste error had
this truncating the input float if !TMP_ABIARGS.
2025-09-15 18:31:55 -07:00
Billy Laws 85e9e255a5 unittests: Add test for x87 mode switches wrongly flushing NZCV 2025-09-15 18:31:50 -07:00
Billy Laws a545865ab7 OpcodeDispatcher: Only flush MMX registers on MMX -> x87 transitions
Flushing other regs is not necessary, and breaks any ConvertNZCVToX87 use
which relies previously saved NZCV values as the flag-setting NZCV op after
the save could trigger a flush of NZCV.
2025-09-15 18:31:44 -07:00
Billy Laws aa8e8f2cb0 OpcodeDispatcher: Don't assert on invalid ALU op encoding 2025-09-15 18:31:37 -07:00
Billy Laws d3a8701e1a WOW64: Fix CsSeg initialization 2025-09-15 18:31:31 -07:00
214 changed files with 33524 additions and 49954 deletions

No files matched your search

-2
View File
@@ -20,5 +20,3 @@
# Whole-tree reformat with clang-format-19
5267cde60e7642852d18f20ae8568643bb5293d5
# Minor reformat with clang-format-19
9fdd96af61c969cb5732471223f00eda64b7a069
+1
View File
@@ -34,6 +34,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -41,6 +41,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -34,6 +34,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+1
View File
@@ -33,6 +33,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+2 -1
View File
@@ -48,6 +48,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
@@ -77,7 +78,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
+1
View File
@@ -35,6 +35,7 @@ jobs:
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
- name: Update RootFS cache
# Use a bash shell so we can use the same syntax for environment variable
+2 -2
View File
@@ -46,12 +46,12 @@ jobs:
- name: Configure CMake arm64ec
shell: bash
working-directory: ${{runner.workspace}}/build_arm64ec
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
- name: Configure CMake wow64
shell: bash
working-directory: ${{runner.workspace}}/build_wow64
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
- name: Build arm64ec
working-directory: ${{runner.workspace}}/build_arm64ec
+54 -12
View File
@@ -4,6 +4,7 @@ project(FEX C CXX ASM)
INCLUDE (CheckIncludeFiles)
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
option(BUILD_THUNKS "Build thunks" FALSE)
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
@@ -303,8 +304,7 @@ set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-poin
include_directories(External/robin-map/include/)
include(CTest)
if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
add_subdirectory(External/vixl/)
include_directories(SYSTEM External/vixl/src/)
endif()
@@ -335,7 +335,7 @@ endif()
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
if (BUILD_TESTING)
if (BUILD_TESTS)
find_package(Catch2 3 QUIET)
if (NOT Catch2_FOUND)
add_subdirectory(External/Catch2/)
@@ -345,9 +345,6 @@ if (BUILD_TESTING)
endif()
include(Catch)
else ()
# Override any previously generated test list to avoid running stale test binaries
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
endif()
find_package(fmt QUIET)
@@ -458,8 +455,13 @@ endif()
add_compile_options(-Wall)
if (BUILD_TESTING)
include(CTest)
if (BUILD_TESTS)
message(STATUS "Unit tests are enabled")
if (NOT BUILD_TESTING)
# CMake checks this variable before generating CTestTestfile.cmake
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
endif()
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
if (TEST_JOB_COUNT)
@@ -490,11 +492,10 @@ file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.js
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/)
endforeach()
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
@@ -555,7 +556,6 @@ if (BUILD_THUNKS)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)"
DEPENDS guest-libs
COMPONENT Runtime
)
install(
@@ -565,7 +565,6 @@ if (BUILD_THUNKS)
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
)"
DEPENDS guest-libs-32
COMPONENT Runtime
)
add_custom_target(uninstall_guest-libs
@@ -607,3 +606,46 @@ if (OVERRIDE_VERSION STREQUAL "detect")
else()
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
endif()
# Parse the version here
# Change something like `FEX-2106.1-76-<hash>` in to a list
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
# Extract the `2106.1` element
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
# Change `2106.1` in to a list
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
# Calculate list size
list(LENGTH DESCRIBE_LIST LIST_SIZE)
# Pull out the major version
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
# Minor version only exists if there is a .1 at the end
# eg: 2106 versus 2106.1
if (LIST_SIZE GREATER 1)
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
endif()
# Package creation
set (CPACK_GENERATOR "DEB")
set (CPACK_PACKAGE_NAME fex-emu)
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/Description.txt")
# Debian defines
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/triggers")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
# binfmt_misc conflicts with qemu-user-static
# We also only install binfmt_misc on aarch64 hosts
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
endif()
include (CPack)
+29 -47
View File
@@ -36,33 +36,24 @@ public:
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
void adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
if (IsADRRange(Imm)) [[likely]] {
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, ForwardLabel* Label) {
void adr(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADR});
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return adr(rd, &Label->Backward);
adr(rd, &Label->Backward);
} else {
return adr(rd, &Label->Forward);
adr(rd, &Label->Forward);
}
}
@@ -71,42 +62,32 @@ public:
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
void adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) [[likely]] {
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
void adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADRP});
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return adrp(rd, &Label->Backward);
adrp(rd, &Label->Backward);
} else {
return adrp(rd, &Label->Forward);
adrp(rd, &Label->Forward);
}
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
if (IsADRRange(Imm)) {
// If the range is in ADR range then we can just use ADR.
return adr(rd, Label);
adr(rd, Label);
} else if (IsADRPRange(Imm)) {
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
@@ -121,28 +102,23 @@ public:
// Now even an add
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
}
return BranchEncodeSucceeded::Success;
} else {
LOGMAN_MSG_A_FMT("Unscaled offset too large");
FEX_UNREACHABLE;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
// Emit a register index and a nop. These will be backpatched.
dc32(rd.Idx());
nop();
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return LongAddressGen(rd, &Label->Backward);
LongAddressGen(rd, &Label->Backward);
} else {
return LongAddressGen(rd, &Label->Forward);
LongAddressGen(rd, &Label->Forward);
}
}
@@ -886,6 +862,12 @@ public:
}
private:
static constexpr Condition InvertCondition(Condition cond) {
// These behave as always, so it makes no sense to allow inverting these.
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
}
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
constexpr uint32_t Op = 0b001'0010'00 << 22;
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
+2 -1
View File
@@ -2244,7 +2244,8 @@ public:
template<IsQOrDRegister T>
void movi(SubRegSize size, T rd, uint64_t Imm, uint16_t Shift = 0) {
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit ||
size == SubRegSize::i64Bit,
"Unsupported movi size");
uint32_t cmode;
+63 -123
View File
@@ -20,31 +20,23 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm);
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
void b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
void b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
void b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return b(Cond, &Label->Backward);
b(Cond, &Label->Backward);
} else {
return b(Cond, &Label->Forward);
b(Cond, &Label->Forward);
}
}
@@ -53,32 +45,24 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm);
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
void bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
void bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return bc(Cond, &Label->Backward);
bc(Cond, &Label->Backward);
} else {
return bc(Cond, &Label->Forward);
bc(Cond, &Label->Forward);
}
}
@@ -114,32 +98,25 @@ public:
UnconditionalBranch(Op, Imm);
}
[[nodiscard]] BranchEncodeSucceeded b(const BackwardLabel* Label) {
void b(const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0001'01 << 26;
// Can't encode.
return BranchEncodeSucceeded::Failure;
UnconditionalBranch(Op, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded b(ForwardLabel* Label) {
void b(ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded b(BiDirectionalLabel* Label) {
void b(BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return b(&Label->Backward);
b(&Label->Backward);
} else {
return b(&Label->Forward);
b(&Label->Forward);
}
}
@@ -149,33 +126,25 @@ public:
UnconditionalBranch(Op, Imm);
}
[[nodiscard]] BranchEncodeSucceeded bl(const BackwardLabel* Label) {
void bl(const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b1001'01 << 26;
return BranchEncodeSucceeded::Success;
}
// Can't encode.
return BranchEncodeSucceeded::Failure;
UnconditionalBranch(Op, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded bl(ForwardLabel* Label) {
void bl(ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded bl(BiDirectionalLabel* Label) {
void bl(BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return bl(&Label->Backward);
bl(&Label->Backward);
} else {
return bl(&Label->Forward);
bl(&Label->Forward);
}
}
@@ -186,35 +155,28 @@ public:
CompareAndBranch(Op, s, rt, Imm);
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0100 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return cbz(s, rt, &Label->Backward);
cbz(s, rt, &Label->Backward);
} else {
return cbz(s, rt, &Label->Forward);
cbz(s, rt, &Label->Forward);
}
}
@@ -224,35 +186,28 @@ public:
CompareAndBranch(Op, s, rt, Imm);
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0101 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return cbnz(s, rt, &Label->Backward);
cbnz(s, rt, &Label->Backward);
} else {
return cbnz(s, rt, &Label->Forward);
cbnz(s, rt, &Label->Forward);
}
}
@@ -262,35 +217,28 @@ public:
TestAndBranch(Op, rt, Bit, Imm);
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0110 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return tbz(rt, Bit, &Label->Backward);
tbz(rt, Bit, &Label->Backward);
} else {
return tbz(rt, Bit, &Label->Forward);
tbz(rt, Bit, &Label->Forward);
}
}
@@ -299,35 +247,27 @@ public:
TestAndBranch(Op, rt, Bit, Imm);
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
return BranchEncodeSucceeded::Success;
}
constexpr uint32_t Op = 0b0011'0111 << 24;
// Can't encode.
return BranchEncodeSucceeded::Failure;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
// Forward label doesn't know if it can encode until Bind.
return BranchEncodeSucceeded::Success;
}
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
if (Label->Backward.Location) {
return tbnz(rt, Bit, &Label->Backward);
tbnz(rt, Bit, &Label->Backward);
} else {
return tbnz(rt, Bit, &Label->Forward);
tbnz(rt, Bit, &Label->Forward);
}
}
+15 -52
View File
@@ -586,11 +586,6 @@ concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegi
template<typename T>
concept IsQOrDRegister = std::is_same_v<T, QRegister> || std::is_same_v<T, DRegister>;
enum class BranchEncodeSucceeded {
Success,
Failure,
};
// Whether or not a given set of vector registers are sequential
// in increasing order as far as the register file is concerned (modulo its size)
//
@@ -643,25 +638,19 @@ public:
// Bind a backward label to an address.
// Address that is bound is the current emitter location.
[[nodiscard]] bool Bind(BackwardLabel* Label) {
void Bind(BackwardLabel* Label) {
LOGMAN_THROW_A_FMT(Label->Location == nullptr, "Trying to bind a label twice");
Label->Location = GetCursorAddress<uint8_t*>();
// Always binds because it is only storing a location.
return true;
}
[[nodiscard]] bool Bind(const ForwardLabel::Reference* Label) {
void Bind(const ForwardLabel::Reference* Label) {
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
// Patch up the instructions
switch (Label->Type) {
case ForwardLabel::InstType::ADR: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!IsADRRange(Imm)) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
@@ -673,12 +662,7 @@ public:
case ForwardLabel::InstType::ADRP: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
@@ -688,13 +672,11 @@ public:
*Instruction = Inst;
break;
}
case ForwardLabel::InstType::B: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -704,13 +686,11 @@ public:
break;
}
case ForwardLabel::InstType::TEST_BRANCH: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -724,10 +704,7 @@ public:
case ForwardLabel::InstType::RELATIVE_LOAD: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) [[unlikely]] {
// Can't bind.
return false;
}
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
@@ -776,41 +753,27 @@ public:
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
return true;
}
// Bind a forward label to a location.
// This walks all the instructions in the label's vector.
// Then backpatching all instructions that have used the label.
[[nodiscard]] bool Bind(ForwardLabel* Label) {
bool Bound = true;
void Bind(ForwardLabel* Label) {
if (Label->FirstInst.Location) {
Bound &= Bind(&Label->FirstInst);
Bind(&Label->FirstInst);
}
for (auto& Inst : Label->Insts) {
Bound &= Bind(&Inst);
Bind(&Inst);
}
return Bound;
}
// Bind a bidirectional location to a location.
// Binds both forwards and backwards depending on how the label was used.
[[nodiscard]] bool Bind(BiDirectionalLabel* Label) {
bool Bound = true;
void Bind(BiDirectionalLabel* Label) {
if (!Label->Backward.Location) {
Bound &= Bind(&Label->Backward);
Bind(&Label->Backward);
}
Bound &= Bind(&Label->Forward);
return Bound;
}
static constexpr Condition InvertCondition(Condition cond) {
// These behave as always, so it makes no sense to allow inverting these.
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
Bind(&Label->Forward);
}
#include <CodeEmitter/VixlUtils.inl>
+2 -4
View File
@@ -4,8 +4,7 @@ file(GLOB GEN_CONFIG_SOURCES CONFIGURE_DEPENDS *.json.in)
# Any application configuration json file gets installed
foreach(CONFIG_SRC ${CONFIG_SOURCES})
install(FILES ${CONFIG_SRC}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
# Any configuration file json file that needs to be generated
@@ -22,6 +21,5 @@ foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
# Then install the configured json
install(
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
DESTINATION ${DATA_DIRECTORY}/AppConfig/
COMPONENT Runtime)
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
endforeach()
+3
View File
@@ -0,0 +1,3 @@
x86 and x86-64 Linux emulator
FEX allows you to run x86 applications on ARM64 Linux devices. It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
+18
View File
@@ -0,0 +1,18 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Setup binfmt_misc
update-binfmts --import FEX-x86
update-binfmts --import FEX-x86_64
}
# Install FEXInterpreter hardlink
# Needs to be done before setting up binfmt_misc
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
+17
View File
@@ -0,0 +1,17 @@
#!/bin/sh
set -e
update_binfmt() {
# Check for update-binfmts
command -v update-binfmts >/dev/null || return 0
# Uninstall
update-binfmts --unimport FEX-x86
update-binfmts --unimport FEX-x86_64
}
if [ $(uname -m) = 'aarch64' ]; then
update_binfmt
fi
# Remove FEXInterpreter hardlink
unlink /usr/bin/FEXInterpreter
+1
View File
@@ -0,0 +1 @@
activate-noawait ldconfig
+1 -1
View File
@@ -14,7 +14,7 @@ RUN mkdir build
ARG CC=clang-13
ARG CXX=clang++-13
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTING=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTS=False -DENABLE_ASSERTIONS=False -G Ninja .
RUN ninja
WORKDIR /FEX/build
+2 -4
View File
@@ -10,8 +10,7 @@ function(GenBinFmt Name)
# Then install the configured binfmt
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/
COMPONENT Runtime)
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
endfunction()
if (NOT USE_LEGACY_BINFMTMISC)
@@ -20,8 +19,7 @@ if (NOT USE_LEGACY_BINFMTMISC)
install(
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/
COMPONENT Runtime)
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
else()
GenBinFmt(FEX-x86.in)
GenBinFmt(FEX-x86_64.in)
+1 -1
View File
@@ -1 +1 @@
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+1 -1
View File
@@ -1 +1 @@
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
+1 -1
View File
@@ -1,5 +1,5 @@
package fex
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
offset 0
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
+1 -1
View File
@@ -45,7 +45,7 @@ pkgs.mkShell {
fi
'';
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False
FEX_CMAKE_TOOLCHAIN_ARM64EC = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_CMAKE_TOOLCHAIN_WOW64 = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
FEX_MESON_CROSSFILE = "--cross-file ${mesonCrossFile}";
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -18,4 +18,4 @@ then
fi
set -o xtrace
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
+1 -1
View File
@@ -14,4 +14,4 @@ fi
rm -rf unittests/FEXLinuxTests
set -o xtrace
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTING=ON -DBUILD_FEX_LINUX_TESTS=ON
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTS=ON -DBUILD_FEX_LINUX_TESTS=ON
+1 -1
+1 -1
View File
@@ -78,6 +78,6 @@ install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
DESTINATION include
COMPONENT Development)
if (BUILD_TESTING)
if (BUILD_TESTS)
add_subdirectory(unittests/)
endif()
+165 -6
View File
@@ -118,6 +118,41 @@ def print_man_env_option(name, desc, default, no_json_key):
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
output_man.write(".Pp\n\n")
def print_man_options(options):
output_man.write(".Sh OPTIONS\n")
output_man.write(".Bl -tag -width -indent\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
default = op_vals["Default"]
value_type = op_vals["Type"]
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = op_vals["TextDefault"]
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
# Wrap the string argument in quotes
default = "'" + default + "'"
print_man_option(
short,
long,
op_vals["Desc"],
default
)
if (value_type == "strenum"):
Enums = op_vals["Enums"]
output_man.write("\\fBAvailable Options:\\fR\n")
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
output_man.write("\n.sp\n")
output_man.write(".El\n")
def print_man_environment(options):
output_man.write(".Sh ENVIRONMENT\n")
output_man.write(".Bl -tag -width -indent\n")
@@ -159,7 +194,7 @@ def print_man_environment_tail():
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
"This will override the full path",
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -173,7 +208,7 @@ def print_man_environment_tail():
"One must be careful with this option as it will override any applications that load with execve as well"
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
"If FEX_PORTABLE is declared then relative paths are also supported",
"For FEX: Relative to the FEX binary",
"For FEXInterpreter: Relative to the FEXInterpreter binary",
"For WINE: Relative to %LOCALAPPDATA%"
],
"''", True)
@@ -192,8 +227,8 @@ def print_man_environment_tail():
"PORTABLE",
[
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
"For FEX on Linux:",
"These files are instead read from <FEXPath>/fex-emu/ by default.",
"For FEXInterpreter on Linux:",
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
"For Arm64ec/Wow64 WINE builds:",
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
@@ -205,12 +240,20 @@ def print_man_header():
.Dt FEX
.Os Linux
.Sh NAME
.Nm FEX
.Nm FEXLoader
.Nm FEXInterpreter
.Nm FEXBash
.Nd Fast x86-64 and x86 emulation.
.Sh SYNOPSIS
.Nm
.Ar <args> ...
.Op options
.Op Ar --
.Ar Application
<args> ...
.Pp
.Nm FEXInterpreter
.Ar Application
<args> ...
.Pp
.Nm FEXBash
.Ar <args> ...
@@ -318,6 +361,82 @@ def print_config_option(type, group_name, json_name, default_value, short, choic
output_argloader.write("\n");
def print_argloader_options(options):
output_argloader.write("#ifdef BEFORE_PARSE\n")
output_argloader.write("#undef BEFORE_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
default = op_vals["Default"]
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
# Wrap the string argument in quotes
default = "\"" + default + "\""
# Textual default rather than enum based
if ("TextDefault" in op_vals):
default = "\"" + op_vals["TextDefault"] + "\""
short = None
choices = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if ("Choices" in op_vals):
choices = op_vals["Choices"]
print_config_option(
op_vals["Type"],
op_group,
op_key,
default,
short,
choices,
op_vals["Desc"])
output_argloader.write("\n")
output_argloader.write("#endif\n")
def print_parse_argloader_options(options):
output_argloader.write("#ifdef AFTER_PARSE\n")
output_argloader.write("#undef AFTER_PARSE\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
output_argloader.write("if (Options.is_set_by_user(\"{0}\")) {{\n".format(op_key))
value_type = op_vals["Type"]
NeedsString = False
conversion_func = "fextl::fmt::format(\"{}\", "
if ("ArgumentHandler" in op_vals):
NeedsString = True
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
if (value_type == "str"):
NeedsString = True
conversion_func = "std::move("
if (value_type == "bool"):
# boolean values need a decimal specifier. Otherwise fmt prints strings.
conversion_func = "fextl::fmt::format(\"{:d}\", "
if (value_type == "strenum"):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key))
elif (value_type == "strarray"):
# these need a bit more help
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
output_argloader.write("\t}\n")
else:
if (NeedsString):
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
else:
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_envloader_options(options):
output_argloader.write("#ifdef ENVLOADER\n")
output_argloader.write("#undef ENVLOADER\n")
@@ -398,6 +517,41 @@ def print_parse_enum_options(options):
output_argloader.write("#endif\n")
def check_for_duplicate_options(options):
short_map = []
long_map = []
# Spin through all the items and see if we have a duplicate option
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
short = None
long = op_key.lower()
long_invert = None
if ("ShortArg" in op_vals):
short = op_vals["ShortArg"]
if (op_vals["Type"] == "bool"):
long_invert = "no-" + long
# Check for short key duplication
if (short != None):
if (short in short_map):
raise Exception("Short config '{0}' for option '{1}' has duplicate entry!".format(short, op_key))
else:
short_map.append(short)
# Check for long key duplication
if (long in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long))
else:
long_map.append(long)
# Check for long key duplication
if (long_invert != None):
if (long_invert in long_map):
raise Exception("Long config '{0}' has duplicate entry!".format(long_invert))
else:
long_map.append(long_invert)
if (len(sys.argv) < 5):
sys.exit()
@@ -414,6 +568,8 @@ json_object = json.loads(json_text)
options = json_object["Options"]
unnamed_options = json_object["UnnamedOptions"]
check_for_duplicate_options(options)
# Generate config include file
output_file = open(output_filename, "w")
print_header()
@@ -425,6 +581,7 @@ output_file.close()
# Generate man file
output_man = open(output_man_page, "w")
print_man_header()
print_man_options(options)
print_man_environment(options)
print_man_tail()
@@ -432,6 +589,8 @@ output_man.close()
# Generate argument loader code
output_argloader = open(output_argumentloader_filename, "w")
print_argloader_options(options);
print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
+2 -3
View File
@@ -18,7 +18,6 @@ set (SRCS
Common/JitSymbols.cpp
Interface/Context/Context.cpp
Interface/Core/LookupCache.cpp
Interface/Core/CodeCache.cpp
Interface/Core/Core.cpp
Interface/Core/CPUBackend.cpp
Interface/Core/Addressing.cpp
@@ -58,6 +57,7 @@ set (SRCS
Interface/Core/X86Tables/VEXTables.cpp
Interface/Core/X86Tables/X87Tables.cpp
Interface/GDBJIT/GDBJIT.cpp
Interface/IR/AOTIR.cpp
Interface/IR/IRDumper.cpp
Interface/IR/IREmitter.cpp
Interface/IR/PassManager.cpp
@@ -66,7 +66,6 @@ set (SRCS
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/x87StackOptimizationPass.cpp
Utils/LongJump.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/Profiler.cpp
@@ -203,7 +202,7 @@ add_custom_target(CONFIG_INC
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
# Install the compressed man page
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
# Add in diagnostic colours if the option is available.
# Ninja code generator will kill colours if this isn't here
-2
View File
@@ -4,8 +4,6 @@
#ifdef _M_X86_64
#include <xmmintrin.h>
#include <immintrin.h>
#else
#include <cstdint>
#endif
namespace FEXCore {
+26 -7
View File
@@ -4,6 +4,7 @@
"Multiblock": {
"Type": "bool",
"Default": "true",
"ShortArg": "m",
"Desc": [
"Controls multiblock code compilation",
"Can cause long JIT compilation times and stutter"
@@ -12,6 +13,7 @@
"MaxInst": {
"Type": "int32",
"Default": "5000",
"ShortArg": "n",
"Desc": [
"Maximum number of instruction to store in a block"
]
@@ -97,6 +99,7 @@
"RootFS": {
"Type": "str",
"Default": "",
"ShortArg": "R",
"Desc": [
"Which Root filesystem prefix to use",
"This can be a filesystem path",
@@ -111,6 +114,7 @@
"ThunkHostLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
"ShortArg": "t",
"Desc": [
"Folder to find the host-side thunking libraries."
]
@@ -118,6 +122,7 @@
"ThunkGuestLibs": {
"Type": "str",
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
"ShortArg": "j",
"Desc": [
"Folder to find the guest-side thunking libraries."
]
@@ -125,6 +130,7 @@
"ThunkConfig": {
"Type": "str",
"Default": "",
"ShortArg": "k",
"Desc": [
"A json file specifying where to overlay the thunks.",
"This can be a filesystem path",
@@ -139,6 +145,7 @@
"Env": {
"Type": "strarray",
"Default": "",
"ShortArg": "E",
"Desc": [
"Adds an environment variable to the emulated environment."
]
@@ -146,6 +153,7 @@
"HostEnv": {
"Type": "strarray",
"Default": "",
"ShortArg": "H",
"Desc": [
"Adds an environment variable to the host environment.",
"This can be useful for setting environment variables that thunks can pick up.",
@@ -164,6 +172,7 @@
"SingleStep": {
"Type": "bool",
"Default": "false",
"ShortArg": "S",
"Desc": [
"Single stepping configuration."
]
@@ -171,6 +180,7 @@
"GdbServer": {
"Type": "bool",
"Default": "false",
"ShortArg": "G",
"Desc": [
"Enables the GDB server."
]
@@ -204,6 +214,7 @@
"DumpGPRs": {
"Type": "bool",
"Default": "false",
"ShortArg": "g",
"Desc": [
"When the test harness ends, print the GPR state."
]
@@ -211,6 +222,7 @@
"O0": {
"Type": "bool",
"Default": "false",
"ShortArg": "O0",
"Desc": [
"Disables optimizations passes for debugging."
]
@@ -300,6 +312,7 @@
"SilentLog": {
"Type": "bool",
"Default": "true",
"ShortArg": "s",
"Desc": [
"Disables logging"
]
@@ -307,6 +320,7 @@
"OutputLog": {
"Type": "str",
"Default": "server",
"ShortArg": "o",
"Desc": [
"File to write FEX output to.",
"[stdout, stderr, server, <Filename>]"
@@ -327,13 +341,6 @@
"Enables FEX's low-overhead sampling profile statistics.",
"Requires a supported version of Mangohud to see the results"
]
},
"TraceProfiler": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enables FEX's trace profiler. Using gpuvis or tracy"
]
}
},
"Hacks": {
@@ -388,6 +395,14 @@
"This is required to ensure a split-lock doesn't tear inside the process"
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
"Desc": [
"Automatically enables TSO when shared memory is used.",
"Should work without issues in most cases."
]
},
"VolatileMetadata": {
"Type": "bool",
"Default": "true",
@@ -499,6 +514,10 @@
},
"UnnamedOptions": {
"Misc": {
"IS_INTERPRETER": {
"Type": "bool",
"Default": "false"
},
"INTERPRETER_INSTALLED": {
"Type": "bool",
"Default": "false"
@@ -1,7 +1,6 @@
// SPDX-License-Identifier: MIT
#include "Interface/Context/Context.h"
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Core/CoreState.h>
+44 -37
View File
@@ -5,46 +5,53 @@
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/X86HelperGen.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/AOTIR.h"
#include <Interface/IR/IntrusiveIRList.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <stdint.h>
#include <atomic>
#include <cstddef>
#include <cstdint>
#include <mutex>
#include <optional>
#include <shared_mutex>
namespace FEXCore {
class SignalDelegator;
class CodeLoader;
class ThunkHandler;
namespace Core {
struct DebugData;
struct InternalThreadState;
} // namespace Core
namespace CPU {
class Arm64JITCore;
class Dispatcher;
} // namespace CPU
namespace HLE {
class SourcecodeResolver;
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
} // namespace HLE
} // namespace FEXCore
namespace FEXCore::IR {
namespace Validation {
class IRValidation;
}
} // namespace FEXCore::IR
namespace FEXCore::Context {
struct FEX_PACKED ExitFunctionLinkData {
uint64_t HostCode;
@@ -64,23 +71,7 @@ struct CustomIRResult {
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
class CodeCache : public AbstractCodeCache {
public:
CodeCache(ContextImpl&);
~CodeCache();
ContextImpl& CTX;
bool IsGeneratingCache = false;
void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
void InitiateCacheGeneration() override {
IsGeneratingCache = true;
}
};
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
public:
// Context base class implementation.
bool InitCore() override;
@@ -150,9 +141,10 @@ public:
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
CodeCache& GetCodeCache() override {
return CodeCache;
}
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
void FinalizeAOTIRCache() override {}
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
@@ -162,6 +154,8 @@ public:
return CodeInvalidationMutex;
}
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
@@ -183,6 +177,13 @@ public:
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
public:
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
friend class FEXCore::IR::Validation::IRValidation;
struct {
uint64_t VirtualMemSize {1ULL << 36};
uint64_t TSCScale = 0;
@@ -195,6 +196,7 @@ public:
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
@@ -225,7 +227,6 @@ public:
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
FEXCore::ThunkHandler* ThunkHandler {};
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CodeCache CodeCache;
SignalDelegator* SignalDelegation {};
X86GeneratedCode X86CodeGen;
@@ -316,9 +317,12 @@ protected:
VectorAtomicTSOEmulationEnabled = true;
MemcpyAtomicTSOEmulationEnabled = true;
} else {
AtomicTSOEmulationEnabled = Config.TSOEnabled;
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
MemcpyAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.MemcpySetTSOEnabled;
// Atomic TSO emulation only enabled if the config option is enabled.
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
}
}
@@ -332,6 +336,9 @@ private:
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
IR::AOTIRCaptureCache IRCaptureCache;
bool IsMemoryShared = false;
bool SupportsHardwareTSO = false;
bool AtomicTSOEmulationEnabled = true;
bool VectorAtomicTSOEmulationEnabled = false;
@@ -344,8 +351,8 @@ private:
std::atomic<bool> HasCustomIRHandlers {};
struct CustomIRHandlerEntry final {
CustomIREntrypointHandler Handler;
void* Creator;
void* Data;
void *Creator;
void *Data;
};
fextl::unordered_map<uint64_t, CustomIRHandlerEntry> CustomIRHandlers;
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
+12 -5
View File
@@ -17,10 +17,6 @@ $end_info$
#include <cstdint>
namespace FEXCore::CPU {
union Relocation;
}
namespace FEXCore {
namespace IR {
@@ -161,7 +157,18 @@ namespace CPU {
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() = 0;
/**
* @brief Relocates a block of code from the JIT code object cache
*
* @param Entry - RIP of the entry
* @param SerializationData - Serialization data referring to the object cache for `Entry`
*
* @return An executable function pointer relocated from the cache object
*/
[[nodiscard]]
virtual void* RelocateJITObjectCode(uint64_t /* Entry */, const CodeSerialize::CodeObjectFileSection* /* SerializationData */) {
return nullptr;
}
virtual void ClearCache() {}
+32 -49
View File
@@ -43,15 +43,12 @@ namespace ProductNames {
static const char ARM_A715[] = "Cortex-A715";
static const char ARM_A720[] = "Cortex-A720";
static const char ARM_A725[] = "Cortex-A725";
static const char ARM_C1Pro[] = "C1-Pro";
static const char ARM_C1Premium[] = "C1-Premium";
static const char ARM_X1[] = "Cortex-X1";
static const char ARM_X1C[] = "Cortex-X1C";
static const char ARM_X2[] = "Cortex-X2";
static const char ARM_X3[] = "Cortex-X3";
static const char ARM_X4[] = "Cortex-X4";
static const char ARM_X925[] = "Cortex-X925";
static const char ARM_C1Ultra[] = "C1-Ultra";
static const char ARM_N1[] = "Neoverse N1";
static const char ARM_N2[] = "Neoverse N2";
static const char ARM_N3[] = "Neoverse N3";
@@ -62,7 +59,6 @@ namespace ProductNames {
static const char ARM_A65[] = "Cortex-A65";
static const char ARM_A510[] = "Cortex-A510";
static const char ARM_A520[] = "Cortex-A520";
static const char ARM_C1Nano[] = "C1-Nano";
static const char ARM_Kryo200[] = "Kryo 2xx";
static const char ARM_Kryo300[] = "Kryo 3xx";
@@ -74,7 +70,6 @@ namespace ProductNames {
static const char ARM_Denver[] = "Nvidia Denver";
static const char ARM_Carmel[] = "Nvidia Carmel";
static const char ARM_Olympus[] = "Nvidia Olympus";
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
@@ -90,9 +85,6 @@ namespace ProductNames {
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
static const char ARM_ORYON_1[] = "Oryon-1";
static const char ARM_Ampere_1[] = "AmpereOne";
static const char ARM_Ampere_1A[] = "AmpereOneA";
static const char ARM_Ampere_1B[] = "AmpereOneB";
#else
#endif
} // namespace ProductNames
@@ -178,7 +170,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
// CPU priority order
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
// Relative list so things they will commonly end up in big.little configurations sort of relate
static constexpr std::array<CPUMIDR, 66> CPUMIDRs = {{
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
// Typically big CPU cores
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
@@ -189,46 +181,38 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
{0x41, 0xd8b, 1, ProductNames::ARM_C1Pro}, // C1-Pro
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
{0xc0, 0xac3, 1, ProductNames::ARM_Ampere_1}, // AmpereOne
{0xc0, 0xac4, 1, ProductNames::ARM_Ampere_1A}, // AmpereOneA
{0xc0, 0xac5, 1, ProductNames::ARM_Ampere_1B}, // AmpereOneB
{0x4e, 0x010, 1, ProductNames::ARM_Olympus}, // Olympus
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
// Denver rated above A57 to match TX2 weirdness
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
@@ -243,7 +227,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
{0x41, 0xd8a, 1, ProductNames::ARM_C1Nano}, // C1-Nano
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
@@ -1,27 +0,0 @@
// SPDX-License-Identifier: MIT
#include <Interface/Context/Context.h>
#include <FEXCore/HLE/SourcecodeResolver.h>
namespace FEXCore {
ExecutableFileInfo::~ExecutableFileInfo() = default;
} // namespace FEXCore
namespace FEXCore::Context {
CodeCache::CodeCache(ContextImpl& CTX_)
: CTX(CTX_) {}
CodeCache::~CodeCache() = default;
void CodeCache::LoadData(Core::InternalThreadState& Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& GuestRIPLookup) {
// TODO
}
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
// TODO
return true;
}
} // namespace FEXCore::Context
+53 -38
View File
@@ -18,7 +18,6 @@ $end_info$
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/X86Tables/X86Tables.h"
#include <Interface/GDBJIT/GDBJIT.h>
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
@@ -78,7 +77,7 @@ namespace FEXCore::Context {
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
: HostFeatures {Features}
, CPUID {this}
, CodeCache {*this} {
, IRCaptureCache {this} {
if (!Config.Is64BitMode()) {
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
Config.VirtualMemSize = 1ULL << 32;
@@ -611,7 +610,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
} else {
ForceTSO = IR::ForceTSOMode::ForceDisabled;
}
} else if (DecodedInfo->Flags & X86Tables::DecodeFlags::FLAG_FORCE_TSO) {
} else if (DecodedInfo->ForceTSO) {
ForceTSO = IR::ForceTSOMode::ForceEnabled;
}
@@ -709,9 +708,9 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
if (SourcecodeResolver && Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
if (MappedSection) {
MappedSection->FileInfo.SourcecodeMap = SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, MappedSection->FileInfo.FileId);
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (AOTIRCacheEntry.Entry) {
AOTIRCacheEntry.Entry->SourcecodeMap = SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
}
}
@@ -786,44 +785,36 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
if (Config.BlockJITNaming()) {
auto FragmentBasePtr = CompiledCode.BlockBegin;
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
if (DebugData) {
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
if (DebugData->Subblocks.size()) {
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
}
}
}
} else {
if (GuestRIPLookup) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
GuestRIP - GuestRIPLookup->FileStartVA);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
}
}
}
if (Config.LibraryJITNaming() || Config.GDBSymbols()) {
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
if (MappedSection) {
if (Config.LibraryJITNaming()) {
Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, MappedSection->FileInfo.Filename);
}
if (Config.GDBSymbols()) {
GDBJITRegister(MappedSection->FileInfo, MappedSection->FileStartVA, GuestRIP, (uintptr_t)CodePtr, *DebugData);
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
}
}
}
}
// Clear any relocations that might have been generated
if (!CodeCache.IsGeneratingCache) {
Thread->CPUBackend->ClearRelocations();
Thread->CPUBackend->ClearRelocations();
if (IRCaptureCache.PostCompileCode(Thread, CompiledCode.BlockBegin, GuestRIP, StartAddr, Length, DebugData.get())) {
// Early exit
return (uintptr_t)CodePtr;
}
if (NeedsAddGuestCodeRanges) {
@@ -902,6 +893,23 @@ void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* T
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
}
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
if (!Thread) {
return;
}
if (!IsMemoryShared) {
IsMemoryShared = true;
UpdateAtomicTSOEmulationConfig();
if (Config.TSOAutoMigration) {
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
Thread->LookupCache->ClearCache();
}
}
}
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
"be unique_locked here");
@@ -1007,9 +1015,9 @@ void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint
auto lk = GuardSignalDeferringSection(CTX->CodeInvalidationMutex, Thread);
if (Size == 8) {
*reinterpret_cast<uint64_t*>(Address) = Value;
*reinterpret_cast<uint64_t *>(Address) = Value;
} else if (Size == 4) {
*reinterpret_cast<uint32_t*>(Address) = Value;
*reinterpret_cast<uint32_t *>(Address) = Value;
} else {
ERROR_AND_DIE_FMT("Unexpected write size for backpatcher: {}", Size);
}
@@ -1018,6 +1026,13 @@ void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint
CTX->SyscallHandler->InvalidateGuestCodeRange(Thread, Address, Size);
}
IR::AOTIRCacheEntry* ContextImpl::LoadAOTIRCacheEntry(const fextl::string& filename) {
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
return rv;
}
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {}
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
@@ -1,6 +1,6 @@
// SPDX-License-Identifier: MIT
#include "Common/VectorRegType.h"
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
@@ -17,7 +17,6 @@
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <CodeEmitter/Emitter.h>
@@ -26,7 +25,9 @@
#endif
#include <array>
#include <atomic>
#include <bit>
#include <condition_variable>
#include <csignal>
#include <cstring>
@@ -91,7 +92,7 @@ void Dispatcher::EmitDispatcher() {
FillStaticRegs();
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
ARMEmitter::BiDirectionalLabel LoopTop {};
@@ -140,7 +141,7 @@ void Dispatcher::EmitDispatcher() {
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
(void)Bind(&LoopTop);
Bind(&LoopTop);
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
// Load in our RIP
@@ -167,11 +168,11 @@ void Dispatcher::EmitDispatcher() {
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
(void)!Bind(&l_NotECCode);
Bind(&l_NotECCode);
#endif
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
@@ -196,7 +197,7 @@ void Dispatcher::EmitDispatcher() {
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
// If page pointer is zero then we have no block
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
@@ -211,11 +212,10 @@ void Dispatcher::EmitDispatcher() {
// If the guest address doesn't match, Compile the block.
sub(TMP2, TMP2, RipReg);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
// Check the host address to see if it matches, else compile the block.
(void)cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
// If we've made it here then we have a real compiled block
{
@@ -303,7 +303,7 @@ void Dispatcher::EmitDispatcher() {
// Need to create the block
{
(void)Bind(&NoBlock);
Bind(&NoBlock);
EmitSignalGuardedRegion([&]() {
SpillStaticRegs(TMP1);
@@ -337,7 +337,7 @@ void Dispatcher::EmitDispatcher() {
}
{
(void)Bind(&CompileSingleStep);
Bind(&CompileSingleStep);
EmitSignalGuardedRegion([&]() {
SpillStaticRegs(TMP1);
@@ -499,7 +499,7 @@ void Dispatcher::EmitDispatcher() {
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
// Now go back to the regular dispatcher loop
(void)b(&LoopTop);
b(&LoopTop);
}
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
@@ -568,15 +568,14 @@ void Dispatcher::EmitDispatcher() {
}
}
(void)Bind(&l_CTX);
Bind(&l_CTX);
dc64(reinterpret_cast<uintptr_t>(CTX));
(void)Bind(&l_Sleep);
Bind(&l_Sleep);
dc64(reinterpret_cast<uint64_t>(SleepThread));
(void)Bind(&l_CompileBlock);
Bind(&l_CompileBlock);
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
dc64(PMFCompileBlock.GetConvertedPointer());
(void)Bind(&l_CompileSingleStep);
Bind(&l_CompileSingleStep);
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
dc64(PMFCompileSingleStep.GetConvertedPointer());
+31 -39
View File
@@ -259,13 +259,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
if (HasSIB) {
FEXCore::X86Tables::SIBDecoded SIB;
if (DecodeInst->Flags & DecodeFlags::FLAG_DECODED_SIB) {
if (DecodeInst->DecodedSIB) {
SIB.Hex = DecodeInst->SIB;
} else {
// Haven't yet grabbed SIB, pull it now
DecodeInst->SIB = ReadByte();
SIB.Hex = DecodeInst->SIB;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_SIB;
DecodeInst->DecodedSIB = true;
}
// If the SIB base is 0b101, aka BP or R13 then we have a 32bit displacement
@@ -401,9 +401,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// If we require ModRM and haven't decoded it yet, do it now
// Some instructions have to read modrm upfront, others do it later
if (HasMODRM && !(DecodeInst->Flags & DecodeFlags::FLAG_DECODED_MODRM)) {
if (HasMODRM && !DecodeInst->DecodedModRM) {
DecodeInst->ModRM = ReadByte();
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
}
// New instruction size decoding
@@ -436,8 +436,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_16BIT);
DestSize = 2;
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) && (HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) &&
(HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_64BIT);
DestSize = 8;
} else {
@@ -464,8 +465,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
// See table 1-2. Operand-Size Overrides for this decoding
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) && (HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) &&
(HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_64BIT);
} else {
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_32BIT);
@@ -634,20 +636,11 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
Literal = static_cast<int32_t>(Literal);
}
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
DecodeInst->Src[CurrentSrc].Data.Literal.SignExtend = true;
}
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
++CurrentSrc;
if (Bytes == 8) [[unlikely]] {
DecodeInst->Src[CurrentSrc].Data.Literal.Size = 4;
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal >> 32;
}
Bytes = 0;
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
}
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
@@ -679,7 +672,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
} else if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -696,18 +689,18 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
constexpr uint16_t PF_F2 = 3;
uint16_t PrefixType = PF_NONE;
if (LastEscapePrefix == 0xF3) {
if (DecodeInst->LastEscapePrefix == 0xF3) {
PrefixType = PF_F3;
} else if (LastEscapePrefix == 0xF2) {
} else if (DecodeInst->LastEscapePrefix == 0xF2) {
PrefixType = PF_F2;
} else if (LastEscapePrefix == 0x66) {
} else if (DecodeInst->LastEscapePrefix == 0x66) {
PrefixType = PF_66;
}
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -734,7 +727,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
return NormalOp(&(*X87Table)[X87Op], X87Op);
@@ -796,7 +789,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
// We have ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -820,7 +813,6 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
InstructionSize = 0;
LastEscapePrefix = 0;
Instruction.fill(0);
DecodeInst = &DecodedBuffer[DecodedSize];
@@ -842,7 +834,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
// Decode ModRM
uint8_t ModRMByte = ReadByte();
DecodeInst->ModRM = ModRMByte;
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
DecodeInst->DecodedModRM = true;
FEXCore::X86Tables::ModRMDecoded ModRM;
ModRM.Hex = DecodeInst->ModRM;
@@ -881,7 +873,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
uint16_t LocalOp = (Prefix << 8) | ReadByte();
bool NoOverlay66 = (FEXCore::X86Tables::H0F38TableOps[LocalOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
if (LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
if (DecodeInst->LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather than modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
@@ -897,7 +889,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
constexpr uint16_t PF_3A_REX = (1 << 1);
uint16_t Prefix = PF_3A_NONE;
if (LastEscapePrefix == 0x66) { // Operand Size
if (DecodeInst->LastEscapePrefix == 0x66) { // Operand Size
Prefix = PF_3A_66;
}
@@ -923,17 +915,17 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
if (NoOverlay) { // This section of the table ignores prefix extention
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0xF3) { // REP
} else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0xF2) { // REPNE
} else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
} else if (LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
} else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
// Remove prefix so it doesn't effect calculations.
// This is only an escape prefix rather tan modifier now
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
@@ -949,7 +941,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
}
case 0x66: // Operand Size prefix
DecodeInst->Flags |= DecodeFlags::FLAG_OPERAND_SIZE;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
break;
case 0x67: // Address Size override prefix
@@ -980,11 +972,11 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
break;
case 0xF2: // REPNE prefix
DecodeInst->Flags |= DecodeFlags::FLAG_REPNE_PREFIX;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
break;
case 0xF3: // REP prefix
DecodeInst->Flags |= DecodeFlags::FLAG_REP_PREFIX;
LastEscapePrefix = Op;
DecodeInst->LastEscapePrefix = Op;
break;
case 0x64: // FS prefix
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_FS_PREFIX;
@@ -1063,10 +1055,10 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
if (DecodeInst->OP == 0x8b && DecodeInst->Src[0].IsGPRIndirect() &&
IsKnownAtomicDisplacement(DecodeInst->Src[0].Data.GPRIndirect.Displacement)) {
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
DecodeInst->ForceTSO = true;
}
if (DecodeInst->OP == 0x89 && DecodeInst->Dest.IsGPRIndirect() && IsKnownAtomicDisplacement(DecodeInst->Dest.Data.GPRIndirect.Displacement)) {
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
DecodeInst->ForceTSO = true;
}
}
@@ -1309,7 +1301,7 @@ const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, u
return _InstStream - EntryPoint + RIP;
}
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
BlockInfo.TotalInstructionCount = 0;
BlockInfo.Blocks.clear();
+1 -2
View File
@@ -49,7 +49,7 @@ public:
};
Decoder(FEXCore::Core::InternalThreadState* Thread);
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
const DecodedBlockInformation* GetDecodedBlockInfo() const {
return &BlockInfo;
@@ -125,7 +125,6 @@ private:
static constexpr size_t MAX_INST_SIZE = 15;
uint8_t InstructionSize {};
std::array<uint8_t, MAX_INST_SIZE> Instruction;
uint8_t LastEscapePrefix {};
FEXCore::X86Tables::DecodedInst* DecodeInst;
// This is for multiblock data tracking
@@ -87,10 +87,12 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SINCOS>::handle)};
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
// Double Precision Binary
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_F64_F64_PTR],
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle)};
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
@@ -218,21 +220,21 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
return true; \
}
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
#define COMMON_UNARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
return true; \
}
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
#define COMMON_UNARYPAIR_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64x2_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
return true; \
}
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
#define COMMON_BINARY_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = {FABI_F64_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
return true; \
return true; \
}
// Unary
+39 -34
View File
@@ -588,7 +588,7 @@ DEF_OP(ShiftFlags) {
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, TMP1, &Done);
cbz(EmitSize, TMP1, &Done);
{
// PF/SF/ZF/OF
if (OpSize >= IR::OpSize::i32Bit) {
@@ -652,7 +652,7 @@ DEF_OP(ShiftFlags) {
msr(ARMEmitter::SystemRegister::NZCV, TMP2);
}
}
(void)Bind(&Done);
Bind(&Done);
// TODO: Make RA less dumb so this can't happen (e.g. with late-kill).
if (PFOutput != PFTemp) {
@@ -669,7 +669,7 @@ DEF_OP(RotateFlags) {
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, Shift, &Done);
cbz(EmitSize, Shift, &Done);
{
// Extract the last bit shifted in to CF
const auto BitSize = IR::OpSizeToSize(Op->Size) * 8;
@@ -701,7 +701,7 @@ DEF_OP(RotateFlags) {
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
}
}
(void)Bind(&Done);
Bind(&Done);
}
DEF_OP(Extr) {
@@ -767,14 +767,14 @@ DEF_OP(PDep) {
// Now, they're copied, so we can start setting Dest (even if it overlaps with
// one of them). Handle early exit case
mov(EmitSize, Dest, 0);
(void)cbz(EmitSize, OrigMask, &Done);
cbz(EmitSize, OrigMask, &Done);
// Setup for first iteration
neg(EmitSize, T0, Mask);
and_(EmitSize, T0, T0, Mask);
// Main loop
(void)Bind(&NextBit);
Bind(&NextBit);
sbfx(EmitSize, T1, Input, 0, 1);
eor(EmitSize, Mask, Mask, T0);
and_(EmitSize, T0, T1, T0);
@@ -782,10 +782,10 @@ DEF_OP(PDep) {
orr(EmitSize, Dest, Dest, T0);
lsr(EmitSize, Input, Input, 1);
and_(EmitSize, T0, Mask, T1);
(void)cbnz(EmitSize, T0, &NextBit);
cbnz(EmitSize, T0, &NextBit);
// All done with nothing to do.
(void)Bind(&Done);
Bind(&Done);
}
}
@@ -821,27 +821,27 @@ DEF_OP(PExt) {
ARMEmitter::BackwardLabel NextBit;
ARMEmitter::ForwardLabel Done;
(void)cbz(EmitSize, Mask, &EarlyExit);
cbz(EmitSize, Mask, &EarlyExit);
mov(EmitSize, MaskReg, Mask);
mov(EmitSize, ValueReg, Input);
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
// Main loop
(void)Bind(&NextBit);
(void)cbz(EmitSize, MaskReg, &Done);
Bind(&NextBit);
cbz(EmitSize, MaskReg, &Done);
clz(EmitSize, BitReg, MaskReg);
lslv(EmitSize, ValueReg, ValueReg, BitReg);
lslv(EmitSize, MaskReg, MaskReg, BitReg);
extr(EmitSize, Dest, Dest, ValueReg, OpSizeBitsM1);
bfc(EmitSize, MaskReg, OpSizeBitsM1, 1);
(void)b(&NextBit);
b(&NextBit);
// Early exit
(void)Bind(&EarlyExit);
Bind(&EarlyExit);
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
// All done with nothing to do.
(void)Bind(&Done);
Bind(&Done);
}
}
@@ -909,7 +909,7 @@ DEF_OP(Div) {
eor(EmitSize, TMP1, TMP1, Upper);
// If the sign bit matches then the result is zero
(void)cbz(EmitSize, TMP1, &Only64Bit);
cbz(EmitSize, TMP1, &Only64Bit);
// Long divide
{
@@ -928,17 +928,17 @@ DEF_OP(Div) {
mov(EmitSize, Remainder, TMP2);
// Skip 64-bit path
(void)b(&LongDIVRet);
b(&LongDIVRet);
}
(void)Bind(&Only64Bit);
Bind(&Only64Bit);
// 64-Bit only
{
sdiv(EmitSize, Quotient, Lower, Divisor);
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
}
(void)Bind(&LongDIVRet);
Bind(&LongDIVRet);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown DIV Size: {}", OpSize); break;
@@ -992,7 +992,7 @@ DEF_OP(UDiv) {
// Check the upper bits for zero
// If the upper bits are zero then we can do a 64-bit divide
(void)cbz(EmitSize, Upper, &Only64Bit);
cbz(EmitSize, Upper, &Only64Bit);
// Long divide
{
@@ -1011,17 +1011,17 @@ DEF_OP(UDiv) {
mov(EmitSize, Remainder, TMP2);
// Skip 64-bit path
(void)b(&LongDIVRet);
b(&LongDIVRet);
}
(void)Bind(&Only64Bit);
Bind(&Only64Bit);
// 64-Bit only
{
udiv(EmitSize, Quotient, Lower, Divisor);
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
}
(void)Bind(&LongDIVRet);
Bind(&LongDIVRet);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown LUDIV Size: {}", OpSize); break;
@@ -1046,19 +1046,24 @@ DEF_OP(Popcount) {
if (CTX->HostFeatures.SupportsCSSC) {
switch (OpSize) {
case IR::OpSize::i8Bit:
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i16Bit:
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i32Bit: cnt(ARMEmitter::Size::i32Bit, Dst, Src); break;
case IR::OpSize::i64Bit: cnt(ARMEmitter::Size::i64Bit, Dst, Src); break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
case IR::OpSize::i8Bit:
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i16Bit:
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
break;
case IR::OpSize::i32Bit:
cnt(ARMEmitter::Size::i32Bit, Dst, Src);
break;
case IR::OpSize::i64Bit:
cnt(ARMEmitter::Size::i64Bit, Dst, Src);
break;
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
}
} else {
}
else {
switch (OpSize) {
case IR::OpSize::i8Bit:
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
@@ -63,7 +63,7 @@ void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
auto CurrentCursor = GetCursorAddress<uint8_t*>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
BindOrRestart(&Lit.Loc);
Bind(&Lit.Loc);
dc64(Lit.Lit);
Relocations.emplace_back(Lit.MoveABI);
}
@@ -81,32 +81,35 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
Relocations.emplace_back(MoveABI);
}
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation> Relocations) {
const auto OrigBase = GetBufferBase();
const auto OrigSize = GetBufferSize();
const auto OrigOffset = GetCursorOffset();
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
const char* EntryRelocations) {
size_t DataIndex {};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
SetBuffer(reinterpret_cast<std::uint8_t*>(Code.data()), Code.size_bytes());
for (auto& Reloc : Relocations) {
switch (Reloc.Header.Type) {
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc.NamedSymbolLiteral.Symbol);
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
SetCursorOffset(Reloc.NamedSymbolLiteral.Offset);
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
dc64(Pointer);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(Reloc.NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
@@ -114,27 +117,18 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Co
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
SetBuffer(OrigBase, OrigSize);
SetCursorOffset(OrigOffset);
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(Reloc.GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIPMove.RegisterIndex), Pointer, true);
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
}
}
SetBuffer(OrigBase, OrigSize);
SetCursorOffset(OrigOffset);
return true;
}
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations() {
return std::move(Relocations);
}
} // namespace FEXCore::CPU
+34 -34
View File
@@ -62,27 +62,27 @@ DEF_OP(CASPair) {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
(void)Bind(&LoopTop);
Bind(&LoopTop);
// This instruction sequence must be synced with HandleCASPAL_Armv8.
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
cmp(EmitSize, TMP2, Expected0);
ccmp(EmitSize, TMP3, Expected1, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
stlxp(EmitSize, TMP2, Desired0, Desired1, MemSrc);
(void)cbnz(EmitSize, TMP2, &LoopTop);
cbnz(EmitSize, TMP2, &LoopTop);
mov(EmitSize, Dst0, Expected0);
mov(EmitSize, Dst1, Expected1);
(void)b(&LoopExpected);
b(&LoopExpected);
(void)Bind(&LoopNotExpected);
Bind(&LoopNotExpected);
mov(EmitSize, Dst0, TMP2.R());
mov(EmitSize, Dst1, TMP3.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
(void)Bind(&LoopExpected);
Bind(&LoopExpected);
// Restore
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
@@ -114,7 +114,7 @@ DEF_OP(CAS) {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
if (IROp->Size == IR::OpSize::i8Bit) {
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
@@ -123,18 +123,18 @@ DEF_OP(CAS) {
} else {
cmp(EmitSize, TMP2, Expected);
}
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
stlxr(SubEmitSize, TMP3, Desired, MemSrc);
(void)cbnz(EmitSize, TMP3, &LoopTop);
cbnz(EmitSize, TMP3, &LoopTop);
mov(EmitSize, Dst, Expected);
(void)b(&LoopExpected);
b(&LoopExpected);
(void)Bind(&LoopNotExpected);
Bind(&LoopNotExpected);
mov(EmitSize, Dst, TMP2.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
(void)Bind(&LoopExpected);
Bind(&LoopExpected);
}
}
@@ -150,11 +150,11 @@ DEF_OP(AtomicXor) {
steorl(SubEmitSize, Src, MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
eor(EmitSize, TMP2, TMP2, Src);
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
(void)cbnz(EmitSize, TMP2, &LoopTop);
cbnz(EmitSize, TMP2, &LoopTop);
}
}
@@ -179,10 +179,10 @@ DEF_OP(AtomicSwap) {
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
stlxr(SubEmitSize, TMP4, Src, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
ubfm(EmitSize, GetReg(Node), TMP2, 0, IR::OpSizeAsBits(OpSize) - 1);
}
}
@@ -199,11 +199,11 @@ DEF_OP(AtomicFetchAdd) {
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
add(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -221,11 +221,11 @@ DEF_OP(AtomicFetchSub) {
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
sub(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -243,11 +243,11 @@ DEF_OP(AtomicFetchAnd) {
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
and_(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -264,11 +264,11 @@ DEF_OP(AtomicFetchCLR) {
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
bic(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -285,11 +285,11 @@ DEF_OP(AtomicFetchOr) {
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
orr(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -306,11 +306,11 @@ DEF_OP(AtomicFetchXor) {
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
eor(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -326,20 +326,20 @@ DEF_OP(AtomicFetchNeg) {
// Use a CAS loop to avoid needing to emulate unaligned LLSC atomics
ldr(SubEmitSize, TMP2, MemSrc);
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
mov(EmitSize, TMP4, TMP2);
neg(EmitSize, TMP3, TMP2);
casal(SubEmitSize, TMP2, TMP3, MemSrc);
sub(EmitSize, TMP3, TMP2, TMP4);
(void)cbnz(EmitSize, TMP3, &LoopTop);
cbnz(EmitSize, TMP3, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
neg(EmitSize, TMP3, TMP2);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
(void)cbnz(EmitSize, TMP4, &LoopTop);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
@@ -359,11 +359,11 @@ DEF_OP(TelemetrySetValue) {
stsetl(ARMEmitter::SubRegSize::i64Bit, TMP1, TMP2);
} else {
ARMEmitter::BackwardLabel LoopTop;
(void)Bind(&LoopTop);
Bind(&LoopTop);
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
}
#endif
}
+18 -18
View File
@@ -141,7 +141,7 @@ DEF_OP(ExitFunction) {
if (!Op->CallReturnBlock.IsInvalid()) {
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
(void)adr(TMP1, &l_CallReturn);
adr(TMP1, &l_CallReturn);
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
} else {
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
@@ -149,16 +149,16 @@ DEF_OP(ExitFunction) {
} else if (Op->Hint == IR::BranchHint::CheckTF) {
ARMEmitter::ForwardLabel TFUnset;
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, NewRIP);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
blr(TMP2);
(void)Bind(&TFUnset);
Bind(&TFUnset);
}
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
(void)Bind(&l_CallReturn);
Bind(&l_CallReturn);
#ifdef _M_ARM_64EC
}
#endif
@@ -170,7 +170,7 @@ DEF_OP(ExitFunction) {
// First try to pop from the call-ret stack, otherwise follow the normal path (but ending in a ret)
ldp<ARMEmitter::IndexType::POST>(TMP1, TMP2, REG_CALLRET_SP, 0x10);
sub(TMP1, TMP1, RipReg.X());
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
}
// L1 Cache
@@ -187,23 +187,23 @@ DEF_OP(ExitFunction) {
// Note: sub+cbnz used over cmp+br to preserve flags.
sub(TMP1, TMP1, RipReg.X());
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
(void)Bind(&SkipFullLookup);
Bind(&SkipFullLookup);
if (Op->Hint == IR::BranchHint::Call) {
ARMEmitter::ForwardLabel l_CallReturn;
if (!Op->CallReturnBlock.IsInvalid()) {
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
(void)adr(TMP1, &l_CallReturn);
adr(TMP1, &l_CallReturn);
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
} else {
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
}
blr(TMP2);
(void)Bind(&l_CallReturn);
Bind(&l_CallReturn);
} else if (Op->Hint == IR::BranchHint::Return) {
ret(TMP2);
} else {
@@ -224,7 +224,7 @@ DEF_OP(CondJump) {
auto TrueTargetLabel = JumpTarget(Op->TrueBlock);
if (Op->FromNZCV) {
b_OrRestart(MapCC(Op->Cond), TrueTargetLabel);
b(MapCC(Op->Cond), TrueTargetLabel);
} else {
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
@@ -237,16 +237,16 @@ DEF_OP(CondJump) {
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
cbz_OrRestart(Size, Reg, TrueTargetLabel);
cbz(Size, Reg, TrueTargetLabel);
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
cbnz_OrRestart(Size, Reg, TrueTargetLabel);
cbnz(Size, Reg, TrueTargetLabel);
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
tbz_OrRestart(Reg, Const, TrueTargetLabel);
tbz(Reg, Const, TrueTargetLabel);
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
tbnz_OrRestart(Reg, Const, TrueTargetLabel);
tbnz(Reg, Const, TrueTargetLabel);
} else {
LOGMAN_THROW_A_FMT(false, "CondJump expected simple condition");
}
@@ -458,7 +458,7 @@ DEF_OP(ValidateCode) {
while (len >= Size) {
LoadData();
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &Fail);
len -= Size;
Offset += Size;
}
@@ -486,10 +486,10 @@ DEF_OP(ValidateCode) {
ARMEmitter::ForwardLabel End;
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
b_OrRestart(&End);
BindOrRestart(&Fail);
b(&End);
Bind(&Fail);
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
BindOrRestart(&End);
Bind(&End);
}
DEF_OP(ThreadRemoveCodeEntry) {
@@ -1,35 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/fextl/vector.h>
#include <cstdint>
namespace FEXCore::CPU {
union Relocation;
} // namespace FEXCore::CPU
namespace FEXCore::Core {
struct DebugDataSubblock {
uint32_t HostCodeOffset;
uint32_t HostCodeSize;
};
struct DebugDataGuestOpcode {
uint64_t GuestEntryOffset;
ptrdiff_t HostEntryOffset;
};
/**
* @brief Contains debug data for a block of code for later debugger analysis
*
* Needs to remain around for as long as the code could be executed at least
*/
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
fextl::vector<DebugDataSubblock> Subblocks;
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
};
} // namespace FEXCore::Core
+73 -80
View File
@@ -11,12 +11,15 @@ desc: Main glue logic of the arm64 splatter backend
$end_info$
*/
#include "Common/SoftFloat.h"
#include "FEXCore/Utils/Telemetry.h"
#include "FEXCore/Utils/TypeDefines.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/JIT/DebugData.h"
#include "Interface/Core/JIT/JITClass.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Utils/MemberFunctionToPointer.h"
@@ -27,16 +30,15 @@ $end_info$
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/LongJump.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <cstdio>
#include <cstring>
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include <stdio.h>
#include <unistd.h>
#include <string.h>
#include <limits>
namespace {
struct DivRem {
@@ -536,8 +538,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
} else {
{
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
auto lk_inval =
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
HostCode = Thread->LookupCache->FindBlock(GuestRip);
}
if (!HostCode) {
@@ -667,6 +668,15 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
CurrentCodeBuffer = CodeBuffers.GetLatest();
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
// Setup dynamic dispatch.
if (ParanoidTSO()) {
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
} else {
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
}
}
void Arm64JITCore::EmitDetectionString() {
@@ -730,48 +740,48 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
}
}
void Arm64JITCore::EmitTFCheck() {
ARMEmitter::ForwardLabel l_TFUnset;
ARMEmitter::ForwardLabel l_TFBlocked;
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
if (CheckTF) {
ARMEmitter::ForwardLabel l_TFUnset;
ARMEmitter::ForwardLabel l_TFBlocked;
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
(void)tbz(TMP1, 1, &l_TFBlocked);
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
tbz(TMP1, 1, &l_TFBlocked);
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
.FaultToTopAndGeneratedException = 1,
.Signal = Core::FAULT_SIGTRAP,
.TrapNo = X86State::X86_TRAPNO_DB,
.si_code = 2,
.err_code = 0,
};
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
.FaultToTopAndGeneratedException = 1,
.Signal = Core::FAULT_SIGTRAP,
.TrapNo = X86State::X86_TRAPNO_DB,
.si_code = 2,
.err_code = 0,
};
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
(void)Bind(&l_TFBlocked);
// If TF was blocked for this instruction, unblock it for the next.
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)Bind(&l_TFUnset);
}
Bind(&l_TFBlocked);
// If TF was blocked for this instruction, unblock it for the next.
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
Bind(&l_TFUnset);
}
void Arm64JITCore::EmitSuspendInterruptCheck() {
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
// Trigger a fault if there are any pending interrupts
// Used only for suspend on WIN32 at the moment
@@ -786,19 +796,17 @@ void Arm64JITCore::EmitSuspendInterruptCheck() {
ARMEmitter::ForwardLabel l_NoSuspend;
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
brk(SuspendMagic);
(void)Bind(&l_NoSuspend);
Bind(&l_NoSuspend);
#endif
}
void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF) {
// Get the address of the JITCodeHeader and store in to the core state.
// Two instruction cost, each 1 cycle.
adr_OrRestart(TMP1, &HeaderLabel);
adr(TMP1, &HeaderLabel);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
if (CheckTF) {
EmitTFCheck();
}
EmitInterruptChecks(CheckTF);
if (SpillSlots) {
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
@@ -815,25 +823,16 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
this->Entry = Entry;
this->DebugData = DebugData;
this->IR = IR;
RequiresFarARM64Jumps = false;
switch (static_cast<RestartOptions::Control>(FEXCore::LongJump::SetJump(RestartControl.RestartJump))) {
case RestartOptions::Control::Incoming:
// Nothing
break;
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
default: ERROR_AND_DIE_FMT("Unhandled Arm64 restart condition!");
}
uint32_t SSACount = IR->GetSSACount();
JumpTargets.clear();
CallReturnTargets.clear();
PendingJumpThunks.clear();
uint32_t SSACount = IR->GetSSACount();
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
this->Entry = Entry;
this->DebugData = DebugData;
this->IR = IR;
CodeData.EntryPoints.clear();
// Fairly excessive buffer range to make sure we don't overflow
@@ -848,7 +847,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Put the code header at the start of the data block.
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
(void)Bind(&JITCodeHeaderLabel);
Bind(&JITCodeHeaderLabel);
JITCodeHeader* CodeHeader = GetCursorAddress<JITCodeHeader*>();
CursorIncrement(sizeof(JITCodeHeader));
@@ -893,10 +892,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// if there's a pending branch, and it is not fall-through
if (PendingTargetLabel && PendingTargetLabel != Target) {
if (PendingTargetLabel->Backward.Location) {
EmitSuspendInterruptCheck();
}
b_OrRestart(PendingTargetLabel);
b(PendingTargetLabel);
PendingTargetLabel = nullptr;
}
@@ -906,14 +902,14 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
const auto IsReturnTarget = CallReturnTargets.try_emplace(Node).first;
if (PendingTargetLabel) {
// If there is a fallthrough branch to this block, skip over the entrypoint code.
b_OrRestart(Target);
b(Target);
} else if (PendingCallReturnTargetLabel && PendingCallReturnTargetLabel != &IsReturnTarget->second) {
// If we just emitted a call, but the block we're now emitting is not the return block so don't fallthrough.
b_OrRestart(PendingCallReturnTargetLabel);
b(PendingCallReturnTargetLabel);
}
PendingCallReturnTargetLabel = nullptr;
BindOrRestart(&IsReturnTarget->second);
Bind(&IsReturnTarget->second);
CodeData.EntryPoints.emplace(BlockStartRIP, GetCursorAddress<uint8_t*>());
DebugData->GuestOpcodes.push_back({BlockIROp->GuestEntryOffset, GetCursorAddress<uint8_t*>() - CodeData.BlockBegin});
@@ -922,12 +918,12 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
if (PendingCallReturnTargetLabel) {
// If there is still a pending call return target, then the block we're emitting is not the return block so don't fallthrough.
b_OrRestart(PendingCallReturnTargetLabel);
b(PendingCallReturnTargetLabel);
PendingCallReturnTargetLabel = nullptr;
}
PendingTargetLabel = nullptr;
BindOrRestart(Target);
Bind(Target);
}
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
@@ -951,10 +947,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Make sure last branch is generated. It certainly can't be eliminated here.
if (PendingTargetLabel) {
if (PendingTargetLabel->Backward.Location) {
EmitSuspendInterruptCheck();
}
b_OrRestart(PendingTargetLabel);
b(PendingTargetLabel);
}
PendingTargetLabel = nullptr;
@@ -965,21 +958,21 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
ARMEmitter::ForwardLabel l_DoLink;
uint64_t ThunkAddress = GetCursorAddress<uint64_t>();
BindOrRestart(&PendingJumpThunk.Label);
b_OrRestart(&l_DoLink);
Bind(&PendingJumpThunk.Label);
b(&l_DoLink);
br(TMP1);
BindOrRestart(&l_DoLink);
Bind(&l_DoLink);
ldr(TMP1, &l_ExitLink);
blr(TMP1);
// This is a ExitFunctionLinkData struct
BindOrRestart(&l_ExitLink);
Bind(&l_ExitLink);
dc64(0); // HostCode
dc64(PendingJumpThunk.GuestRIP); // GuestRIP
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
}
BindOrRestart(&l_ExitLink);
Bind(&l_ExitLink);
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
// CodeSize not including the header or tail data.
+14 -209
View File
@@ -19,7 +19,6 @@ $end_info$
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <FEXCore/Utils/LongJump.h>
#include <CodeEmitter/Emitter.h>
@@ -55,7 +54,6 @@ public:
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
const bool HostSupportsSVE128 {};
const bool HostSupportsSVE256 {};
@@ -63,19 +61,6 @@ private:
const bool HostSupportsRPRES {};
const bool HostSupportsAFP {};
struct RestartOptions {
FEXCore::LongJump::JumpBuf RestartJump;
enum class Control : uint64_t {
Incoming = 0,
EnableFarARM64Jumps = 1,
};
};
// FEXCore makes assumptions in the JIT about certain conditions being true.
// In the rare case when those assumptions are broken, FEX needs to safely restart the JIT.
RestartOptions RestartControl {};
bool RequiresFarARM64Jumps {};
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
ARMEmitter::BiDirectionalLabel* PendingCallReturnTargetLabel {};
FEXCore::Context::ContextImpl* CTX {};
@@ -330,199 +315,14 @@ private:
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
auto& Thunk = PendingJumpThunks.back();
BindOrRestart(&Thunk.Label);
Bind(&Thunk.Label);
if (Call) {
bl_OrRestart(&Thunk.Label);
bl(&Thunk.Label);
} else {
b_OrRestart(&Thunk.Label);
b(&Thunk.Label);
}
}
// Restart helpers
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void bl_OrRestart(T* Label) {
if (bl(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void b_OrRestart(T* Label) {
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void b_OrRestart(ARMEmitter::Condition Cond, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)b(InvertCondition(Cond), &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (b(Cond, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void cbz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)cbnz(s, rt, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (cbz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void cbnz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)cbz(s, rt, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (cbnz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void tbz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)tbnz(rt, Bit, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (tbz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void tbnz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
if (RequiresFarARM64Jumps) {
ARMEmitter::ForwardLabel Skip {};
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
(void)tbz(rt, Bit, &Skip);
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
}
(void)Bind(&Skip);
return;
}
if (tbnz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void adr_OrRestart(ARMEmitter::Register rd, T* Label) {
if (adr(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Long ADR currently unsupported!");
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void adrp_OrRestart(ARMEmitter::Register rd, T* Label) {
if (adrp(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
return;
}
// We can support this but currently unnecessary.
ERROR_AND_DIE_FMT("Long ADRP currently unsupported!");
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
template<typename T>
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
void BindOrRestart(T* Label) {
if (Bind(Label)) {
return;
}
if (RequiresFarARM64Jumps) {
// This should have been caught before this point.
ERROR_AND_DIE_FMT("Oops. Unhandled long bind.");
return;
}
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
IR::RegisterAllocationPass* RAPass {};
@@ -581,9 +381,7 @@ private:
fextl::vector<FEXCore::CPU::Relocation> Relocations;
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation>);
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() override;
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
/** @} */
@@ -608,14 +406,21 @@ private:
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
void EmitTFCheck();
void EmitSuspendInterruptCheck();
void EmitInterruptChecks(bool CheckTF);
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
// Runtime selection;
// Load and store TSO memory style
OpType RT_LoadMemTSO;
OpType RT_StoreMemTSO;
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
// Dynamic Dispatcher supporting operations
DEF_OP(ParanoidLoadMemTSO);
DEF_OP(ParanoidStoreMemTSO);
///< Unhandled handler
DEF_OP(Unhandled);
+235 -99
View File
@@ -348,25 +348,6 @@ DEF_OP(StoreContextIndexed) {
}
}
DEF_OP(FormContextAddress) {
const auto Op = IROp->C<IR::IROp_FormContextAddress>();
const auto Index = GetReg(Op->Index);
const auto Dst = GetReg(Node);
switch (Op->Stride) {
case 1:
case 2:
case 4:
case 8:
case 16:
case 32: {
add(ARMEmitter::Size::i64Bit, Dst, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled FormContextAddress stride: {}", Op->Stride); break;
}
}
DEF_OP(SpillRegister) {
const auto Op = IROp->C<IR::IROp_SpillRegister>();
const auto OpSize = IROp->Size;
@@ -776,10 +757,8 @@ DEF_OP(LoadMemTSO) {
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
}
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
@@ -793,10 +772,8 @@ DEF_OP(LoadMemTSO) {
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
}
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
@@ -810,10 +787,8 @@ DEF_OP(LoadMemTSO) {
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
}
} else {
const auto Dst = GetVReg(Node);
@@ -917,7 +892,7 @@ DEF_OP(VLoadVectorMasked) {
// If the sign bit is zero then skip the load
ARMEmitter::ForwardLabel Skip {};
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
// Do the gather load for this element into the destination
switch (IROp->ElementSize) {
case IR::OpSize::i8Bit: ld1<ARMEmitter::SubRegSize::i8Bit>(TempDst.Q(), i, TempMemReg); break;
@@ -928,7 +903,7 @@ DEF_OP(VLoadVectorMasked) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
}
(void)Bind(&Skip);
Bind(&Skip);
if ((i + 1) != NumElements) {
// Handle register rename to save a move.
@@ -1018,7 +993,7 @@ DEF_OP(VStoreVectorMasked) {
// If the sign bit is zero then skip the load
ARMEmitter::ForwardLabel Skip {};
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
// Do the gather load for this element into the destination
switch (IROp->ElementSize) {
case IR::OpSize::i8Bit: st1<ARMEmitter::SubRegSize::i8Bit>(RegData.Q(), i, TempMemReg); break;
@@ -1029,7 +1004,7 @@ DEF_OP(VStoreVectorMasked) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
}
(void)Bind(&Skip);
Bind(&Skip);
if ((i + 1) != NumElements) {
// Handle register rename to save a move.
@@ -1107,7 +1082,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
PerformMove(ElementSize, WorkingReg, MaskReg, i);
// Skip if the mask's sign bit isn't set
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
// Extract Index Element
if ((IndexElement * IR::OpSizeToSize(VectorIndexSize)) >= 16) {
@@ -1145,7 +1120,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, ElementSize); FEX_UNREACHABLE;
}
(void)Bind(&Skip);
Bind(&Skip);
}
if (NeedsDestTmp) {
@@ -1783,10 +1758,8 @@ DEF_OP(StoreMemTSO) {
// 8bit load is always aligned to natural alignment
stlurb(Src, MemReg, Offset);
} else {
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
switch (OpSize) {
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
@@ -1801,10 +1774,8 @@ DEF_OP(StoreMemTSO) {
// 8bit load is always aligned to natural alignment
stlrb(Src, MemReg);
} else {
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Half-barrier once back-patched.
nop();
}
// Half-barrier once back-patched.
nop();
switch (OpSize) {
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
@@ -1880,7 +1851,7 @@ DEF_OP(MemSet) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
(void)tbnz(DirectionReg, 1, &BackwardImpl);
tbnz(DirectionReg, 1, &BackwardImpl);
}
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
@@ -1898,9 +1869,7 @@ DEF_OP(MemSet) {
// 8bit load is always aligned to natural alignment
stlrb(Value.W(), TMP2);
} else {
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
nop();
}
nop();
switch (OpSize) {
case 2: stlrh(Value.W(), TMP2); break;
case 4: stlr(Value.W(), TMP2); break;
@@ -1930,7 +1899,7 @@ DEF_OP(MemSet) {
ARMEmitter::ForwardLabel DoneInternal {};
// Early exit if zero count.
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AgainInternal256Exit {};
@@ -1947,50 +1916,50 @@ DEF_OP(MemSet) {
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
tbnz(TMP1, 63, &AgainInternal128Exit);
// Fill VTMP2 with the set pattern
dup(SubRegSize, VTMP2.Q(), Value);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
tbnz(TMP1, 63, &AgainInternal256Exit);
(void)Bind(&AgainInternal256);
Bind(&AgainInternal256);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)tbz(TMP1, 63, &AgainInternal256);
tbz(TMP1, 63, &AgainInternal256);
(void)Bind(&AgainInternal256Exit);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
(void)Bind(&AgainInternal128);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbz(TMP1, 63, &AgainInternal128);
tbz(TMP1, 63, &AgainInternal128);
(void)Bind(&AgainInternal128Exit);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
}
}
(void)Bind(&AgainInternal);
Bind(&AgainInternal);
if (IsAtomic) {
MemStoreTSO(Value, OpSize, SizeDirection);
} else {
MemStore(Value, OpSize, SizeDirection);
}
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
(void)Bind(&DoneInternal);
Bind(&DoneInternal);
if (SizeDirection >= 0) {
switch (OpSize) {
@@ -2020,12 +1989,12 @@ DEF_OP(MemSet) {
EmitMemset(Direction);
if (Direction == 1) {
(void)b(&Done);
(void)Bind(&BackwardImpl);
b(&Done);
Bind(&BackwardImpl);
}
}
(void)Bind(&Done);
Bind(&Done);
// Destination already set to the final pointer.
}
}
@@ -2075,7 +2044,7 @@ DEF_OP(MemCpy) {
if (!DirectionIsInline) {
// Backward or forwards implementation depends on flag
(void)tbnz(DirectionReg, 1, &BackwardImpl);
tbnz(DirectionReg, 1, &BackwardImpl);
}
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
@@ -2118,11 +2087,9 @@ DEF_OP(MemCpy) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Placeholders for backpatching barriers (one per load/store)
nop();
nop();
}
// Placeholders for backpatching barriers (one per load/store)
nop();
nop();
switch (OpSize) {
case 2: stlrh(TMP4.W(), TMP2); break;
@@ -2144,11 +2111,9 @@ DEF_OP(MemCpy) {
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
}
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
// Placeholders for backpatching barriers (one per load/store)
nop();
nop();
}
// Placeholders for backpatching barriers (one per load/store)
nop();
nop();
switch (OpSize) {
case 2: stlrh(TMP4.W(), TMP2); break;
@@ -2176,7 +2141,7 @@ DEF_OP(MemCpy) {
ARMEmitter::ForwardLabel DoneInternal {};
// Early exit if zero count.
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (!IsAtomic) {
ARMEmitter::ForwardLabel AbsPos {};
@@ -2186,11 +2151,11 @@ DEF_OP(MemCpy) {
ARMEmitter::BackwardLabel AgainInternal256 {};
sub(ARMEmitter::Size::i64Bit, TMP4, TMP2, TMP3);
(void)tbz(TMP4, 63, &AbsPos);
tbz(TMP4, 63, &AbsPos);
neg(ARMEmitter::Size::i64Bit, TMP4, TMP4);
(void)Bind(&AbsPos);
Bind(&AbsPos);
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
(void)tbnz(TMP4, 63, &AgainInternal);
tbnz(TMP4, 63, &AgainInternal);
if (Direction == -1) {
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
@@ -2202,30 +2167,30 @@ DEF_OP(MemCpy) {
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
// single copy loop if size < 64.
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
tbnz(TMP1, 63, &AgainInternal128Exit);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
tbnz(TMP1, 63, &AgainInternal256Exit);
(void)Bind(&AgainInternal256);
Bind(&AgainInternal256);
MemCpy(32, 32 * Direction);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)tbz(TMP1, 63, &AgainInternal256);
tbz(TMP1, 63, &AgainInternal256);
(void)Bind(&AgainInternal256Exit);
Bind(&AgainInternal256Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
(void)Bind(&AgainInternal128);
tbnz(TMP1, 63, &AgainInternal128Exit);
Bind(&AgainInternal128);
MemCpy(32, 32 * Direction);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)tbz(TMP1, 63, &AgainInternal128);
tbz(TMP1, 63, &AgainInternal128);
(void)Bind(&AgainInternal128Exit);
Bind(&AgainInternal128Exit);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
if (Direction == -1) {
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
@@ -2233,16 +2198,16 @@ DEF_OP(MemCpy) {
}
}
(void)Bind(&AgainInternal);
Bind(&AgainInternal);
if (IsAtomic) {
MemCpyTSO(OpSize, SizeDirection);
} else {
MemCpy(OpSize, SizeDirection);
}
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
(void)Bind(&DoneInternal);
Bind(&DoneInternal);
// Needs to use temporaries just in case of overwrite
mov(TMP1, MemRegDest.X());
@@ -2300,15 +2265,186 @@ DEF_OP(MemCpy) {
for (int32_t Direction : {1, -1}) {
EmitMemcpy(Direction);
if (Direction == 1) {
(void)b(&Done);
(void)Bind(&BackwardImpl);
b(&Done);
Bind(&BackwardImpl);
}
}
(void)Bind(&Done);
Bind(&Done);
// Destination already set to the final pointer.
}
}
DEF_OP(ParanoidLoadMemTSO) {
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
const auto OpSize = IROp->Size;
auto MemReg = GetReg(Op->Addr);
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
if (!IsInlineConstant(Op->Offset, &Offset)) {
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
}
}
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
const auto Dst = GetReg(Node);
ldapurb(Dst, MemReg, Offset);
} else {
switch (OpSize) {
case IR::OpSize::i16Bit: ldapurh(Dst, MemReg, Offset); break;
case IR::OpSize::i32Bit: ldapur(Dst.W(), MemReg, Offset); break;
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
ldaprb(Dst.W(), MemReg);
} else {
switch (OpSize) {
case IR::OpSize::i16Bit: ldaprh(Dst.W(), MemReg); break;
case IR::OpSize::i32Bit: ldapr(Dst.W(), MemReg); break;
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Dst = GetReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit: ldarb(Dst, MemReg); break;
case IR::OpSize::i16Bit: ldarh(Dst, MemReg); break;
case IR::OpSize::i32Bit: ldar(Dst.W(), MemReg); break;
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
} else {
const auto Dst = GetVReg(Node);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit:
ldarb(TMP1, MemReg);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
break;
case IR::OpSize::i16Bit:
ldarh(TMP1, MemReg);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
break;
case IR::OpSize::i32Bit:
ldar(TMP1.W(), MemReg);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
break;
case IR::OpSize::i64Bit:
ldar(TMP1, MemReg);
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
break;
case IR::OpSize::i128Bit:
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
clrex();
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
break;
case IR::OpSize::i256Bit:
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
dmb(ARMEmitter::BarrierScope::ISH);
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
dmb(ARMEmitter::BarrierScope::ISH);
break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
}
}
}
DEF_OP(ParanoidStoreMemTSO) {
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
const auto OpSize = IROp->Size;
auto MemReg = GetReg(Op->Addr);
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
uint64_t Offset = 0;
if (!Op->Offset.IsInvalid()) {
if (!IsInlineConstant(Op->Offset, &Offset)) {
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
}
}
if (OpSize == IR::OpSize::i8Bit) {
// 8bit load is always aligned to natural alignment
stlurb(Src, MemReg, Offset);
} else {
switch (OpSize) {
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
case IR::OpSize::i64Bit: stlur(Src.X(), MemReg, Offset); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
}
} else if (Op->Class == FEXCore::IR::GPRClass) {
const auto Src = GetZeroableReg(Op->Value);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
case IR::OpSize::i64Bit: stlr(Src.X(), MemReg); break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
} else {
const auto Src = GetVReg(Op->Value);
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
switch (OpSize) {
case IR::OpSize::i8Bit:
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
stlrb(TMP1, MemReg);
break;
case IR::OpSize::i16Bit:
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
stlrh(TMP1, MemReg);
break;
case IR::OpSize::i32Bit:
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
stlr(TMP1.W(), MemReg);
break;
case IR::OpSize::i64Bit:
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
stlr(TMP1, MemReg);
break;
case IR::OpSize::i128Bit: {
// Move vector to GPRs
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
umov<ARMEmitter::SubRegSize::i64Bit>(TMP2, Src, 1);
ARMEmitter::BackwardLabel B;
Bind(&B);
// ldaxp must not have both the destination registers be the same
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
break;
}
case IR::OpSize::i256Bit: {
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
dmb(ARMEmitter::BarrierScope::ISH);
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
dmb(ARMEmitter::BarrierScope::ISH);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
}
}
}
DEF_OP(CacheLineClear) {
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
dmb(ARMEmitter::BarrierScope::SY);
@@ -10,11 +10,10 @@ $end_info$
#endif
#include "Interface/Context/Context.h"
#include "Interface/Core/JIT/DebugData.h"
#include "Interface/Core/JIT/JITClass.h"
#include "FEXCore/Debug/InternalThreadState.h"
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
namespace FEXCore::CPU {
@@ -789,7 +789,7 @@ void OpDispatchBuilder::CondJUMPRCXOp(OpcodeArgs) {
StartNewBlock();
// Store the new RIP
ExitRelocatedPC(Op, Op->Src[0].Literal());
ExitRelocatedPC(Op, Op->Src[0].Data.Literal.Value);
}
// Failure to take branch
@@ -858,7 +858,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
StartNewBlock();
// Store the new RIP
ExitRelocatedPC(Op, Op->Src[1].Literal());
ExitRelocatedPC(Op, Op->Src[1].Data.Literal.Value);
}
// Failure to take branch
@@ -947,7 +947,7 @@ void OpDispatchBuilder::JUMPFARIndirectOp(OpcodeArgs) {
// This uses ModRM to determine its location
// No way to use this effectively in multiblock
Ref Src = MakeSegmentAddress(Op, Op->Dest);
AddressMode SrcCS = {.Base = Src, .Offset = 4, .AddrSize = OpSize::i64Bit};
AddressMode SrcCS = {.Base = Src, .Offset = 4, .AddrSize = OpSize::i64Bit};
auto RIPOffset = _LoadMemAutoTSO(GPRClass, OpSize::i32Bit, Src, OpSize::i8Bit);
auto NewSegmentCS = _LoadMemAutoTSO(GPRClass, OpSize::i16Bit, SrcCS, OpSize::i8Bit);
@@ -968,7 +968,7 @@ void OpDispatchBuilder::CALLFARIndirectOp(OpcodeArgs) {
BlockSetRIP = true;
Ref Src = MakeSegmentAddress(Op, Op->Dest);
AddressMode SrcCS = {.Base = Src, .Offset = 4, .AddrSize = OpSize::i64Bit};
AddressMode SrcCS = {.Base = Src, .Offset = 4, .AddrSize = OpSize::i64Bit};
auto RIPOffset = _LoadMemAutoTSO(GPRClass, OpSize::i32Bit, Src, OpSize::i8Bit);
auto NewSegmentCS = _LoadMemAutoTSO(GPRClass, OpSize::i16Bit, SrcCS, OpSize::i8Bit);
auto CurrentCS = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, cs_idx));
@@ -1285,7 +1285,10 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) {
DecodeFailure = true;
}
break;
default: UnimplementedOp(Op); return;
default:
LogMan::Msg::EFmt("Unknown segment register: {}", Op->Dest.Data.GPR.GPR);
DecodeFailure = true;
break;
}
} else {
Ref Segment {};
@@ -1323,7 +1326,10 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) {
Segment = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, fs_idx));
}
break;
default: UnimplementedOp(Op); return;
default:
LogMan::Msg::EFmt("Unknown segment register: {}", Op->Dest.Data.GPR.GPR);
DecodeFailure = true;
return;
}
if (DestIsMem(Op)) {
// If the destination is memory then we always store 16-bits only
@@ -1336,56 +1342,26 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) {
}
void OpDispatchBuilder::MOVOffsetOp(OpcodeArgs) {
Ref Src;
auto GenMemSrcFromOp = [&](size_t StartingSource) -> AddressMode {
const uint64_t Lower = Op->Src[StartingSource].Literal();
const uint64_t Upper = Op->Src[StartingSource + 1].Literal();
const uint64_t Combined = (Upper << 32) | Lower;
const auto GPRSize = GetGPROpSize();
AddressMode A {
.Segment = GetSegment(Op->Flags),
.Offset = static_cast<int64_t>(Combined),
.AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize,
.NonTSO = false,
};
return A;
};
switch (Op->OP) {
case 0xA0:
case 0xA1: {
case 0xA1:
// Source is memory(literal)
// Dest is GPR
Ref Src {};
if (Op->Src[0].Data.Literal.Size <= 4) {
Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.ForceLoad = true});
} else {
const auto OpSize = OpSizeFromSrc(Op);
auto A = GenMemSrcFromOp(0);
Src = _LoadMemAutoTSO(GPRClass, OpSize, A, OpSize::i8Bit);
}
Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.ForceLoad = true});
StoreResult(GPRClass, Op, Op->Dest, Src, OpSize::iInvalid);
break;
}
case 0xA2:
case 0xA3: {
case 0xA3:
// Source is GPR
// Dest is memory(literal)
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
// This one is a bit special since the destination is a literal
// So the destination gets stored in Src[1]
if (Op->Src[1].Data.Literal.Size <= 4) {
StoreResult(GPRClass, Op, Op->Src[1], Src, OpSize::iInvalid);
} else {
const auto OpSize = OpSizeFromSrc(Op);
auto A = GenMemSrcFromOp(1);
_StoreMemAutoTSO(GPRClass, OpSize, A, Src, OpSize::i8Bit);
}
StoreResult(GPRClass, Op, Op->Src[1], Src, OpSize::iInvalid);
break;
}
}
}
void OpDispatchBuilder::CPUIDOp(OpcodeArgs) {
@@ -2424,7 +2400,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
} else {
// Can only be an immediate
// Masked by operand size
Src = ARef(Op->Src[SrcIndex].Literal() & Mask);
Src = ARef(Op->Src[SrcIndex].Data.Literal.Value & Mask);
}
if (Op->Dest.IsGPR()) {
@@ -2925,7 +2901,7 @@ void OpDispatchBuilder::AASOp(OpcodeArgs) {
void OpDispatchBuilder::AAMOp(OpcodeArgs) {
auto AL = LoadGPRRegister(X86State::REG_RAX, OpSize::i8Bit);
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF);
auto Imm8 = Constant(Op->Src[0].Data.Literal.Value & 0xFF);
Ref Quotient = _AllocateGPR(true);
Ref Remainder = _AllocateGPR(true);
_UDiv(OpSize::i64Bit, AL, Invalid(), Imm8, Quotient, Remainder);
@@ -2940,7 +2916,7 @@ void OpDispatchBuilder::AAMOp(OpcodeArgs) {
void OpDispatchBuilder::AADOp(OpcodeArgs) {
auto A = LoadGPRRegister(X86State::REG_RAX);
auto AH = _Lshr(OpSize::i32Bit, A, Constant(8));
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF);
auto Imm8 = Constant(Op->Src[0].Data.Literal.Value & 0xFF);
auto NewAL = Add(OpSize::i64Bit, A, _Mul(OpSize::i64Bit, AH, Imm8));
auto Result = _And(OpSize::i64Bit, NewAL, Constant(0xFF));
StoreGPRRegister(X86State::REG_RAX, Result, OpSize::i16Bit);
@@ -3064,19 +3040,19 @@ void OpDispatchBuilder::SMSWOp(OpcodeArgs) {
IR::OpSize DstSize {OpSize::iInvalid};
Ref Const = Constant((1U << 31) | ///< PG - Paging
(0U << 30) | ///< CD - Cache Disable
(0U << 29) | ///< NW - Not Writethrough (Legacy, now ignored)
///< [28:19] - Reserved
(1U << 18) | ///< AM - Alignment Mask
///< 17 - Reserved
(1U << 16) | ///< WP - Write Protect
///< [15:6] - Reserved
(1U << 5) | ///< NE - Numeric Error
(1U << 4) | ///< ET - Extension Type (Legacy, now reserved and 1)
(0U << 3) | ///< TS - Task Switched
(0U << 2) | ///< EM - Emulation
(1U << 1) | ///< MP - Monitor Coprocessor
(1U << 0)); ///< PE - Protection Enabled
(0U << 30) | ///< CD - Cache Disable
(0U << 29) | ///< NW - Not Writethrough (Legacy, now ignored)
///< [28:19] - Reserved
(1U << 18) | ///< AM - Alignment Mask
///< 17 - Reserved
(1U << 16) | ///< WP - Write Protect
///< [15:6] - Reserved
(1U << 5) | ///< NE - Numeric Error
(1U << 4) | ///< ET - Extension Type (Legacy, now reserved and 1)
(0U << 3) | ///< TS - Task Switched
(0U << 2) | ///< EM - Emulation
(1U << 1) | ///< MP - Monitor Coprocessor
(1U << 0)); ///< PE - Protection Enabled
if (Is64BitMode) {
DstSize = X86Tables::DecodeFlags::GetOpAddr(Op->Flags, 0) == X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST ? OpSize::i16Bit :
@@ -3941,8 +3917,7 @@ void OpDispatchBuilder::CreateJumpBlocks(const fextl::vector<FEXCore::Frontend::
}
}
void OpDispatchBuilder::BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks,
uint32_t NumInstructions, bool _Is64BitMode, bool MonoBackpatcherBlock) {
void OpDispatchBuilder::BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions, bool _Is64BitMode, bool MonoBackpatcherBlock) {
Entry = RIP;
Is64BitMode = _Is64BitMode;
LOGMAN_THROW_A_FMT(Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
@@ -4019,8 +3994,7 @@ Ref OpDispatchBuilder::GetSegment(uint32_t Flags, uint32_t DefaultPrefix, bool O
// With the segment register optimization we store the GDT bases directly in the segment register to remove indexed loads
Ref SegmentResult {};
switch (Prefix) {
[[likely]] case FEXCore::X86Tables::DecodeFlags::FLAG_NO_PREFIX:
return nullptr;
[[likely]] case FEXCore::X86Tables::DecodeFlags::FLAG_NO_PREFIX: return nullptr;
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_cached));
break;
@@ -4256,10 +4230,6 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
if (gpr >= FEXCore::X86State::REG_MM_0) {
LOGMAN_THROW_A_FMT(OpSize == OpSize::i64Bit, "full");
if (MMXState != MMXState_MMX) {
ChgStateX87_MMX();
}
A.Base = LoadContext(OpSize::i64Bit, MM0Index + gpr - FEXCore::X86State::REG_MM_0);
} else if (gpr >= FEXCore::X86State::REG_XMM_0) {
const auto gprIndex = gpr - X86State::REG_XMM_0;
@@ -4471,20 +4441,6 @@ void OpDispatchBuilder::MOVGPROp(OpcodeArgs, uint32_t SrcIndex) {
StoreResult(GPRClass, Op, Src, OpSize::i8Bit);
}
void OpDispatchBuilder::MOVGPRImmediate(OpcodeArgs) {
Ref Src {};
if (Op->Src[0].Data.Literal.Size <= 4) {
Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit, .AllowUpperGarbage = true});
} else {
// 8-byte literal is special cased.
const uint64_t Lower = Op->Src[0].Literal();
const uint64_t Upper = Op->Src[1].Literal();
const uint64_t Combined = (Upper << 32) | Lower;
Src = _Constant(Combined);
}
StoreResult(GPRClass, Op, Src, OpSize::i8Bit);
}
void OpDispatchBuilder::MOVGPRNTOp(OpcodeArgs) {
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit});
StoreResult(GPRClass, Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM);
@@ -4606,7 +4562,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
switch (Op->OP) {
case 0xCD: { // INT imm8
uint8_t Literal = Op->Src[0].Literal();
uint8_t Literal = Op->Src[0].Data.Literal.Value;
#ifndef _WIN32
constexpr uint8_t SYSCALL_LITERAL = 0x80;
@@ -4747,18 +4703,18 @@ void OpDispatchBuilder::MOVBEOp(OpcodeArgs) {
const auto SrcSize = OpSizeFromSrc(Op);
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit});
Src = _Rev(std::max(OpSize::i32Bit, SrcSize), Src);
if (DestIsMem(Op) || SrcSize != OpSize::i16Bit) {
Src = _Rev(SrcSize, Src);
StoreResult(GPRClass, Op, Op->Dest, Src, OpSize::iInvalid);
} else {
Src = _Rev(std::max(OpSize::i32Bit, SrcSize), Src);
if (SrcSize == OpSize::i16Bit) {
// 16-bit does an insert.
// Rev of 16-bit value as 32-bit replaces the result in the upper 16-bits of the result.
// bfxil the 16-bit result in to the GPR.
Ref Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags);
auto Result = _Bfxil(GPRSize, 16, 16, Dest, Src);
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
} else {
// 32-bit does regular zext
StoreResult(GPRClass, Op, Op->Dest, Src, OpSize::iInvalid);
}
}
@@ -4973,7 +4929,7 @@ void OpDispatchBuilder::InvalidOp(OpcodeArgs) {
void OpDispatchBuilder::NoExecOp(OpcodeArgs) {
BreakOp(Op, FEXCore::IR::BreakDefinition {
.ErrorRegister = X86State::X86_PF_PROT | X86State::X86_PF_USER | X86State::X86_PF_INSTR,
.ErrorRegister = 0,
.Signal = Core::FAULT_SIGSEGV,
.TrapNumber = X86State::X86_TRAPNO_PF,
.si_code = 2, // SEGV_ACCERR
@@ -209,17 +209,10 @@ public:
}
static bool CanHaveSideEffects(const FEXCore::X86Tables::X86InstInfo* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
if (TableInfo) {
if (TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
// If it is marked as having memory access then always say it has a side-effect.
// Not always true but better to be safe.
return true;
}
if (TableInfo->Flags & (X86Tables::InstFlags::FLAGS_SETS_RIP | X86Tables::InstFlags::FLAGS_BLOCK_END)) {
// Cooperative suspend interrupts can be triggered at any back-edge, the RIP must be reconstructed correctly in such cases
return true;
}
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
// If it is marked as having memory access then always say it has a side-effect.
// Not always true but better to be safe.
return true;
}
auto CanHaveSideEffects = false;
@@ -300,8 +293,7 @@ public:
return ShouldDump;
}
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions,
bool Is64BitMode, bool MonoBackpatcherBlock);
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions, bool Is64BitMode, bool MonoBackpatcherBlock);
void Finalize();
// Dispatch builder functions
@@ -318,7 +310,6 @@ public:
void UnhandledOp(OpcodeArgs);
void MOVGPROp(OpcodeArgs, uint32_t SrcIndex);
void MOVGPRImmediate(OpcodeArgs);
void MOVGPRNTOp(OpcodeArgs);
void MOVVectorAlignedOp(OpcodeArgs);
void MOVVectorUnalignedOp(OpcodeArgs);
@@ -1854,10 +1845,10 @@ private:
static const int PFIndex = 16;
static const int AFIndex = 17;
/* Gap 18..19 */
/* Note this range is only valid if MMXState = MMXState_MMX */
static const int MM0Index = 20;
static const int MM7Index = 27;
/* Gap 28..30 */
static const int AbridgedFTWIndex = 28;
/* Gap 29..30 */
static const int DFIndex = 31;
static const int FPR0Index = 32;
static const int FPR15Index = 47;
@@ -1869,6 +1860,7 @@ private:
switch (Index) {
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
default: return ~0U;
}
}
@@ -1930,7 +1922,7 @@ private:
if (!(RegCache.Cached & Bit)) {
if (Index == DFIndex) {
RegCache.Value[Index] = _LoadDF();
} else if ((Index >= MM0Index && Index <= MM7Index) || Index >= AVXHigh0Index) {
} else if ((Index >= MM0Index && Index <= AbridgedFTWIndex) || Index >= AVXHigh0Index) {
RegCache.Value[Index] = _LoadContext(Size, RegClass, Offset);
// We may have done a partial load, this requires special handling.
@@ -2364,8 +2356,8 @@ private:
void ChgStateX87_MMX() override {
LOGMAN_THROW_A_FMT(MMXState == MMXState_X87, "Expected state to be x87");
_StackForceSlow();
SetX87Top(Constant(0)); // top reset to zero
_StoreContext(OpSize::i8Bit, GPRClass, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
SetX87Top(Constant(0)); // top reset to zero
StoreContext(AbridgedFTWIndex, Constant(0xFFFFUL)); // all valid
MMXState = MMXState_MMX;
}
@@ -53,7 +53,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
{0xAA, 2, &OpDispatchBuilder::STOSOp},
{0xAC, 2, &OpDispatchBuilder::LODSOp},
{0xAE, 2, &OpDispatchBuilder::SCASOp},
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPRImmediate>},
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
{0xC2, 2, &OpDispatchBuilder::RETOp},
{0xC8, 1, &OpDispatchBuilder::EnterOp},
{0xC9, 1, &OpDispatchBuilder::LEAVEOp},
@@ -2557,8 +2557,6 @@ void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
}
void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
_SyncStackToSlow();
// Saves 512bytes to the memory location provided
// Header changes depending on if REX.W is set or not
if (Op->Flags & X86Tables::DecodeFlags::FLAG_REX_WIDENING) {
@@ -2582,8 +2580,7 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
{
// Abridged FTW
auto FTW = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
_StoreMem(GPRClass, OpSize::i8Bit, FTW, MemBase, Constant(4), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
_StoreMem(GPRClass, OpSize::i8Bit, LoadContext(AbridgedFTWIndex), MemBase, Constant(4), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
}
// BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f |
@@ -2630,19 +2627,9 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
// MXCSR_MASK: Mask for writes to the MXCSR register
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
// This is implementation dependent
//
// x87 registers are stored rotated depending on the current TOP.
Ref Top = GetX87Top();
auto SevenConst = Constant(7);
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
if (ReducedPrecisionMode) {
data = _F80CVTTo(data, OpSize::i64Bit);
}
_StoreMem(FPRClass, OpSize::i128Bit, data, MemBase, Constant(16 * i + 32), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
RefPair MMRegs = LoadContextPair(OpSize::i128Bit, MM0Index + i);
_StoreMemPair(FPRClass, OpSize::i128Bit, MMRegs.Low, MMRegs.High, MemBase, i * 16 + 32);
}
}
@@ -2753,8 +2740,6 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
}
void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
_StackForceSlow();
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, MemBase, OpSize::i16Bit);
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
@@ -2765,14 +2750,14 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
{
// Abridged FTW
auto NewFTW = _LoadMem(GPRClass, OpSize::i8Bit, MemBase, Constant(4), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
_StoreContext(OpSize::i8Bit, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, _LoadMem(GPRClass, OpSize::i8Bit, MemBase, Constant(4), OpSize::i8Bit, MEM_OFFSET_SXTX, 1));
}
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
auto MMRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 32);
_StoreContext(OpSize::i128Bit, FPRClass, MMRegs.Low, MMBaseOffset() + i * 16);
_StoreContext(OpSize::i128Bit, FPRClass, MMRegs.High, MMBaseOffset() + (i + 1) * 16);
StoreContext(MM0Index + i, MMRegs.Low);
StoreContext(MM0Index + i + 1, MMRegs.High);
}
}
@@ -2818,7 +2803,7 @@ void OpDispatchBuilder::DefaultX87State(OpcodeArgs) {
// all of the ST0-7/MM0-7 registers to zero.
Ref ZeroVector = LoadZeroVector(OpSize::i64Bit);
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
_StoreContext(OpSize::i128Bit, FPRClass, ZeroVector, MMBaseOffset() + i * 16);
StoreContext(MM0Index + i, ZeroVector);
}
}
@@ -32,8 +32,6 @@ Ref OpDispatchBuilder::GetX87Top() {
}
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
_StackForceSlow(); // Invalidate x87 FTW register cache
// For the output, we want a 1-bit for each pair not equal to 11 (Empty).
static_assert(static_cast<uint8_t>(FPState::X87Tag::Empty) == 0b11);
@@ -52,7 +50,7 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4);
// ...and that's it. StoreContext implicitly does the final masking.
_StoreContext(OpSize::i8Bit, GPRClass, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, FTW);
}
void OpDispatchBuilder::SetX87Top(Ref Value) {
@@ -340,7 +338,7 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() {
// bytes, we use the well-known bit twiddling algorithm:
//
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
Ref X = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref X = LoadContext(AbridgedFTWIndex);
X = _Orlshl(OpSize::i32Bit, X, X, 4);
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f));
X = _Orlshl(OpSize::i32Bit, X, X, 2);
@@ -591,7 +589,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
// Lower 64bits [63:0]
// upper 16 bits [79:64]
Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Ref RegHigh = _LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Ref RegHigh =
_LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
if (ReducedPrecisionMode) {
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
@@ -775,8 +774,6 @@ void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
}
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
_SyncStackToSlow(); // Invalidate x87 register caches
auto Zero = Constant(0);
if (ReducedPrecisionMode) {
@@ -790,7 +787,7 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
// Set top to zero
SetX87Top(Zero);
// Tags all get marked as invalid
_StoreContext(OpSize::i8Bit, GPRClass, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, Zero);
// Reinits the simulated stack
_InitStack();
@@ -60,7 +60,7 @@ void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
// Float load op with memory operand
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
// Convert to 64bit float
Ref ConvertedData = Data;
if (Width == OpSize::i32Bit) {
@@ -73,7 +73,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
// Read from memory
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i128Bit, Op->Flags);
Ref ConvertedData = _F80BCDLoad(Data);
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
@@ -42,9 +42,8 @@ constexpr uint32_t FLAG_DS_PREFIX = (0b100 << 11);
constexpr uint32_t FLAG_FS_PREFIX = (0b101 << 11);
constexpr uint32_t FLAG_GS_PREFIX = (0b110 << 11);
constexpr uint32_t FLAG_SEGMENTS = (0b111 << 11);
constexpr uint32_t FLAG_FORCE_TSO = (1 << 14);
constexpr uint32_t FLAG_DECODED_MODRM = (1 << 15);
constexpr uint32_t FLAG_DECODED_SIB = (1 << 16);
// Bits 14, 15, 16 - Unused
constexpr uint32_t FLAG_REP_PREFIX = (1 << 17);
constexpr uint32_t FLAG_REPNE_PREFIX = (1 << 18);
// Size flags
@@ -144,9 +143,6 @@ struct DecodedOperand {
}
uint64_t Literal() const {
LOGMAN_THROW_A_FMT(IsLiteral(), "Precondition: must be a literal");
if (Data.Literal.SignExtend) {
return static_cast<int64_t>(static_cast<int32_t>(Data.Literal.Value));
}
return Data.Literal.Value;
}
@@ -170,9 +166,8 @@ struct DecodedOperand {
} RIPLiteral;
struct LiteralType {
uint32_t Value;
uint8_t Size : 7 ;
bool SignExtend : 1;
uint64_t Value;
uint8_t Size;
auto operator<=>(const LiteralType&) const = default;
} Literal;
@@ -198,12 +193,16 @@ struct DecodedInst {
X86InstInfo const* TableInfo;
uint32_t Flags;
uint16_t OP;
uint8_t OPRaw;
uint16_t OP;
uint8_t ModRM;
uint8_t SIB;
uint8_t InstSize;
uint8_t LastEscapePrefix;
bool DecodedModRM;
bool DecodedSIB;
bool ForceTSO;
};
union ModRMDecoded {
+7 -7
View File
@@ -42,19 +42,19 @@ void __attribute__((noinline)) __jit_debug_register_code() {
namespace FEXCore {
void GDBJITRegister(FEXCore::ExecutableFileInfo& Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
FEXCore::Core::DebugData& DebugData) {
auto map = Entry.SourcecodeMap.get();
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
FEXCore::Core::DebugData* DebugData) {
auto map = Entry->SourcecodeMap.get();
if (map) {
auto FileOffset = GuestRIP - VAFileStart;
auto Sym = map->FindSymbolMapping(FileOffset);
auto SymName = HLE::SourcecodeSymbolMapping::SymName(Sym, Entry.Filename, HostEntry, FileOffset);
auto SymName = HLE::SourcecodeSymbolMapping::SymName(Sym, Entry->Filename, HostEntry, FileOffset);
fextl::vector<gdb_line_mapping> Lines;
for (const auto& GuestOpcode : DebugData.GuestOpcodes) {
for (const auto& GuestOpcode : DebugData->GuestOpcodes) {
auto Line = map->FindLineMapping(GuestRIP + GuestOpcode.GuestEntryOffset - VAFileStart);
if (Line) {
Lines.push_back({Line->LineNumber, HostEntry + GuestOpcode.HostEntryOffset});
@@ -80,7 +80,7 @@ void GDBJITRegister(FEXCore::ExecutableFileInfo& Entry, uintptr_t VAFileStart, u
for (int i = 0; i < info->nblocks; i++) {
strncpy(blocks[i].name, SymName.c_str(), 511);
blocks[i].start = HostEntry;
blocks[i].end = HostEntry + DebugData.HostCodeSize;
blocks[i].end = HostEntry + DebugData->HostCodeSize;
}
info->nlines = Lines.size();
@@ -113,7 +113,7 @@ void GDBJITRegister(FEXCore::ExecutableFileInfo& Entry, uintptr_t VAFileStart, u
} // namespace FEXCore
#else
namespace FEXCore {
void GDBJITRegister(FEXCore::ExecutableFileInfo&, uintptr_t, uint64_t, uintptr_t, FEXCore::Core::DebugData&) {
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry*, uintptr_t, uint64_t, uintptr_t, FEXCore::Core::DebugData*) {
ERROR_AND_DIE_FMT("GDBSymbols support not compiled in");
}
} // namespace FEXCore
+5 -4
View File
@@ -1,8 +1,9 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/Core/CodeCache.h>
#include <Interface/Core/JIT/DebugData.h>
#include <Interface/IR/AOTIR.h>
namespace FEXCore {
void GDBJITRegister(FEXCore::ExecutableFileInfo&, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry, FEXCore::Core::DebugData&);
}
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
FEXCore::Core::DebugData* DebugData);
}
+60
View File
@@ -0,0 +1,60 @@
// SPDX-License-Identifier: MIT
#include "FEXHeaderUtils/Filesystem.h"
#include "Interface/Context/Context.h"
#include "Interface/IR/AOTIR.h"
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/fextl/fmt.h>
#include <Interface/Core/LookupCache.h>
#include <Interface/GDBJIT/GDBJIT.h>
#include <xxhash.h>
namespace FEXCore::IR {
bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr,
uint64_t Length, FEXCore::Core::DebugData* DebugData) {
// Both generated ir and LibraryJITName need a named region lookup
if (CTX->Config.LibraryJITNaming() || CTX->Config.GDBSymbols()) {
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
if (AOTIRCacheEntry.Entry) {
if (DebugData && CTX->Config.LibraryJITNaming()) {
CTX->Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, AOTIRCacheEntry.Entry->Filename);
}
if (CTX->Config.GDBSymbols()) {
GDBJITRegister(AOTIRCacheEntry.Entry, AOTIRCacheEntry.VAFileStart, GuestRIP, (uintptr_t)CodePtr, DebugData);
}
}
}
return false;
}
AOTIRCacheEntry* AOTIRCaptureCache::LoadAOTIRCacheEntry(const fextl::string& filename) {
fextl::string base_filename = FHU::Filesystem::GetFilename(filename);
if (!base_filename.empty()) {
auto filename_hash = XXH3_64bits(filename.c_str(), filename.size());
auto fileid = fextl::fmt::format("{}-{}-{}{}{}", base_filename, filename_hash,
(CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) ? 'S' : 's',
CTX->Config.TSOEnabled ? 'T' : 't', CTX->Config.ABILocalFlags ? 'L' : 'l');
std::unique_lock lk(AOTIRCacheLock);
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry {.FileId = fileid, .Filename = filename}});
auto Entry = &(Inserted.first->second);
return Entry;
}
return nullptr;
}
} // namespace FEXCore::IR
+72
View File
@@ -0,0 +1,72 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/HLE/SourcecodeResolver.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <cstdint>
#include <shared_mutex>
namespace FEXCore::CPU {
union Relocation;
} // namespace FEXCore::CPU
namespace FEXCore::Core {
struct InternalThreadState;
struct DebugDataSubblock {
uint32_t HostCodeOffset;
uint32_t HostCodeSize;
};
struct DebugDataGuestOpcode {
uint64_t GuestEntryOffset;
ptrdiff_t HostEntryOffset;
};
/**
* @brief Contains debug data for a block of code for later debugger analysis
*
* Needs to remain around for as long as the code could be executed at least
*/
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
fextl::vector<DebugDataSubblock> Subblocks;
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
};
} // namespace FEXCore::Core
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::IR {
struct AOTIRCacheEntry {
fextl::unique_ptr<FEXCore::HLE::SourcecodeMap> SourcecodeMap;
fextl::string FileId;
fextl::string Filename;
};
class AOTIRCaptureCache final {
public:
AOTIRCaptureCache(FEXCore::Context::ContextImpl* ctx)
: CTX {ctx} {}
bool PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr, uint64_t Length,
FEXCore::Core::DebugData* DebugData);
AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& filename);
private:
FEXCore::Context::ContextImpl* CTX;
std::shared_mutex AOTIRCacheLock;
fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCacheEntry> AOTIRCache;
};
} // namespace FEXCore::IR
+4 -9
View File
@@ -1,9 +1,7 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/ThreadPoolAllocator.h>
#include <FEXCore/IR/IR.h>
@@ -11,11 +9,6 @@
#include <FEXCore/fextl/sstream.h>
#include <array>
#include <cstddef>
#include <cstdint>
#include <functional>
#include <iterator>
#include <type_traits>
namespace FEXCore::IR {
@@ -242,6 +235,8 @@ static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3);
* The second region is contiguous but they don't have any relationship with one another directly
*/
class OrderedNode final {
friend class NodeWrapperIterator;
friend class OrderedList;
public:
// These three values are laid out very specifically to make it fast to access the NodeWrappers specifically
OrderedNodeHeader Header;
@@ -526,8 +521,8 @@ class NodeIterator;
class NodeIterator {
public:
struct value_type final {
OrderedNode* Node;
IROp_Header* Header;
OrderedNode *Node;
IROp_Header *Header;
};
using size_type = std::size_t;
using difference_type = std::ptrdiff_t;
+4 -13
View File
@@ -505,17 +505,6 @@
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContextIndexed to XMM\""
]
},
"GPR = FormContextAddress OpSize:#Size, GPR:$Index, u32:$Stride": {
"Desc": ["Forms an address into the context structure indexed by SSA value",
"Dest = Ctx + Index * Stride",
"This allows backends to compute the address once and reuse it for multiple memory operations",
"Stride must be a power of 2"
],
"DestSize": "Size",
"EmitValidation": [
"#Size == IR::OpSize::i64Bit"
]
},
"SpillRegister SSA:$Value, u32:$Slot, RegisterClass:$Class": {
"HasSideEffects": true,
@@ -604,7 +593,8 @@
"Desc": ["Does a x86 TSO compatible load from memory. Offset must be Invalid()."
],
"Inline": ["", "Memtso"],
"DestSize": "Size"
"DestSize": "Size",
"DynamicDispatch": true
},
"StoreMemTSO RegisterClass:$Class, OpSize:#Size, SSA:$Value, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
@@ -612,7 +602,8 @@
],
"Inline": ["Zero", "", "Memtso"],
"HasSideEffects": true,
"DestSize": "Size"
"DestSize": "Size",
"DynamicDispatch": true
},
"FPR = VLoadVectorMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Mask, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
+1 -2
View File
@@ -315,8 +315,7 @@ void Dump(fextl::stringstream* out, const IRListView* IR) {
++CurrentIndent;
AddIndent();
*out << fextl::fmt::format("(%0) IRHeader %{}, #{:#x}, #{}, #{}\n", HeaderOp->Blocks.ID(), HeaderOp->OriginalRIP, HeaderOp->BlockCount,
HeaderOp->NumHostInstructions);
*out << fextl::fmt::format("(%0) IRHeader %{}, #{:#x}, #{}, #{}\n", HeaderOp->Blocks.ID(), HeaderOp->OriginalRIP, HeaderOp->BlockCount, HeaderOp->NumHostInstructions);
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
{
+5
View File
@@ -16,8 +16,13 @@
#include <string.h>
namespace FEXCore::IR {
class Pass;
class PassManager;
class IREmitter {
friend class FEXCore::IR::Pass;
friend class FEXCore::IR::PassManager;
public:
IREmitter(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, bool SupportsTSOImm9)
: DualListData {ThreadAllocator, 8 * 1024 * 1024}
@@ -255,6 +255,18 @@ private:
}
}
// Helper to check if a Ref is a Zero constant
bool IsZero(Ref Node) {
auto Header = IR->GetOp<IR::IROp_Header>(Node);
if (Header->Op != OP_CONSTANT) {
return false;
}
auto Const = Header->C<IROp_Constant>();
return Const->Constant == 0;
}
// Handles a Unary operation.
// Takes the op we are handling, the Node for the reduced precision case and the node for the normal case.
// Depending on the type of Op64, we might need to pass a couple of extra constant arguments, this happens
@@ -267,7 +279,7 @@ private:
// Top Management Helpers
/// Set the valid tag for Value as valid (if Valid is true), or invalid (if Valid is false).
void SetX87ValidTag(uint8_t Offset, bool Valid);
void SetX87ValidTag(Ref Value, bool Valid);
// Generates slow code to load/store a value from an offset from the top of the stack
Ref LoadStackValueAtOffset_Slow(uint8_t Offset = 0);
void StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset = 0, bool SetValid = true);
@@ -282,12 +294,11 @@ private:
void MigrateToSlowPathIf(bool ShouldMigrate);
// Top Cache Management
Ref GetTopWithCache_Slow();
Ref GetOffsetTopWithCache_Slow(uint8_t Offset, bool Reverse = false);
Ref GetOffsetTopAddressWithCache_Slow(uint8_t Offset);
Ref GetOffsetTopWithCache_Slow(uint8_t Offset);
void SetTopWithCache_Slow(Ref Value);
Ref GetX87ValidTag_Slow(uint8_t Offset);
// Resets fields to initial values
void Reset();
void Reset(bool AlsoSlowPath = true);
struct StackMemberInfo {
StackMemberInfo() {}
@@ -319,7 +330,7 @@ private:
FixedSizeStack<StackMemberInfo> StackData;
void InvalidateCaches();
void InvalidateCachedRegs();
void InvalidateTopOffsetCache();
// Path Migration helper management
std::optional<StackMemberInfo> MigrateToSlowPath_IfInvalid(uint8_t Offset = 0);
@@ -334,19 +345,7 @@ private:
// Cached value for Top
// If slowpath is false, then TopCache is nullptr.
bool FlushTopPending = false;
std::array<bool, 8> FlushValuesPending {};
bool FlushValidPending = false;
void FlushCachedRegs();
Ref GetFTW();
Ref FTWCached {};
std::array<Ref, 8> TopOffsetCache {};
std::array<Ref, 8> TopOffsetAddressCache {};
std::array<Ref, 8> TopValueCache {};
std::array<StackSlot, 8> TopValidCache {};
// Are we on the slow path?
// Once we enter the slow path, we never come out.
// This just simplifies the code atm. If there's a need to return to the fast path in the future
@@ -360,21 +359,18 @@ private:
};
inline void X87StackOptimization::InvalidateCaches() {
InvalidateCachedRegs();
InvalidateTopOffsetCache();
ConstantPool.fill(nullptr);
}
inline void X87StackOptimization::InvalidateCachedRegs() {
FlushCachedRegs();
FTWCached = {};
inline void X87StackOptimization::InvalidateTopOffsetCache() {
TopOffsetCache.fill(nullptr);
TopOffsetAddressCache.fill(nullptr);
TopValueCache.fill(nullptr);
TopValidCache.fill(StackSlot::UNUSED);
}
inline void X87StackOptimization::Reset() {
SlowPath = false;
inline void X87StackOptimization::Reset(bool AlsoSlowPath) {
if (AlsoSlowPath) {
SlowPath = false;
}
StackData.clear();
InvalidateCaches();
}
@@ -394,7 +390,7 @@ inline Ref X87StackOptimization::GetConstant(ssize_t Offset) {
inline void X87StackOptimization::MigrateToSlowPathIf(bool ShouldMigrate) {
if (ShouldMigrate && !SlowPath) {
SynchronizeStackValues();
StackData.clear();
Reset(false); // Reset everything but no need to change slowpath
SlowPath = true;
}
}
@@ -407,13 +403,7 @@ inline Ref X87StackOptimization::GetTopWithCache_Slow() {
return TopOffsetCache[0];
}
inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset, bool Reverse) {
if (Reverse) {
Offset = 8 - Offset;
}
Offset &= 7;
inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset) {
if (TopOffsetCache[Offset]) {
return TopOffsetCache[Offset];
}
@@ -428,60 +418,38 @@ inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset, bool
return OffsetTop;
}
inline Ref X87StackOptimization::GetOffsetTopAddressWithCache_Slow(uint8_t Offset) {
if (TopOffsetAddressCache[Offset]) {
return TopOffsetAddressCache[Offset];
}
Ref OffsetRef = GetOffsetTopWithCache_Slow(Offset);
TopOffsetAddressCache[Offset] = IREmit->_FormContextAddress(OpSize::i64Bit, OffsetRef, 16);
return TopOffsetAddressCache[Offset];
}
inline void X87StackOptimization::SetTopWithCache_Slow(Ref Value) {
InvalidateCachedRegs();
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
InvalidateTopOffsetCache();
TopOffsetCache[0] = Value;
FlushTopPending = true;
}
inline Ref X87StackOptimization::GetFTW() {
if (!FTWCached) {
FTWCached = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
}
return FTWCached;
}
inline void X87StackOptimization::SetX87ValidTag(uint8_t Offset, bool Valid) {
TopValidCache[Offset] = Valid ? StackSlot::VALID : StackSlot::INVALID;
FlushValidPending = true;
inline void X87StackOptimization::SetX87ValidTag(Ref Value, bool Valid) {
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), Value);
Ref NewAbridgedFTW = Valid ? IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RegMask) : IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RegMask);
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
}
inline Ref X87StackOptimization::GetX87ValidTag_Slow(uint8_t Offset) {
switch (TopValidCache[Offset]) {
case StackSlot::UNUSED:
return IREmit->_And(OpSize::i32Bit, IREmit->_Lshr(OpSize::i32Bit, GetFTW(), GetOffsetTopWithCache_Slow(Offset)), GetConstant(1));
case StackSlot::INVALID: return GetConstant(0);
case StackSlot::VALID: return GetConstant(1);
}
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
return IREmit->_And(OpSize::i32Bit, IREmit->_Lshr(OpSize::i32Bit, AbridgedFTW, GetOffsetTopWithCache_Slow(Offset)), GetConstant(1));
}
inline Ref X87StackOptimization::LoadStackValueAtOffset_Slow(uint8_t Offset) {
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(Offset);
auto Size = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
if (!TopValueCache[Offset]) {
TopValueCache[Offset] = IREmit->_LoadMem(FPRClass, Size, TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MEM_OFFSET_SXTX, 1);
}
return TopValueCache[Offset];
return IREmit->_LoadContextIndexed(GetOffsetTopWithCache_Slow(Offset), ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
MMBaseOffset(), 16, FPRClass);
}
inline void X87StackOptimization::StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset, bool SetValid) {
TopValueCache[Offset] = Value;
FlushValuesPending[Offset] = true;
OrderedNode* TopOffset = GetOffsetTopWithCache_Slow(Offset);
// store
IREmit->_StoreContextIndexed(Value, TopOffset, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit, MMBaseOffset(), 16, FPRClass);
// mark it valid
// In some cases we might already know it has been previously set as valid so we don't need to do it again
if (SetValid) {
SetX87ValidTag(Offset, true);
SetX87ValidTag(TopOffset, true);
}
}
@@ -573,100 +541,25 @@ void X87StackOptimization::HandleBinopStack(IROps Op64, bool VFOp64, IROps Op80,
inline void X87StackOptimization::UpdateTopForPop_Slow() {
// Pop the top of the x87 stack
GetOffsetTopWithCache_Slow(1);
std::rotate(TopOffsetCache.begin(), std::next(TopOffsetCache.begin()), TopOffsetCache.end());
std::rotate(TopOffsetAddressCache.begin(), std::next(TopOffsetAddressCache.begin()), TopOffsetAddressCache.end());
std::rotate(TopValueCache.begin(), std::next(TopValueCache.begin()), TopValueCache.end());
std::rotate(FlushValuesPending.begin(), std::next(FlushValuesPending.begin()), FlushValuesPending.end());
std::rotate(TopValidCache.begin(), std::next(TopValidCache.begin()), TopValidCache.end());
FlushTopPending = true;
auto* TopOffset = GetTopWithCache_Slow();
TopOffset = IREmit->Add(OpSize::i32Bit, TopOffset, 1);
TopOffset = IREmit->_And(OpSize::i32Bit, TopOffset, GetConstant(7));
SetTopWithCache_Slow(TopOffset);
}
inline void X87StackOptimization::UpdateTopForPush_Slow() {
// Pop the top of the x87 stack
GetOffsetTopWithCache_Slow(1, true);
std::rotate(TopOffsetCache.begin(), std::prev(TopOffsetCache.end()), TopOffsetCache.end());
std::rotate(TopOffsetAddressCache.begin(), std::prev(TopOffsetAddressCache.end()), TopOffsetAddressCache.end());
std::rotate(TopValueCache.begin(), std::prev(TopValueCache.end()), TopValueCache.end());
std::rotate(FlushValuesPending.begin(), std::prev(FlushValuesPending.end()), FlushValuesPending.end());
std::rotate(TopValidCache.begin(), std::prev(TopValidCache.end()), TopValidCache.end());
FlushTopPending = true;
}
void X87StackOptimization::FlushCachedRegs() {
if (FlushTopPending) {
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, TopOffsetCache[0], offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
FlushTopPending = false;
}
auto Size = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (size_t i = 0; i < FlushValuesPending.size(); i++) {
if (FlushValuesPending[i]) {
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(i);
IREmit->_StoreMem(FPRClass, Size, TopValueCache[i], TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MEM_OFFSET_SXTX, 1);
// store
FlushValuesPending[i] = false;
}
}
if (FlushValidPending) {
uint8_t ValidMask = 0;
uint8_t InvalidMask = 0;
for (auto It = TopValidCache.rbegin(); It != TopValidCache.rend(); It++) {
ValidMask <<= 1;
InvalidMask <<= 1;
if (*It == StackSlot::VALID) {
ValidMask |= 1;
} else if (*It == StackSlot::INVALID) {
InvalidMask |= 1;
}
}
if (ValidMask || InvalidMask) {
Ref NewFTW = [&]() {
if (ValidMask == 0xff || InvalidMask == 0xff) {
// If InvalidMask == 0xff then ValidMask = 0
return GetConstant(ValidMask);
} else {
Ref NewFTW = GetFTW();
Ref RotAmount {};
if (std::popcount(ValidMask) == 1) {
uint8_t BitIdx = std::countr_zero(ValidMask);
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), GetOffsetTopWithCache_Slow(BitIdx));
NewFTW = IREmit->_Or(OpSize::i32Bit, NewFTW, RegMask);
} else if (ValidMask) {
RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), GetTopWithCache_Slow());
// perform a rotate right on mask by top
NewFTW = IREmit->_Or(OpSize::i32Bit, NewFTW, RotateRight8(ValidMask, RotAmount));
}
if (std::popcount(InvalidMask) == 1) {
uint8_t BitIdx = std::countr_zero(InvalidMask);
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), GetOffsetTopWithCache_Slow(BitIdx));
NewFTW = IREmit->_Andn(OpSize::i32Bit, NewFTW, RegMask);
} else if (InvalidMask) {
if (!RotAmount) {
RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), GetTopWithCache_Slow());
}
NewFTW = IREmit->_Andn(OpSize::i32Bit, NewFTW, RotateRight8(InvalidMask, RotAmount));
}
return NewFTW;
}
}();
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
FTWCached = NewFTW;
}
FlushValidPending = false;
}
auto* TopOffset = GetTopWithCache_Slow();
TopOffset = IREmit->Sub(OpSize::i32Bit, TopOffset, 1);
TopOffset = IREmit->_And(OpSize::i32Bit, TopOffset, GetConstant(7));
SetTopWithCache_Slow(TopOffset);
}
// We synchronize stack values in a few occasions but one of the most important of those,
// is when we move from fast to a slow path and need to make sure that the context is properly
// written.
Ref X87StackOptimization::SynchronizeStackValues() {
if (SlowPath) {
if (SlowPath) { // Nothing to do here.
return GetTopWithCache_Slow();
}
@@ -675,7 +568,8 @@ Ref X87StackOptimization::SynchronizeStackValues() {
const auto TopOffset = StackData.TopOffset;
if (TopOffset != 0) {
Ref NewTop = GetOffsetTopWithCache_Slow(TopOffset, true);
auto* OrigTop = GetTopWithCache_Slow();
Ref NewTop = IREmit->_And(OpSize::i32Bit, IREmit->Sub(OpSize::i32Bit, OrigTop, TopOffset), GetConstant(0x7));
SetTopWithCache_Slow(NewTop);
}
StackData.TopOffset = 0;
@@ -687,22 +581,51 @@ Ref X87StackOptimization::SynchronizeStackValues() {
for (size_t i = 0; i < StackData.size; ++i) {
const auto& [Valid, StackMember] = StackData.top(i);
if (Valid == StackSlot::UNUSED) {
continue;
}
Ref TopIndex = GetOffsetTopWithCache_Slow(i);
if (Valid == StackSlot::VALID) {
StoreStackValueAtOffset_Slow(StackMember.StackDataNode, i, false);
IREmit->_StoreContextIndexed(StackMember.StackDataNode, TopIndex, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
MMBaseOffset(), 16, FPRClass);
}
}
{ // Set valid tags
uint8_t ValidMask = StackData.getValidMask();
uint8_t InvalidMask = StackData.getInvalidMask();
for (auto& Elem : TopValidCache) {
Elem = (ValidMask & 1) ? StackSlot::VALID : ((InvalidMask & 1) ? StackSlot::INVALID : StackSlot::UNUSED);
ValidMask >>= 1;
InvalidMask >>= 1;
uint8_t Mask = StackData.getValidMask();
if (Mask == 0xff) {
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(Mask), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
} else if (Mask != 0) {
if (std::popcount(Mask) == 1) {
uint8_t BitIdx = __builtin_ctz(Mask);
SetX87ValidTag(GetOffsetTopWithCache_Slow(BitIdx), true);
} else {
// perform a rotate right on mask by top
auto* TopValue = GetTopWithCache_Slow();
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref NewAbridgedFTW = IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
}
}
}
{ // Set invalid tags
uint8_t Mask = StackData.getInvalidMask();
if (Mask == 0xff) {
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
} else if (Mask != 0) {
if (std::popcount(Mask)) {
uint8_t BitIdx = __builtin_ctz(Mask);
SetX87ValidTag(GetOffsetTopWithCache_Slow(BitIdx), false);
} else {
// Same rotate right as above but this time on the invalid mask
auto* TopValue = GetTopWithCache_Slow();
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref NewAbridgedFTW = IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
}
}
FlushValidPending = true;
}
return TopValue;
}
@@ -892,7 +815,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
case OP_INITSTACK: {
StackData.clear();
InvalidateCachedRegs();
break;
}
@@ -902,14 +824,18 @@ void X87StackOptimization::Run(IREmitter* Emit) {
if (Offset != 0xff) { // invalidate single offset
if (SlowPath) {
SetX87ValidTag(Offset, false);
auto* TopValue = GetTopWithCache_Slow();
if (Offset != 0) {
auto* Mask = GetConstant(7);
TopValue = IREmit->_And(OpSize::i32Bit, IREmit->Add(OpSize::i32Bit, TopValue, Offset), Mask);
}
SetX87ValidTag(TopValue, false);
} else {
StackData.setTagInvalid(Offset);
}
} else { // invalidate all
if (SlowPath) {
TopValidCache.fill(StackSlot::INVALID);
FlushValidPending = true;
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
} else {
for (size_t i = 0; i < StackData.size; i++) {
StackData.setTagInvalid(i);
@@ -1027,7 +953,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
}
case OP_POPSTACKDESTROY: {
if (SlowPath) {
SetX87ValidTag(0, false);
SetX87ValidTag(GetTopWithCache_Slow(), false);
}
StackPop();
break;
@@ -1126,14 +1052,12 @@ void X87StackOptimization::Run(IREmitter* Emit) {
case OP_SYNCSTACKTOSLOW: {
// This synchronizes stack values but doesn't necessarily moves us off the FastPath!
Ref NewTop = SynchronizeStackValues();
FlushCachedRegs();
IREmit->ReplaceUsesWithAfter(CodeNode, NewTop, CodeNode);
break;
}
case OP_STACKFORCESLOW: {
MigrateToSlowPathIf(true);
InvalidateCachedRegs();
break;
}
@@ -1192,7 +1116,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
LOGMAN_THROW_A_FMT(IsBlockExit(LastIROp->Op), "must be exit");
IREmit->SetWriteCursorBefore(LastCodeNode);
SynchronizeStackValues();
FlushCachedRegs();
}
return;
@@ -165,8 +165,7 @@ private:
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
auto Res = mprotect(reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData, PROT_READ | PROT_WRITE);
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno,
strerror(errno));
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno, strerror(errno));
LiveVMARegion* LiveRange = new (reinterpret_cast<void*>(ReservedRegion->Base)) LiveVMARegion();
@@ -274,8 +273,8 @@ void* OSAllocator_64Bit::Mmap(void* addr, size_t length, int prot, int flags, in
again:
struct RangeResult final {
LiveVMARegion* RegionInsertedInto;
void* Ptr;
LiveVMARegion *RegionInsertedInto;
void *Ptr;
};
auto CheckIfRangeFits = [&AllocatedOffset](LiveVMARegion* Region, uint64_t length, int prot, int flags, int fd, off_t offset,
-119
View File
@@ -1,119 +0,0 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/Utils/LongJump.h>
namespace FEXCore::LongJump {
#if defined(_M_ARM_64)
[[nodiscard]]
FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
__asm volatile(R"(
// x0 contains the jumpbuffer
stp x19, x20, [x0, #( 0 * 8)];
stp x21, x22, [x0, #( 2 * 8)];
stp x23, x24, [x0, #( 4 * 8)];
stp x25, x26, [x0, #( 6 * 8)];
stp x27, x28, [x0, #( 8 * 8)];
stp x29, x30, [x0, #(10 * 8)];
// FPRs
stp d8, d9, [x0, #(12 * 8)];
stp d10, d11, [x0, #(14 * 8)];
stp d12, d13, [x0, #(16 * 8)];
stp d14, d15, [x0, #(18 * 8)];
// Move SP in to a temporary to store.
mov x1, sp;
str x1, [x0, #(20 * 8)];
// Return zero to signify this is the SetJump.
mov x0, #0;
ret;
)" ::
: "memory");
}
[[noreturn]]
FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(JumpBuf& Buffer, uint64_t Value) {
__asm volatile(R"(
// x0 contains the jumpbuffer
ldp x19, x20, [x0, #( 0 * 8)];
ldp x21, x22, [x0, #( 2 * 8)];
ldp x23, x24, [x0, #( 4 * 8)];
ldp x25, x26, [x0, #( 6 * 8)];
ldp x27, x28, [x0, #( 8 * 8)];
ldp x29, x30, [x0, #(10 * 8)];
// FPRs
ldp d8, d9, [x0, #(12 * 8)];
ldp d10, d11, [x0, #(14 * 8)];
ldp d12, d13, [x0, #(16 * 8)];
ldp d14, d15, [x0, #(18 * 8)];
// Load SP in to temporary then move
ldr x0, [x0, #(20 * 8)];
mov sp, x0;
// Move value in to result register
mov x0, x1;
ret;
)" ::
: "memory");
}
#else
[[nodiscard]]
FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
__asm volatile(R"(
.intel_syntax noprefix;
// rdi contains the jumpbuffer
mov [rdi + (0 * 8)], rbx;
mov [rdi + (1 * 8)], rsp;
mov [rdi + (2 * 8)], rbp;
mov [rdi + (3 * 8)], r12;
mov [rdi + (4 * 8)], r13;
mov [rdi + (5 * 8)], r14;
mov [rdi + (6 * 8)], r15;
// Return address is on the stack, load it and store
mov rsi, [rsp];
mov [rdi + (7 * 8)], rsi;
// Return zero to signify this is the SetJump.
mov rax, 0;
ret;
.att_syntax prefix;
)" ::
: "memory");
}
[[noreturn]]
FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(JumpBuf& Buffer, uint64_t Value) {
__asm volatile(R"(
.intel_syntax noprefix;
// rdi contains the jumpbuffer
mov rbx, [rdi + (0 * 8)];
mov rsp, [rdi + (1 * 8)];
mov rbp, [rdi + (2 * 8)];
mov r12, [rdi + (3 * 8)];
mov r13, [rdi + (4 * 8)];
mov r14, [rdi + (5 * 8)];
mov r15, [rdi + (6 * 8)];
// Move value in to result register
mov rax, rsi;
// Pop the dead return address off the stack
pop rsi;
// Load the original return address from the jumpbuffer
mov rsi, [rdi + (7 * 8)];
// Return using a jump
jmp rsi;
.att_syntax prefix;
)" ::
: "memory");
}
#endif
} // namespace FEXCore::LongJump
+8 -3
View File
@@ -78,9 +78,14 @@ When generating IR inside of the `OpDispatchBuilder` it is straight forward, jus
This is an intrusive allocator that is used by the `OpDispatchBuilder` for storing IR data. It is a simple linear arena allocator without resizing capabilities.
### OpDispatchBuilder
OpDispatchBuilder provides `IRListView ViewIR()` for handling the IR outside of the class:
* Returns a wrapper container class the allows you to view the IR. This doesn't take ownership of the IR data.
* If the OpDispatcherBuilder changes its IR then changes are also visible to this class
OpDispatchBuilder provides two routines for handling the IR outside of the class
* `IRListView ViewIR();`
* Returns a wrapper container class the allows you to view the IR. This doesn't take ownership of the IR data.
* If the OpDispatcherBuilder changes its IR then changes are also visible to this class
* `IRListView *CreateIRCopy()`
* As the name says, it creates a new copy of the IR that is in the OpDispatchBuilder
* Copying the IR only copies the memory used and doesn't have any free space for optimizations after this copy operation
* Useful for tiered recompilers, AOT, and offline analysis
This class uses two IntrusiveAllocator objects for tracking IR data. `ListData` and `Data` are the object names.
* `ListData` is for tracking the doubly linked list of nodes
-58
View File
@@ -1,58 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <cstdint>
namespace FEXCore {
namespace Core {
struct InternalThreadState;
} // namespace Core
namespace HLE {
struct SourcecodeMap;
} // namespace HLE
// Generic information associated with an executable file.
struct ExecutableFileInfo {
~ExecutableFileInfo();
fextl::unique_ptr<HLE::SourcecodeMap> SourcecodeMap;
fextl::string FileId;
fextl::string Filename;
};
// Information associated with a specific section of an executable file
struct ExecutableFileSectionInfo {
ExecutableFileInfo& FileInfo;
// Start address that the file is mapped to.
uintptr_t FileStartVA;
};
class AbstractCodeCache {
public:
virtual ~AbstractCodeCache() = default;
/**
* Loads a code cache from mapped memory and appends it to the current Core state.
* TODO: Optionally recompiles all contained code blocks at runtime for validation.
*/
virtual void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) = 0;
/**
* Bundles the current Core state (CodeBuffer, GuestToHostMapping, ...) to a code cache and writes it to the given file descriptor.
* Returns true on success.
*/
virtual bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) = 0;
/**
* Function to be called before compiling any code for caching purposes
*/
virtual void InitiateCacheGeneration() = 0;
};
} // namespace FEXCore
+17 -2
View File
@@ -4,7 +4,6 @@
#include <stdint.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/CodeCache.h>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
@@ -14,7 +13,12 @@
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <istream>
#include <ostream>
#include <span>
namespace FEXCore {
class CodeLoader;
struct HostFeatures;
class ForkableSharedMutex;
class ThunkHandler;
@@ -25,11 +29,17 @@ struct CPUState;
struct InternalThreadState;
} // namespace FEXCore::Core
namespace FEXCore::CPU {
class CPUBackend;
}
namespace FEXCore::HLE {
struct SyscallArguments;
class SyscallHandler;
} // namespace FEXCore::HLE
namespace FEXCore::IR {
struct AOTIRCacheEntry;
class IREmitter;
} // namespace FEXCore::IR
@@ -138,13 +148,18 @@ public:
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) = 0;
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) = 0;
virtual AbstractCodeCache& GetCodeCache() = 0;
FEX_DEFAULT_VISIBILITY virtual FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) = 0;
FEX_DEFAULT_VISIBILITY virtual void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) = 0;
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) = 0;
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(
FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start, uint64_t Length) = 0;
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
FEX_DEFAULT_VISIBILITY virtual void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) = 0;
FEX_DEFAULT_VISIBILITY virtual void
ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) = 0;
+6 -6
View File
@@ -146,15 +146,15 @@ struct CPUState {
// - Three are reserved for user-space to setup TLS segments in
// LDT segments are entirely controlled by userspace.
// - Kernel allocates up to 8192 ldt segments.
gdt_segment* segment_arrays[2] {};
gdt_segment *segment_arrays[2] {};
static gdt_segment* GetSegmentFromIndex(CPUState& State, uint16_t Selector) {
static gdt_segment* GetSegmentFromIndex(CPUState &State, uint16_t Selector) {
auto base = State.segment_arrays[(Selector >> 2) & 1];
return &base[Selector >> 3];
}
static uint32_t CalculateGDTBase(gdt_segment GDT) {
uint32_t Base {};
uint32_t Base{};
Base |= GDT.Base2 << 24;
Base |= GDT.Base1 << 16;
Base |= GDT.Base0;
@@ -162,19 +162,19 @@ struct CPUState {
}
static uint32_t CalculateGDTLimit(gdt_segment GDT) {
uint32_t Limit {};
uint32_t Limit{};
Limit |= GDT.Limit1 << 16;
Limit |= GDT.Limit0;
return Limit;
}
static void SetGDTBase(gdt_segment* GDT, uint32_t Base) {
static void SetGDTBase(gdt_segment *GDT, uint32_t Base) {
GDT->Base0 = Base;
GDT->Base1 = Base >> 16;
GDT->Base2 = Base >> 24;
}
static void SetGDTLimit(gdt_segment* GDT, uint32_t Limit) {
static void SetGDTLimit(gdt_segment *GDT, uint32_t Limit) {
GDT->Limit0 = Limit;
GDT->Limit1 = Limit >> 16;
}
@@ -9,6 +9,10 @@
#include <memory>
#include <filesystem>
namespace FEXCore::IR {
struct AOTIRCacheEntry;
}
namespace FEXCore::HLE {
struct SourcecodeLineMapping {
+19 -4
View File
@@ -1,11 +1,13 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <cstdint>
#include <optional>
#include <shared_mutex>
#include <FEXCore/Core/CodeCache.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/fextl/string.h>
namespace FEXCore::IR {
struct AOTIRCacheEntry;
}
namespace FEXCore::Context {
class Context;
@@ -49,6 +51,19 @@ struct ExecutableRangeInfo {
class SyscallHandler;
class SourcecodeResolver;
struct AOTIRCacheEntryLookupResult {
AOTIRCacheEntryLookupResult(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart)
: Entry(Entry)
, VAFileStart(VAFileStart) {}
AOTIRCacheEntryLookupResult(AOTIRCacheEntryLookupResult&&) = default;
FEXCore::IR::AOTIRCacheEntry* Entry;
uintptr_t VAFileStart;
friend class SyscallHandler;
};
class SyscallHandler {
public:
virtual ~SyscallHandler() = default;
@@ -67,7 +82,7 @@ public:
virtual void MarkOvercommitRange(uint64_t Start, uint64_t Length) {}
virtual void UnmarkOvercommitRange(uint64_t Start, uint64_t Length) {}
virtual ExecutableRangeInfo QueryGuestExecutableRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) = 0;
virtual std::optional<ExecutableFileSectionInfo> LookupExecutableFileSection(Core::InternalThreadState& Thread, uint64_t GuestAddr) = 0;
virtual AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddr) = 0;
virtual void PreCompile() {}
+13 -13
View File
@@ -2,24 +2,24 @@
#pragma once
#include <type_traits>
#define FEX_DEF_ENUM_CLASS_BIN_OP(Enum, Op) \
#define FEX_DEF_ENUM_CLASS_BIN_OP(Enum, Op) \
inline constexpr Enum operator Op(Enum lhs, Enum rhs) { \
using Type = std::underlying_type_t<Enum>; \
Type _lhs = static_cast<Type>(lhs); \
Type _rhs = static_cast<Type>(rhs); \
return static_cast<Enum>(_lhs Op _rhs); \
} \
using Type = std::underlying_type_t<Enum>; \
Type _lhs = static_cast<Type>(lhs); \
Type _rhs = static_cast<Type>(rhs); \
return static_cast<Enum>(_lhs Op _rhs); \
} \
inline constexpr uint64_t operator Op(uint64_t lhs, Enum rhs) { \
using Type = std::underlying_type_t<Enum>; \
Type _rhs = static_cast<Type>(rhs); \
return lhs Op _rhs; \
using Type = std::underlying_type_t<Enum>; \
Type _rhs = static_cast<Type>(rhs); \
return lhs Op _rhs; \
}
#define FEX_DEF_ENUM_CLASS_UNARY_OP(Enum, Op) \
#define FEX_DEF_ENUM_CLASS_UNARY_OP(Enum, Op) \
inline constexpr Enum operator Op(Enum rhs) { \
using Type = std::underlying_type_t<Enum>; \
Type _rhs = static_cast<Type>(rhs); \
return static_cast<Enum>(Op _rhs); \
using Type = std::underlying_type_t<Enum>; \
Type _rhs = static_cast<Type>(rhs); \
return static_cast<Enum>(Op _rhs); \
}
#define FEX_DEF_NUM_OPS(Enum) \
-36
View File
@@ -1,36 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/CompilerDefs.h>
#include <cstdint>
// It's longjump without glibc fortification checks.
namespace FEXCore::LongJump {
// JumpBuf definition needs to be public because the frontend needs to understand it.
#if defined(_M_ARM_64)
struct JumpBuf {
// All the registers that are required by AAPCS64 to save.
// GPRs
// X19, X20, X21, X22,
// X23, X24, X25, X26,
// X27, X28, X29, X30,
//
// Lower 64-bits:
// V8, V9, V10, V11,
// V12, V13, V14, V15,
//
// SP,
uint64_t Registers[21];
};
#else
struct JumpBuf {
// Registers to preserve
// RBX, RSP, RBP, R12, R13, R14, R15,
// <return address>
uint64_t Registers[8];
};
#endif
[[nodiscard]] FEX_DEFAULT_VISIBILITY uint64_t SetJump(JumpBuf& Buffer);
[[noreturn]] FEX_DEFAULT_VISIBILITY void LongJump(JumpBuf& Buffer, uint64_t Value);
} // namespace FEXCore::LongJump
+48 -48
View File
@@ -9,17 +9,17 @@ using namespace ARMEmitter;
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)adr(Reg::r30, &Label);
adr(Reg::r30, &Label);
CHECK(DisassembleEncoding(1) == 0x10fffffe);
}
{
ForwardLabel Label;
(void)adr(Reg::r30, &Label);
(void)Bind(&Label);
adr(Reg::r30, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x1000003e);
@@ -27,17 +27,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)adr(Reg::r30, &Label);
adr(Reg::r30, &Label);
CHECK(DisassembleEncoding(1) == 0x10fffffe);
}
{
BiDirectionalLabel Label;
(void)adr(Reg::r30, &Label);
(void)Bind(&Label);
adr(Reg::r30, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x1000003e);
@@ -45,80 +45,80 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)adrp(Reg::r30, &Label);
adrp(Reg::r30, &Label);
CHECK(DisassembleEncoding(1) == 0x9000001e);
}
{
ForwardLabel Label;
(void)adrp(Reg::r30, &Label);
adrp(Reg::r30, &Label);
// Move label a page away
for (size_t i = 0; i < 1023; ++i) {
nop();
}
(void)Bind(&Label);
Bind(&Label);
CHECK(DisassembleEncoding(0) == 0xb000001e);
}
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)adrp(Reg::r30, &Label);
adrp(Reg::r30, &Label);
CHECK(DisassembleEncoding(1) == 0x9000001e);
}
{
BiDirectionalLabel Label;
(void)adrp(Reg::r30, &Label);
adrp(Reg::r30, &Label);
// Move label a page away
for (size_t i = 0; i < 1023; ++i) {
nop();
}
(void)Bind(&Label);
Bind(&Label);
CHECK(DisassembleEncoding(0) == 0xb000001e);
}
{
// Will generate (void)adr.
// Will generate adr.
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)LongAddressGen(Reg::r30, &Label);
LongAddressGen(Reg::r30, &Label);
CHECK(DisassembleEncoding(1) == 0x10fffffe);
}
{
// Will generate nop + (void)adr.
// Will generate nop + adr.
ForwardLabel Label;
(void)LongAddressGen(Reg::r30, &Label);
(void)Bind(&Label);
LongAddressGen(Reg::r30, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xd503201f);
CHECK(DisassembleEncoding(1) == 0x1000003e);
}
{
// Will generate (void)adr.
// Will generate adr.
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)LongAddressGen(Reg::r30, &Label);
LongAddressGen(Reg::r30, &Label);
CHECK(DisassembleEncoding(1) == 0x10fffffe);
}
{
// Will generate nop + (void)adr.
// Will generate nop + adr.
BiDirectionalLabel Label;
(void)LongAddressGen(Reg::r30, &Label);
(void)Bind(&Label);
LongAddressGen(Reg::r30, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xd503201f);
@@ -126,33 +126,33 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
}
{
// Will generate (void)adrp.
// Will generate adrp.
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
// Move (void)adrp 1MB away.
// Move adrp 1MB away.
for (size_t i = 0; i < (1 * 1024 * 1024 / 4); ++i) {
nop();
}
(void)LongAddressGen(Reg::r30, &Label);
LongAddressGen(Reg::r30, &Label);
nop();
CHECK(DisassembleEncoding(262145) == 0x90fff81e);
CHECK(DisassembleEncoding(262146) == 0xd503201f);
}
{
// Will generate nop + (void)adrp.
// Will generate nop + adrp.
ForwardLabel Label;
(void)LongAddressGen(Reg::r30, &Label);
LongAddressGen(Reg::r30, &Label);
// Move label 1MB away, plus a page, and then aligned to a page.
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 2); ++i) {
nop();
}
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xd503201f);
@@ -160,16 +160,16 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
}
{
// Will generate (void)adrp + add.
// Will generate adrp + add.
ForwardLabel Label;
(void)LongAddressGen(Reg::r30, &Label);
LongAddressGen(Reg::r30, &Label);
// Move label 1MB away, plus a page, plus one instruction.
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 1); ++i) {
nop();
}
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb000081e);
@@ -178,33 +178,33 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
{
// Will generate (void)adrp.
// Will generate adrp.
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
// Move (void)adrp 1MB away.
// Move adrp 1MB away.
for (size_t i = 0; i < (1 * 1024 * 1024 / 4); ++i) {
nop();
}
(void)LongAddressGen(Reg::r30, &Label);
LongAddressGen(Reg::r30, &Label);
nop();
CHECK(DisassembleEncoding(262145) == 0x90fff81e);
CHECK(DisassembleEncoding(262146) == 0xd503201f);
}
{
// Will generate nop + (void)adrp.
// Will generate nop + adrp.
BiDirectionalLabel Label;
(void)LongAddressGen(Reg::r30, &Label);
LongAddressGen(Reg::r30, &Label);
// Move label 1MB away, plus a page, and then aligned to a page.
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 2); ++i) {
nop();
}
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xd503201f);
@@ -212,16 +212,16 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
}
{
// Will generate (void)adrp + add.
// Will generate adrp + add.
BiDirectionalLabel Label;
(void)LongAddressGen(Reg::r30, &Label);
LongAddressGen(Reg::r30, &Label);
// Move label 1MB away, plus a page, plus one instruction.
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 1); ++i) {
nop();
}
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb000081e);
+96 -96
View File
@@ -9,17 +9,17 @@ using namespace ARMEmitter;
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Conditional branch immediate") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)b(Condition::CC_PL, &Label);
b(Condition::CC_PL, &Label);
CHECK(DisassembleEncoding(1) == 0x54ffffe5);
}
{
ForwardLabel Label;
(void)b(Condition::CC_PL, &Label);
(void)Bind(&Label);
b(Condition::CC_PL, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x54000025);
@@ -27,17 +27,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Conditional branch immediat
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)b(Condition::CC_PL, &Label);
b(Condition::CC_PL, &Label);
CHECK(DisassembleEncoding(1) == 0x54ffffe5);
}
{
BiDirectionalLabel Label;
(void)b(Condition::CC_PL, &Label);
(void)Bind(&Label);
b(Condition::CC_PL, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x54000025);
@@ -46,17 +46,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Conditional branch immediat
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Branch consistent conditional") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)bc(Condition::CC_PL, &Label);
bc(Condition::CC_PL, &Label);
CHECK(DisassembleEncoding(1) == 0x54fffff5);
}
{
ForwardLabel Label;
(void)bc(Condition::CC_PL, &Label);
(void)Bind(&Label);
bc(Condition::CC_PL, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x54000035);
@@ -64,17 +64,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Branch consistent condition
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)bc(Condition::CC_PL, &Label);
bc(Condition::CC_PL, &Label);
CHECK(DisassembleEncoding(1) == 0x54fffff5);
}
{
BiDirectionalLabel Label;
(void)bc(Condition::CC_PL, &Label);
(void)Bind(&Label);
bc(Condition::CC_PL, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x54000035);
@@ -89,17 +89,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch regist
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immediate") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)b(&Label);
b(&Label);
CHECK(DisassembleEncoding(1) == 0x17ffffff);
}
{
ForwardLabel Label;
(void)b(&Label);
(void)Bind(&Label);
b(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x14000001);
@@ -107,17 +107,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immedi
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)b(&Label);
b(&Label);
CHECK(DisassembleEncoding(1) == 0x17ffffff);
}
{
BiDirectionalLabel Label;
(void)b(&Label);
(void)Bind(&Label);
b(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x14000001);
@@ -125,17 +125,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immedi
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)bl(&Label);
bl(&Label);
CHECK(DisassembleEncoding(1) == 0x97ffffff);
}
{
ForwardLabel Label;
(void)bl(&Label);
(void)Bind(&Label);
bl(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x94000001);
@@ -143,17 +143,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immedi
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)bl(&Label);
bl(&Label);
CHECK(DisassembleEncoding(1) == 0x97ffffff);
}
{
BiDirectionalLabel Label;
(void)bl(&Label);
(void)Bind(&Label);
bl(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x94000001);
@@ -162,17 +162,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immedi
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)cbz(Size::i32Bit, Reg::r29, &Label);
cbz(Size::i32Bit, Reg::r29, &Label);
CHECK(DisassembleEncoding(1) == 0x34fffffd);
}
{
ForwardLabel Label;
(void)cbz(Size::i32Bit, Reg::r29, &Label);
(void)Bind(&Label);
cbz(Size::i32Bit, Reg::r29, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x3400003d);
@@ -180,17 +180,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)cbz(Size::i32Bit, Reg::r29, &Label);
cbz(Size::i32Bit, Reg::r29, &Label);
CHECK(DisassembleEncoding(1) == 0x34fffffd);
}
{
BiDirectionalLabel Label;
(void)cbz(Size::i32Bit, Reg::r29, &Label);
(void)Bind(&Label);
cbz(Size::i32Bit, Reg::r29, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x3400003d);
@@ -198,17 +198,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)cbz(Size::i64Bit, Reg::r29, &Label);
cbz(Size::i64Bit, Reg::r29, &Label);
CHECK(DisassembleEncoding(1) == 0xb4fffffd);
}
{
ForwardLabel Label;
(void)cbz(Size::i64Bit, Reg::r29, &Label);
(void)Bind(&Label);
cbz(Size::i64Bit, Reg::r29, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb400003d);
@@ -216,17 +216,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)cbz(Size::i64Bit, Reg::r29, &Label);
cbz(Size::i64Bit, Reg::r29, &Label);
CHECK(DisassembleEncoding(1) == 0xb4fffffd);
}
{
BiDirectionalLabel Label;
(void)cbz(Size::i64Bit, Reg::r29, &Label);
(void)Bind(&Label);
cbz(Size::i64Bit, Reg::r29, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb400003d);
@@ -234,17 +234,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)cbnz(Size::i32Bit, Reg::r29, &Label);
cbnz(Size::i32Bit, Reg::r29, &Label);
CHECK(DisassembleEncoding(1) == 0x35fffffd);
}
{
ForwardLabel Label;
(void)cbnz(Size::i32Bit, Reg::r29, &Label);
(void)Bind(&Label);
cbnz(Size::i32Bit, Reg::r29, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x3500003d);
@@ -252,17 +252,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)cbnz(Size::i32Bit, Reg::r29, &Label);
cbnz(Size::i32Bit, Reg::r29, &Label);
CHECK(DisassembleEncoding(1) == 0x35fffffd);
}
{
BiDirectionalLabel Label;
(void)cbnz(Size::i32Bit, Reg::r29, &Label);
(void)Bind(&Label);
cbnz(Size::i32Bit, Reg::r29, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x3500003d);
@@ -270,17 +270,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)cbnz(Size::i64Bit, Reg::r29, &Label);
cbnz(Size::i64Bit, Reg::r29, &Label);
CHECK(DisassembleEncoding(1) == 0xb5fffffd);
}
{
ForwardLabel Label;
(void)cbnz(Size::i64Bit, Reg::r29, &Label);
(void)Bind(&Label);
cbnz(Size::i64Bit, Reg::r29, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb500003d);
@@ -288,17 +288,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)cbnz(Size::i64Bit, Reg::r29, &Label);
cbnz(Size::i64Bit, Reg::r29, &Label);
CHECK(DisassembleEncoding(1) == 0xb5fffffd);
}
{
BiDirectionalLabel Label;
(void)cbnz(Size::i64Bit, Reg::r29, &Label);
(void)Bind(&Label);
cbnz(Size::i64Bit, Reg::r29, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb500003d);
@@ -307,17 +307,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)tbz(Reg::r29, 0, &Label);
tbz(Reg::r29, 0, &Label);
CHECK(DisassembleEncoding(1) == 0x3607fffd);
}
{
ForwardLabel Label;
(void)tbz(Reg::r29, 0, &Label);
(void)Bind(&Label);
tbz(Reg::r29, 0, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x3600003d);
@@ -325,17 +325,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)tbz(Reg::r29, 0, &Label);
tbz(Reg::r29, 0, &Label);
CHECK(DisassembleEncoding(1) == 0x3607fffd);
}
{
BiDirectionalLabel Label;
(void)tbz(Reg::r29, 0, &Label);
(void)Bind(&Label);
tbz(Reg::r29, 0, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x3600003d);
@@ -343,17 +343,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)tbz(Reg::r29, 63, &Label);
tbz(Reg::r29, 63, &Label);
CHECK(DisassembleEncoding(1) == 0xb6fffffd);
}
{
ForwardLabel Label;
(void)tbz(Reg::r29, 63, &Label);
(void)Bind(&Label);
tbz(Reg::r29, 63, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb6f8003d);
@@ -361,17 +361,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)tbz(Reg::r29, 63, &Label);
tbz(Reg::r29, 63, &Label);
CHECK(DisassembleEncoding(1) == 0xb6fffffd);
}
{
BiDirectionalLabel Label;
(void)tbz(Reg::r29, 63, &Label);
(void)Bind(&Label);
tbz(Reg::r29, 63, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb6f8003d);
@@ -379,17 +379,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)tbnz(Reg::r29, 0, &Label);
tbnz(Reg::r29, 0, &Label);
CHECK(DisassembleEncoding(1) == 0x3707fffd);
}
{
ForwardLabel Label;
(void)tbnz(Reg::r29, 0, &Label);
(void)Bind(&Label);
tbnz(Reg::r29, 0, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x3700003d);
@@ -397,17 +397,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)tbnz(Reg::r29, 0, &Label);
tbnz(Reg::r29, 0, &Label);
CHECK(DisassembleEncoding(1) == 0x3707fffd);
}
{
BiDirectionalLabel Label;
(void)tbnz(Reg::r29, 0, &Label);
(void)Bind(&Label);
tbnz(Reg::r29, 0, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x3700003d);
@@ -415,17 +415,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)tbnz(Reg::r29, 63, &Label);
tbnz(Reg::r29, 63, &Label);
CHECK(DisassembleEncoding(1) == 0xb7fffffd);
}
{
ForwardLabel Label;
(void)tbnz(Reg::r29, 63, &Label);
(void)Bind(&Label);
tbnz(Reg::r29, 63, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb7f8003d);
@@ -433,17 +433,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
{
BiDirectionalLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
(void)tbnz(Reg::r29, 63, &Label);
tbnz(Reg::r29, 63, &Label);
CHECK(DisassembleEncoding(1) == 0xb7fffffd);
}
{
BiDirectionalLabel Label;
(void)tbnz(Reg::r29, 63, &Label);
(void)Bind(&Label);
tbnz(Reg::r29, 63, &Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xb7f8003d);
+14 -14
View File
@@ -1323,7 +1323,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: LDAPR/STLR unscaled imme
TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal") {
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
ldr(WReg::w30, &Label);
@@ -1332,7 +1332,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
ldr(SReg::s30, &Label);
@@ -1341,7 +1341,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
ldr(XReg::x30, &Label);
@@ -1350,7 +1350,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
ldr(DReg::d30, &Label);
@@ -1359,7 +1359,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
ldrsw(XReg::x30, &Label);
@@ -1368,7 +1368,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
ldr(QReg::q30, &Label);
@@ -1377,7 +1377,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
BackwardLabel Label;
(void)Bind(&Label);
Bind(&Label);
dc32(0);
prfm(Prefetch::PLDL1KEEP, &Label);
@@ -1387,7 +1387,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
ForwardLabel Label;
ldr(WReg::w30, &Label);
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x1800003e);
@@ -1396,7 +1396,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
ForwardLabel Label;
ldr(SReg::s30, &Label);
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x1c00003e);
@@ -1405,7 +1405,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
ForwardLabel Label;
ldr(XReg::x30, &Label);
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x5800003e);
@@ -1414,7 +1414,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
ForwardLabel Label;
ldr(DReg::d30, &Label);
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x5c00003e);
@@ -1423,7 +1423,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
ForwardLabel Label;
ldrsw(XReg::x30, &Label);
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x9800003e);
@@ -1432,7 +1432,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
ForwardLabel Label;
ldr(QReg::q30, &Label);
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0x9c00003e);
@@ -1441,7 +1441,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
{
ForwardLabel Label;
prfm(Prefetch::PLDL1KEEP, &Label);
(void)Bind(&Label);
Bind(&Label);
dc32(0);
CHECK(DisassembleEncoding(0) == 0xd8000020);
+1 -6
View File
@@ -8,7 +8,7 @@
#include <algorithm>
#include <fcntl.h>
#include <memory_resource>
#include <string_view>
#include <string>
#ifndef _WIN32
#include <linux/limits.h>
#include <sys/sendfile.h>
@@ -354,11 +354,6 @@ inline fextl::string GetFilename(const fextl::string& Path) {
return PathToString(std::filesystem::path(Path).filename());
}
inline std::string_view GetFilename(std::string_view Path) {
auto Filename = PathToString(std::filesystem::path(Path).filename());
return Path.substr(Path.size() - Filename.size());
}
inline fextl::string ParentPath(const fextl::string& Path) {
return PathToString(std::filesystem::path(Path).parent_path());
}
+3 -3
View File
@@ -333,7 +333,7 @@ def TryInstallRootFS():
return DidInstall
def TryBasicProgramExecution():
return subprocess.call(["FEX", "/usr/bin/uname", "-a"]) == 0
return subprocess.call(["FEXInterpreter", "/usr/bin/uname", "-a"]) == 0
def ExitWithStatus(Status):
# Remove the cached credentials
@@ -380,7 +380,7 @@ def main():
print ("FEX is now installed. Trying basic program run")
if not TryBasicProgramExecution():
print ("FEX failed to run. Not continuing")
print ("FEXInterpreter failed to run. Not continuing")
ExitWithStatus(-1)
print ("")
@@ -391,7 +391,7 @@ def main():
print ("# steam is a bash script. Wrap with FEXBash")
print ("\tFEXBash steam")
print ("# Full path execution execution will wrap the application if it exists in the rootfs")
print ("\tFEX /usr/bin/uname")
print ("\tFEXInterpreter /usr/bin/uname")
print ("# Freestanding x86/x86-64 programs can be executed directly. binfmt_misc will redirect to FEX")
print ("\t$HOME/PetalCrashOnline.AppImage")
print ("# If you need a terminal that emulates everything.")
+12 -1
View File
@@ -6,10 +6,15 @@ import subprocess
# Check if FEX indicates support for AVX
def DoesFEXSupportAVX(mode):
fex_interpreter_path = os.path.dirname(sys.argv[7]) + "/FEX"
fex_interpreter_path = os.path.dirname(sys.argv[7]) + "/FEXInterpreter"
args = list()
args.append(fex_interpreter_path)
if (mode == "guest"):
ROOTFS_ENV = os.getenv("ROOTFS")
if ROOTFS_ENV != None:
args.append("-R")
args.append(ROOTFS_ENV)
args.append('/bin/cat')
args.append('/proc/cpuinfo')
@@ -105,6 +110,12 @@ RunnerArgs = []
RunnerArgs.append(fexecutable)
if (mode == "guest"):
ROOTFS_ENV = os.getenv("ROOTFS")
if ROOTFS_ENV != None:
RunnerArgs.append("-R")
RunnerArgs.append(ROOTFS_ENV)
# Add the rest of the arguments
for i in range(len(sys.argv) - StartingFEXArgsOffset):
RunnerArgs.append(sys.argv[StartingFEXArgsOffset + i])
+5 -9
View File
@@ -5,9 +5,9 @@ import os.path
from os import path
from shutil import which
# Args: <Known Failures file> <Known Failures Type File> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <FullTestName> <Test Harness Executable> <Args>...
# Args: <Known Failures file> <Known Failures Type File> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <Test Harness Executable> <Args>...
if (len(sys.argv) < 8):
if (len(sys.argv) < 7):
sys.exit()
known_failures = {}
@@ -19,9 +19,8 @@ disabled_tests_type_file = sys.argv[4]
disabled_tests_runner_file = sys.argv[5]
current_test = sys.argv[6]
full_test_name = sys.argv[7]
runner = sys.argv[8]
args_start_index = 9
runner = sys.argv[7]
args_start_index = 8
# Open the known failures file and add it to a dictionary
with open(known_failures_file) as kff:
@@ -64,10 +63,7 @@ Process = subprocess.Popen(RunnerArgs)
Process.wait()
ResultCode = Process.returncode
# Check for known failures - try full test name first, then partial test name
is_known_failure = known_failures.get(full_test_name) or known_failures.get(current_test)
if (is_known_failure):
if (known_failures.get(current_test)):
# If the test is on the known failures list
if (ResultCode):
# If we errored but are on the known failures list then "pass" the test
+37
View File
@@ -5,13 +5,50 @@
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include "cpp-optparse/OptionParser.h"
#include "git_version.h"
#include <stdint.h>
namespace FEX::ArgLoader {
void FEX::ArgLoader::ArgLoader::PreLoad() {
RemainingArgs.clear();
ProgramArguments.clear();
if (Type == LoadType::WITHOUT_FEXLOADER_PARSER) {
LoadWithoutArguments();
return;
}
optparse::OptionParser Parser {};
Parser.version("FEX-Emu (" GIT_DESCRIBE_STRING ") ");
optparse::OptionGroup CPUGroup(Parser, "CPU Core options");
optparse::OptionGroup EmulationGroup(Parser, "Emulation options");
optparse::OptionGroup DebugGroup(Parser, "Debug options");
optparse::OptionGroup HacksGroup(Parser, "Hacks options");
optparse::OptionGroup MiscGroup(Parser, "Miscellaneous options");
optparse::OptionGroup LoggingGroup(Parser, "Logging options");
#define BEFORE_PARSE
#include <FEXCore/Config/ConfigOptions.inl>
Parser.add_option_group(CPUGroup);
Parser.add_option_group(EmulationGroup);
Parser.add_option_group(DebugGroup);
Parser.add_option_group(HacksGroup);
Parser.add_option_group(MiscGroup);
Parser.add_option_group(LoggingGroup);
optparse::Values Options = Parser.parse_args(argc, argv);
using int32 = int32_t;
using uint32 = uint32_t;
#define AFTER_PARSE
#include <FEXCore/Config/ConfigOptions.inl>
RemainingArgs = Parser.args();
ProgramArguments = Parser.parsed_args();
}
void FEX::ArgLoader::ArgLoader::LoadWithoutArguments() {
// Skip argument 0, which will be the interpreter
for (int i = 1; i < argc; ++i) {
RemainingArgs.emplace_back(argv[i]);
+13 -1
View File
@@ -8,8 +8,14 @@
namespace FEX::ArgLoader {
class ArgLoader final : public FEXCore::Config::Layer {
public:
explicit ArgLoader(int argc, char** argv)
enum class LoadType {
WITH_FEXLOADER_PARSER,
WITHOUT_FEXLOADER_PARSER,
};
explicit ArgLoader(LoadType Type, int argc, char** argv)
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ARGUMENTS)
, Type {Type}
, argc {argc}
, argv {argv} {
PreLoad();
@@ -19,6 +25,7 @@ public:
// Intentional no-op.
}
void PreLoad();
void LoadWithoutArguments();
fextl::vector<fextl::string> Get() {
return RemainingArgs;
}
@@ -26,7 +33,12 @@ public:
return ProgramArguments;
}
LoadType GetLoadType() const {
return Type;
}
private:
LoadType Type;
int argc {};
char** argv {};
+9 -4
View File
@@ -373,11 +373,11 @@ fextl::string RecoverGuestProgramFilename(fextl::string Program, bool ExecFDInte
//
// Examples:
// - Regular execve. Application must exist on disk.
// execve binfmt_misc args layout: `FEX <Path provided to execve pathname> <user provided argv[0]> <user provided argv[n]>...`
// execve binfmt_misc args layout: `FEXInterpreter <Path provided to execve pathname> <user provided argv[0]> <user provided argv[n]>...`
// - Regular execveat with FD. FD is backed by application on disk.
// execveat binfmt_misc args layout: `FEX <Path provided to execve pathname> <user provided argv[0]> <user provided argv[n]>...`
// execveat binfmt_misc args layout: `FEXInterpreter <Path provided to execve pathname> <user provided argv[0]> <user provided argv[n]>...`
// - Regular execveat with FD. FD points to file on disk that has been deleted.
// execveat binfmt_misc args layout: `FEX /dev/fd/<FD> <user provided argv[0]> <user provided argv[n]>...`
// execveat binfmt_misc args layout: `FEXInterpreter /dev/fd/<FD> <user provided argv[0]> <user provided argv[n]>...`
#ifndef _WIN32
if (ExecFDInterp || ProgramFDFromEnv != -1) {
// Only in the case that FEX is executing an FD will the program argument potentially be a symlink.
@@ -448,7 +448,8 @@ ApplicationNames GetApplicationNames(const fextl::vector<fextl::string>& Args, b
return ApplicationNames {std::move(Program), std::move(ProgramName)};
}
void LoadConfig(fextl::string ProgramName, char** const envp, const PortableInformation& PortableInfo) {
void LoadConfig(fextl::unique_ptr<FEX::ArgLoader::ArgLoader> ArgsLoader, fextl::string ProgramName, char** const envp,
const PortableInformation& PortableInfo) {
const bool IsPortable = PortableInfo.IsPortable;
FEX::Config::InitializeConfigs(PortableInfo);
FEXCore::Config::Initialize();
@@ -475,6 +476,10 @@ void LoadConfig(fextl::string ProgramName, char** const envp, const PortableInfo
}
}
if (ArgsLoader && ArgsLoader->GetLoadType() == FEX::ArgLoader::ArgLoader::LoadType::WITH_FEXLOADER_PARSER) {
FEXCore::Config::AddLayer(std::move(ArgsLoader));
}
const char* AppConfig = getenv("FEX_APP_CONFIG");
if (AppConfig) {
fextl::string AppConfigStr = AppConfig;
+4 -2
View File
@@ -37,7 +37,7 @@ struct ApplicationNames {
struct PortableInformation {
bool IsPortable;
// Path of folder containing FEX (including / at the end)
// Path of folder containing FEXInterpreter (including / at the end)
fextl::string InterpreterPath;
};
@@ -52,10 +52,12 @@ ApplicationNames GetApplicationNames(const fextl::vector<fextl::string>& Args, b
/**
* @brief Loads the FEX and application configurations for the application that is getting ready to run.
*
* @param ArgLoader Optional argument loader for argument based config options
* @param ProgramName Optional program name, if non-empty application specific configurations will be loaded
* @param envp Optional `envp` passed to main(...)
*/
void LoadConfig(fextl::string ProgramName = {}, char** const envp = nullptr, const PortableInformation& PortableInfo = {});
void LoadConfig(fextl::unique_ptr<FEX::ArgLoader::ArgLoader> ArgLoader = {}, fextl::string ProgramName = {}, char** const envp = nullptr,
const PortableInformation& PortableInfo = {});
const char* GetHomeDirectory();
+2 -2
View File
@@ -121,7 +121,7 @@ fextl::string GetServerMountFolder() {
// This is due to pressure-vesssel being a chroot environment.
// It by default maps the host-filesystem to `/run/host/` so we need to redirect.
// After pressure-vessel is fully set up it will set the `FEX_ROOTFS` environment variable,
// which FEX will pick up.
// which the FEXInterpreter will pick up on.
Folder = "/run/host/" + Folder;
}
@@ -244,7 +244,7 @@ int ConnectToAndStartServer(std::string_view InterpreterPath) {
}
fextl::string FEXServerPath = fextl::fmt::format("{}/FEXServer", InterpreterDir);
// Check if a local FEXServer next to FEX exists
// Check if a local FEXServer next to FEXInterpreter exists
// If it does then it takes priority over the installed one
if (!FHU::Filesystem::Exists(FEXServerPath)) {
FEXServerPath = "FEXServer";
+5 -4
View File
@@ -21,9 +21,10 @@ if (NOT MINGW_BUILD)
add_subdirectory(CodeSizeValidation/)
add_subdirectory(LinuxEmulation/)
add_subdirectory(FEXInterpreter/)
add_subdirectory(FEXLoader/)
add_subdirectory(pidof/)
if (BUILD_TESTING)
add_subdirectory(TestHarnessRunner/)
endif()
endif()
if (BUILD_TESTS)
add_subdirectory(TestHarnessRunner/)
endif()
+9 -11
View File
@@ -460,9 +460,8 @@ public:
}
// These are no-ops implementations of the SyscallHandler API
std::optional<FEXCore::ExecutableFileSectionInfo>
LookupExecutableFileSection(FEXCore::Core::InternalThreadState& Thread, uint64_t GuestAddr) override {
return std::nullopt;
FEXCore::HLE::AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddr) override {
return {0, 0};
}
FEXCore::HLE::ExecutableRangeInfo QueryGuestExecutableRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) override {
@@ -510,8 +509,6 @@ int main(int argc, char** argv, char** const envp) {
// Disable vixl simulator indirect calls as it can affect instruction counts.
FEXCore::Config::Set(FEXCore::Config::CONFIG_DISABLE_VIXL_INDIRECT_RUNTIME_CALLS, "1");
FEXCore::Config::Set(FEXCore::Config::CONFIG_TSOENABLED, "0");
// Host feature override. Only supports overriding SVE width.
enum HostFeatures {
FEATURE_SVE128 = (1U << 0),
@@ -583,12 +580,11 @@ int main(int argc, char** argv, char** const envp) {
}
if (TestHeaderData->EnabledHostFeatures & FEATURE_TSO) {
// Always disable auto migration.
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOAUTOMIGRATION, "0");
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "1");
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, "1");
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, "1");
} else {
// Override the TSO default setting, since TSO is not relevant for most tests
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0");
}
// Always enable ARMv8.1 LSE atomics.
@@ -638,6 +634,8 @@ int main(int argc, char** argv, char** const envp) {
}
if (TestHeaderData->DisabledHostFeatures & FEATURE_TSO) {
// Always disable auto migration.
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOAUTOMIGRATION, "0");
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0");
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, "0");
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, "0");
@@ -682,13 +680,13 @@ int main(int argc, char** argv, char** const envp) {
auto GDT = FEXCore::Core::CPUState::GetSegmentFromIndex(Frame->State, Frame->State.cs_idx);
FEXCore::Core::CPUState::SetGDTBase(GDT, 0);
FEXCore::Core::CPUState::SetGDTLimit(GDT, 0xF'FFFFU);
Frame->State.cs_cached =
FEXCore::Core::CPUState::CalculateGDTBase(*FEXCore::Core::CPUState::GetSegmentFromIndex(Frame->State, Frame->State.cs_idx));
Frame->State.cs_cached = FEXCore::Core::CPUState::CalculateGDTBase(*FEXCore::Core::CPUState::GetSegmentFromIndex(Frame->State, Frame->State.cs_idx));
if (TestHeaderData->Bitness == 64) {
GDT->L = 1; // L = Long Mode = 64-bit
GDT->D = 0; // D = Default Operand SIze = Reserved
} else {
}
else {
GDT->L = 0; // L = Long Mode = 32-bit
GDT->D = 1; // D = Default Operand Size = 32-bit
}
+2 -2
View File
@@ -22,8 +22,8 @@ public:
}
// These are no-ops implementations of the SyscallHandler API
std::optional<FEXCore::ExecutableFileSectionInfo> LookupExecutableFileSection(FEXCore::Core::InternalThreadState&, uint64_t) override {
return std::nullopt;
FEXCore::HLE::AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddr) override {
return {0, 0};
}
FEXCore::HLE::ExecutableRangeInfo QueryGuestExecutableRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) override {
+2 -2
View File
@@ -4,7 +4,7 @@
namespace FEX {
static inline std::optional<fextl::string> GetSelfPath() {
// Read the FEX path from `/proc/self/exe` which is always a symlink to the absolute path of the executable running.
// Read the FEXInterpreter path from `/proc/self/exe` which is always a symlink to the absolute path of the executable running.
// This way we can get the parent path that the application is executing from.
char SelfPath[PATH_MAX];
auto Result = readlink("/proc/self/exe", SelfPath, PATH_MAX);
@@ -34,7 +34,7 @@ static inline FEX::Config::PortableInformation ReadPortabilityInformation() {
return {false, {}};
}
// Extract the absolute path from the FEX path
// Extract the absolute path from the FEXInterpreter path
return {true, *SelfPath};
}
} // namespace FEX
+1 -1
View File
@@ -20,5 +20,5 @@ endif()
install(TARGETS FEXBash
RUNTIME
DESTINATION bin
COMPONENT Runtime
COMPONENT runtime
)
+20 -19
View File
@@ -6,7 +6,9 @@ desc: Launches bash under FEX and passes arguments via -c to it
$end_info$
*/
#include <FEXCore/fextl/fmt.h>
#include "Common/ArgumentLoader.h"
#include <FEXCore/Config/Config.h>
#include <filesystem>
#include <string>
@@ -14,56 +16,55 @@ $end_info$
#include <vector>
int main(int argc, char** argv, char** const envp) {
// Skip argv[0].
const int ArgCount = argc - 1;
const bool EmptyArgs = ArgCount == 0;
FEX::ArgLoader::ArgLoader ArgsLoader(FEX::ArgLoader::ArgLoader::LoadType::WITHOUT_FEXLOADER_PARSER, argc, argv);
auto Args = ArgsLoader.Get();
std::vector<const char*> Argv;
// FEX will handle finding bash in the rootfs
// FEXInterpreter will handle finding bash in the rootfs
// Use /bin/sh for -c commands and /bin/bash for interactive mode
const char* BashPath = EmptyArgs ? "/bin/bash" : "/bin/sh";
const char* BashPath = Args.empty() ? "/bin/bash" : "/bin/sh";
std::string FEXPath = std::filesystem::path(argv[0]).parent_path().string() + "/FEX";
std::string FEXInterpreterPath = std::filesystem::path(argv[0]).parent_path().string() + "/FEXInterpreter";
// Check if a local FEX to FEXBash exists
// Check if a local FEXInterpreter to FEXBash exists
// If it does then it takes priority over the installed one
if (!std::filesystem::exists(FEXPath)) {
if (!std::filesystem::exists(FEXInterpreterPath)) {
char FEXBashPath[PATH_MAX];
auto Result = readlink("/proc/self/exe", FEXBashPath, PATH_MAX);
if (Result != -1) {
FEXPath = std::filesystem::path(&FEXBashPath[0], &FEXBashPath[Result]).parent_path().string() + "/FEX";
FEXInterpreterPath = std::filesystem::path(&FEXBashPath[0], &FEXBashPath[Result]).parent_path().string() + "/FEXInterpreter";
}
if (!std::filesystem::exists(FEXPath)) {
fmt::print(stderr, "Could not locate FEX executable\n");
if (!std::filesystem::exists(FEXInterpreterPath)) {
fmt::print(stderr, "Could not locate FEXInterpreter executable\n");
std::abort();
}
}
const char* FEXArgs[] = {
FEXPath.c_str(),
FEXInterpreterPath.c_str(),
BashPath,
"-c",
};
// Remove -c argument if arguments are empty
// Lets us start an emulated bash instance
const size_t FEXArgsCount = std::size(FEXArgs) - (EmptyArgs ? 1 : 0);
const size_t FEXArgsCount = std::size(FEXArgs) - (Args.empty() ? 1 : 0);
Argv.resize(ArgCount + FEXArgsCount);
Argv.resize(Args.size() + FEXArgsCount);
// Pass in the FEX arguments
// Pass in the FEXInterpreter arguments
for (size_t i = 0; i < FEXArgsCount; ++i) {
Argv[i] = FEXArgs[i];
}
// Bring in passed in arguments
for (size_t i = 0; i < ArgCount; ++i) {
Argv[i + FEXArgsCount] = argv[i + 1];
for (size_t i = 0; i < Args.size(); ++i) {
Argv[i + FEXArgsCount] = Args[i].c_str();
}
// Set --norc when no arguments are passed so PS1 doesn't get overwritten
const char* NoRC = "--norc";
if (EmptyArgs) {
if (Args.empty()) {
Argv.emplace_back(NoRC);
}
+1 -1
View File
@@ -25,4 +25,4 @@ endif()
install(TARGETS FEXConfig
RUNTIME
DESTINATION bin
COMPONENT Runtime)
COMPONENT runtime)
+2
View File
@@ -175,6 +175,7 @@ static void LoadDefaultSettings() {
#include <FEXCore/Config/ConfigValues.inl>
// Erase unnamed options which shouldn't be set
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS_INTERPRETER);
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_INTERPRETER_INSTALLED);
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_FILENAME);
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_CONFIG_NAME);
@@ -383,6 +384,7 @@ static bool OpenFile(fextl::string Filename) {
#include <FEXCore/Config/ConfigValues.inl>
// Erase unnamed options which shouldn't be set
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS_INTERPRETER);
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_INTERPRETER_INSTALLED);
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_FILENAME);
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_CONFIG_NAME);
+1 -1
View File
@@ -6,7 +6,7 @@ add_library(${NAME} SHARED ${SRCS})
install(TARGETS ${NAME}
RUNTIME
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}/gdb
COMPONENT Development)
COMPONENT LIBRARY)
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
target_include_directories(${NAME} PRIVATE ${CMAKE_BINARY_DIR}/generated)
Loaded 100 of 214 files, more files were not shown because too many files have changed in this diff. Show more