Compare commits

..
2 Commits
Author SHA1 Message Date
Ryan Houdek 4a7839b5ac Docs: Update for release FEX-2311.1 2023-11-11 11:59:57 -08:00
Ryan Houdek d8efcb39b8 FEX: Only pass CPU tunables to FEXCore and FEXLoader
This fixes an issue where CPU tunables were ending up in the thunk
generator which means if your CPU doesn't support all the features on
the *Builder* then it would crash with SIGILL. This was happening with
Canonical's runners because they typically only support ARMv8.2 but we
are compiling packages to run on ARMv8.4 devices.

cc: FEX-2311.1
2023-11-11 11:58:25 -08:00
765 changed files with 142462 additions and 120799 deletions

No files matched your search

-109
View File
@@ -1,109 +0,0 @@
Language: Cpp
BasedOnStyle: WebKit
AccessModifierOffset: -2
AlignAfterOpenBracket: Align
AlignArrayOfStructures: None
AlignConsecutiveAssignments: None
AlignConsecutiveBitFields: Consecutive
AlignConsecutiveDeclarations: None
AlignConsecutiveMacros: None
AlignEscapedNewlines: DontAlign
AlignOperands: Align
AlignTrailingComments: true
AllowAllParametersOfDeclarationOnNextLine: false
AllowShortCaseLabelsOnASingleLine: true
AllowShortEnumsOnASingleLine: true
AllowShortFunctionsOnASingleLine: Empty
AllowShortIfStatementsOnASingleLine: WithoutElse
AllowShortLambdasOnASingleLine: Inline
AlwaysBreakAfterDefinitionReturnType: None
AlwaysBreakAfterReturnType: None
AlwaysBreakBeforeMultilineStrings: false
AlwaysBreakTemplateDeclarations: true
AttributeMacros:
- JEMALLOC_NOTHROW
- FEX_ALIGNED
- FEX_ANNOTATE
- FEX_DEFAULT_VISIBILITY
- FEX_NAKED
- FEX_PACKED
- FEXCORE_PRESERVE_ALL_ATTR
- GLIBC_ALIAS_FUNCTION
BinPackArguments: true
BinPackParameters: true
BitFieldColonSpacing: Both
BreakAfterAttributes: Always # clang 16 required
BreakBeforeBraces: Attach
BreakBeforeBinaryOperators: None
BreakBeforeInlineASMColon: OnlyMultiline # clang 16 required
BreakBeforeTernaryOperators: false
BreakConstructorInitializers: BeforeComma
BreakInheritanceList: BeforeColon
ColumnLimit: 140
CompactNamespaces: false
ConstructorInitializerIndentWidth: 2
ContinuationIndentWidth: 2
Cpp11BracedListStyle: true
DerivePointerAlignment: false
EmptyLineAfterAccessModifier: Leave
EmptyLineBeforeAccessModifier: Leave
ExperimentalAutoDetectBinPacking: false
FixNamespaceComments: true
IncludeBlocks: Preserve
IndentAccessModifiers: false
IndentCaseBlocks: false
IndentCaseLabels: false
IndentExternBlock: AfterExternBlock
IndentGotoLabels: false
IndentPPDirectives: None
IndentRequires: false
IndentWidth: 2
InsertBraces: true
KeepEmptyLinesAtTheStartOfBlocks: true
LambdaBodyIndentation: OuterScope
LineEnding: LF # clang 16 required
MaxEmptyLinesToKeep: 2
NamespaceIndentation: Inner
QualifierAlignment: Left
PackConstructorInitializers: Never
PenaltyBreakAssignment: 2
PenaltyBreakBeforeFirstCallParameter: 2
PenaltyBreakOpenParenthesis: 2
PenaltyBreakString: 10
PenaltyBreakTemplateDeclaration: 8
PenaltyExcessCharacter: 2
PenaltyReturnTypeOnItsOwnLine: 16
PointerAlignment: Left
RemoveBracesLLVM: false
ReferenceAlignment: Left
ReflowComments: true
RequiresClausePosition: WithPreceding
SeparateDefinitionBlocks: Leave
SortIncludes: Never
SpaceAfterCStyleCast: false
SpaceAfterLogicalNot: false
SpaceAfterTemplateKeyword: false
SpaceAroundPointerQualifiers: Default
SpaceBeforeAssignmentOperators: true
SpaceBeforeCaseColon: false
SpaceBeforeCpp11BracedList: true
SpaceBeforeInheritanceColon: true
SpaceBeforeParens: Custom
SpaceBeforeParensOptions:
AfterControlStatements: true
AfterFunctionDeclarationName: false
AfterFunctionDefinitionName: false
AfterOverloadedOperator: false
AfterRequiresInClause: true
BeforeNonEmptyParentheses: false
SpaceBeforeRangeBasedForLoopColon: true
SpaceBeforeSquareBrackets: false
SpaceInEmptyBlock: false
SpaceInEmptyParentheses: false
SpacesBeforeTrailingComments: 1
SpacesInAngles: Leave
SpacesInCStyleCastParentheses: false
SpacesInConditionalStatement: false
SpacesInParentheses: false
Standard: c++20
UseTab: Never
-14
View File
@@ -1,14 +0,0 @@
# This file is used to ignore files and directories from clang-format
# Ignore all files in the External directory
External/*
# SoftFloat-3e code doesn't belong to us
FEXCore/Source/Common/SoftFloat-3e/*
Source/Common/cpp-optparse/*
# Files with human-indented tables for readability - don't mess with these
FEXCore/Source/Interface/Core/X86Tables/X87Tables.cpp
FEXCore/Source/Interface/Core/X86Tables/XOPTables.cpp
FEXCore/Source/Interface/Core/X86Tables/*
-15
View File
@@ -1,15 +0,0 @@
# Since version 2.23 (released in August 2019), git-blame has a feature
# to ignore or bypass certain commits.
#
# This file contains a list of commits that are not likely what you
# are looking for in a blame, such as mass reformatting or renaming.
# You can set this file as a default ignore file for blame by running
# the following command.
#
# $ git config blame.ignoreRevsFile .git-blame-ignore-revs
# Whole tree reformat PR#3571
2b4ec88daebd35fefb5bf5c73d7fc2b4155771ed
# Second reformat to find fixed point PR#3577
905aa935f5ce344a48ef4d5edab3c31efa8d793e
@@ -37,6 +37,7 @@ If applicable, add screenshots and video to help explain your problem.
**Additional context**
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
- Is this a Vulkan game: [Yes/No/Unknown]
- If Yes, What is your Vulkan driver:
+15 -2
View File
@@ -13,6 +13,7 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
FEX_ENABLEAVX: 1
jobs:
@@ -64,7 +65,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -77,6 +78,18 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -237,7 +250,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
+15 -2
View File
@@ -20,6 +20,7 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
FEX_ENABLEAVX: 1
jobs:
@@ -71,7 +72,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -84,6 +85,18 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -171,7 +184,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
+2 -2
View File
@@ -64,7 +64,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -90,7 +90,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+2 -2
View File
@@ -74,7 +74,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -121,7 +121,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+1 -1
View File
@@ -74,7 +74,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
- name: Build
working-directory: ${{runner.workspace}}/build
-75
View File
@@ -1,75 +0,0 @@
# Inspired by LLVM's pr-code-format.yml at
# https://github.com/llvm/llvm-project/blob/main/.github/workflows/pr-code-format.yml
name: "Check code formatting"
on:
pull_request:
branches:
- main
jobs:
code_formatter:
runs-on: [self-hosted, X64]
if: github.repository == 'FEX-Emu/FEX'
steps:
- name: Fetch FEX sources
uses: actions/checkout@v4
with:
ref: ${{ github.event.pull_request.head.sha }}
- name: Checkout through merge base
uses: rmacklin/fetch-through-merge-base@v0
with:
base_ref: ${{ github.event.pull_request.base.ref }}
head_ref: ${{ github.event.pull_request.head.sha }}
deepen_length: 500
- name: Get changed files
id: changed-files
uses: tj-actions/changed-files@v39
with:
separator: ","
skip_initial_fetch: true
- name: "Listed files"
env:
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
run: |
echo "Formatting files:"
echo "$CHANGED_FILES"
- name: Check for correct clang-format version
run: clang-format --version | grep -qF '16.0.6'
- name: Check git-clang-format-16 exists
run: which git-clang-format-16
- name: Setup Python env
uses: actions/setup-python@v4
with:
python-version: '3.11'
cache: 'pip'
cache-dependency-path: './External/code-format-helper/requirements_formatting.txt'
- name: Install python dependencies
run: pip install -r ./External/code-format-helper/requirements_formatting.txt
- name: Run code formatter
env:
CLANG_FORMAT_PATH: 'git-clang-format-16'
GITHUB_PR_NUMBER: ${{ github.event.pull_request.number }}
START_REV: ${{ github.event.pull_request.base.sha }}
END_REV: ${{ github.event.pull_request.head.sha }}
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
# TODO(pmatos): Once we adopt v18, we should be able
# to take advantage of the new --diff_from_common_commit option
# explicitly in code-format-helper.py and not have to diff starting at
# the merge base.
run: |
python ./External/code-format-helper/code-format-helper.py \
--repo "FEX-emu/FEX" \
--issue-number $GITHUB_PR_NUMBER \
--start-rev $(git merge-base $START_REV $END_REV) \
--end-rev $END_REV \
--changed-files "$CHANGED_FILES"
+14 -2
View File
@@ -65,7 +65,7 @@ jobs:
# Note the current convention is to use the -S and -B options here to specify source
# and build directories, but this is only available with CMake 3.13 and higher.
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
- name: Build
working-directory: ${{runner.workspace}}/build
@@ -100,13 +100,25 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM128bit.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+6 -6
View File
@@ -15,19 +15,19 @@
path = External/tiny-json
url = https://github.com/Sonicadvance1/tiny-json.git
[submodule "External/xbyak"]
shallow = true
shallow = true
path = External/xbyak
url = https://github.com/herumi/xbyak.git
url = https://github.com/FEX-Emu/xbyak.git
[submodule "External/fex-posixtest-bins"]
shallow = true
shallow = true
path = External/fex-posixtest-bins
url = https://github.com/FEX-Emu/fex-posixtest-bins.git
[submodule "External/fex-gvisor-tests-bins"]
shallow = true
shallow = true
path = External/fex-gvisor-tests-bins
url = https://github.com/FEX-Emu/fex-gvisor-tests-bins.git
[submodule "External/fex-gcc-target-tests-bins"]
shallow = true
shallow = true
path = External/fex-gcc-target-tests-bins
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
[submodule "External/jemalloc"]
@@ -41,7 +41,7 @@
url = https://github.com/FEX-Emu/drm-headers.git
[submodule "External/xxhash"]
path = External/xxhash
url = https://github.com/Cyan4973/xxHash.git
url = https://github.com/FEX-Emu/xxHash.git
[submodule "External/Catch2"]
path = External/Catch2
url = https://github.com/catchorg/Catch2.git
+67 -12
View File
@@ -9,6 +9,7 @@ option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
option(BUILD_THUNKS "Build thunks" FALSE)
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
option(ENABLE_IWYU "Enables include what you use program" FALSE)
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
@@ -25,6 +26,7 @@ option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
@@ -120,14 +122,6 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
add_definitions(-D_M_ARM_64=1)
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
set(_M_ARM_64EC 1)
add_definitions(-D_M_ARM_64EC=1)
# Required as FEX is not allowed to lock the CRT heap lock during compilation or callbacks
set(ENABLE_JEMALLOC TRUE)
endif()
if (ENABLE_CCACHE)
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
@@ -164,6 +158,18 @@ if (NOT ENABLE_OFFLINE_TELEMETRY)
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
endif()
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
add_definitions(-DTERMUX_BUILD=1)
set(TERMUX_BUILD 1)
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
set(ENABLE_JEMALLOC FALSE)
# Termux builds can't rely on X11 packages
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
set(BUILD_FEXCONFIG FALSE)
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
@@ -226,10 +232,8 @@ endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
set(XXHASH_BUNDLED_MODE TRUE)
set(XXHASH_BUILD_XXHSUM FALSE)
set(BUILD_SHARED_LIBS OFF)
add_subdirectory(External/xxhash/cmake_unofficial/)
add_subdirectory(External/xxhash/)
include_directories(External/xxhash/)
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
@@ -346,6 +350,57 @@ if (ENABLE_IWYU)
endif()
endif()
if (ENABLE_CLANG_FORMAT)
find_program(CLANG_TIDY_EXE "clang-tidy")
if (NOT CLANG_TIDY_EXE)
message(FATAL_ERROR "Couldn't find clang-tidy")
endif()
set(CLANG_TIDY_FLAGS
"-checks=*"
"-fuchsia*"
"-bugprone-macro-parentheses"
"-clang-analyzer-core.*"
"-cppcoreguidelines-pro-type-*"
"-cppcoreguidelines-pro-bounds-array-to-pointer-decay"
"-cppcoreguidelines-pro-bounds-pointer-arithmetic"
"-cppcoreguidelines-avoid-c-arrays"
"-cppcoreguidelines-avoid-magic-numbers"
"-cppcoreguidelines-pro-bounds-constant-array-index"
"-cppcoreguidelines-no-malloc"
"-cppcoreguidelines-special-member-functions"
"-cppcoreguidelines-owning-memory"
"-cppcoreguidelines-macro-usage"
"-cppcoreguidelines-avoid-goto"
"-google-readability-function-size"
"-google-readability-namespace-comments"
"-google-readability-braces-around-statements"
"-google-build-using-namespace"
"-hicpp-*"
"-llvm-namespace-comment"
"-llvm-include-order" # Messes up with case sensitivity
"-llvmlibc-*"
"-misc-unused-parameters"
"-modernize-loop-convert"
"-modernize-use-auto"
"-modernize-avoid-c-arrays"
"-modernize-use-nodiscard"
"readability-*"
"-readability-function-size"
"-readability-implicit-bool-conversion"
"-readability-braces-around-statements"
"-readability-else-after-return"
"-readability-magic-numbers"
"-readability-named-parameter"
"-readability-uppercase-literal-suffix"
"-cert-err34-c"
"-cert-err58-cpp"
"-bugprone-exception-escape"
)
string(REPLACE ";" "," CLANG_TIDY_FLAGS "${CLANG_TIDY_FLAGS}")
set(CMAKE_CXX_CLANG_TIDY ${CLANG_TIDY_EXE} "${CLANG_TIDY_FLAGS}")
endif()
add_compile_options(-Wall)
configure_file(
+2 -3
View File
@@ -1,6 +1,5 @@
{
"Comment": "Bypasses libGL's glX and instead sends GLX requests directly via xcb",
"ThunksDB": {
"GL": 0
"Config": {
"AdditionalArguments": "--no-sandbox"
}
}
+9
View File
@@ -2,6 +2,9 @@
"DB": {
"GL": {
"Library" : "libGL-guest.so",
"Depends": [
"X11"
],
"Overlay": [
"@PREFIX_LIB@/libGL.so",
"@PREFIX_LIB@/libGL.so.1",
@@ -30,10 +33,16 @@
},
"Vulkan": {
"Library": "libvulkan-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"@PREFIX_LIB@/libvulkan.so",
"@PREFIX_LIB@/libvulkan.so.1",
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
],
"Comment": [
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
]
},
"xcb": {
+4 -5
View File
@@ -3,12 +3,11 @@ FROM ubuntu:20.04 as builder
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
clang-10 llvm-10 nasm ninja-build pkg-config \
libcap-dev libglfw3-dev libepoxy-dev python3-dev libsdl2-dev \
python3 linux-headers-generic \
git
clang-10 llvm-10 nasm ninja-build \
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
python3 linux-headers-generic
RUN git clone --recurse-submodules https://github.com/FEX-Emu/FEX.git
COPY . /opt/FEX
CMD [ "mkdir /opt/FEX/build" ]
+1 -1
-394
View File
@@ -1,394 +0,0 @@
#!/usr/bin/env python3
#
# ====- code-format-helper, runs code formatters from the ci or in a hook --*- python -*--==#
#
# Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
# See https://llvm.org/LICENSE.txt for license information.
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#
# ==--------------------------------------------------------------------------------------==#
import argparse
import os
import subprocess
import sys
from typing import List, Optional
"""
This script is run by GitHub actions to ensure that the code in PR's conform to
the coding style of LLVM. It can also be installed as a pre-commit git hook to
check the coding style before submitting it. The canonical source of this script
is in the LLVM source tree under llvm/utils/git.
For C/C++ code it uses clang-format and for Python code it uses darker (which
in turn invokes black).
You can learn more about the LLVM coding style on llvm.org:
https://llvm.org/docs/CodingStandards.html
You can install this script as a git hook by symlinking it to the .git/hooks
directory:
ln -s $(pwd)/llvm/utils/git/code-format-helper.py .git/hooks/pre-commit
You can control the exact path to clang-format or darker with the following
environment variables: $CLANG_FORMAT_PATH and $DARKER_FORMAT_PATH.
"""
class FormatArgs:
start_rev: str = None
end_rev: str = None
repo: str = None
changed_files: List[str] = []
token: str = None
verbose: bool = True
issue_number: int = 0
write_comment_to_file: str = None
def __init__(self, args: argparse.Namespace = None) -> None:
if not args is None:
self.start_rev = args.start_rev
self.end_rev = args.end_rev
self.repo = args.repo
self.token = args.token
self.changed_files = args.changed_files
self.issue_number = args.issue_number
self.write_comment_to_file = args.write_comment_to_file
class FormatHelper:
COMMENT_TAG = "<!--CODE FORMAT COMMENT: {fmt}-->"
name: str
friendly_name: str
comment: dict = None
@property
def comment_tag(self) -> str:
return self.COMMENT_TAG.replace("fmt", self.name)
@property
def instructions(self) -> str:
raise NotImplementedError()
def has_tool(self) -> bool:
raise NotImplementedError()
def format_run(self, changed_files: List[str], args: FormatArgs) -> Optional[str]:
raise NotImplementedError()
def pr_comment_text_for_diff(self, diff: str) -> str:
return f"""
:warning: {self.friendly_name}, {self.name} found issues in your code. :warning:
<details>
<summary>
You can test this locally with the following command:
</summary>
``````````bash
{self.instructions}
``````````
</details>
<details>
<summary>
View the diff from {self.name} here.
</summary>
``````````diff
{diff}
``````````
</details>
"""
# TODO: any type should be replaced with the correct github type, but it requires refactoring to
# not require the github module to be installed everywhere.
def find_comment(self, pr: any) -> any:
for comment in pr.as_issue().get_comments():
if self.comment_tag in comment.body:
return comment
return None
def update_pr(self, comment_text: str, args: FormatArgs, create_new: bool) -> None:
import github
from github import IssueComment, PullRequest
repo = github.Github(args.token).get_repo(args.repo)
pr = repo.get_issue(args.issue_number).as_pull_request()
comment_text = self.comment_tag + "\n\n" + comment_text
existing_comment = self.find_comment(pr)
if args.write_comment_to_file:
if create_new or existing_comment:
self.comment = {"body": comment_text}
if existing_comment:
self.comment["id"] = existing_comment.id
return
if existing_comment:
existing_comment.edit(comment_text)
elif create_new:
pr.as_issue().create_comment(comment_text)
def run(self, changed_files: List[str], args: FormatArgs) -> bool:
changed_files = [arg for arg in changed_files if "third-party" not in arg]
diff = self.format_run(changed_files, args)
should_update_gh = args.token is not None and args.repo is not None
if diff is None:
if should_update_gh:
comment_text = (
":white_check_mark: With the latest revision "
f"this PR passed the {self.friendly_name}."
)
self.update_pr(comment_text, args, create_new=False)
return True
elif len(diff) > 0:
if should_update_gh:
comment_text = self.pr_comment_text_for_diff(diff)
self.update_pr(comment_text, args, create_new=True)
else:
print(
f"Warning: {self.friendly_name}, {self.name} detected "
"some issues with your code formatting..."
)
return False
else:
# The formatter failed but didn't output a diff (e.g. some sort of
# infrastructure failure).
comment_text = (
f":warning: The {self.friendly_name} failed without printing "
"a diff. Check the logs for stderr output. :warning:"
)
self.update_pr(comment_text, args, create_new=False)
return False
class ClangFormatHelper(FormatHelper):
name = "clang-format"
friendly_name = "C/C++ code formatter"
@property
def instructions(self) -> str:
return " ".join(self.cf_cmd)
def should_include_extensionless_file(self, path: str) -> bool:
return path.startswith("libcxx/include")
def filter_changed_files(self, changed_files: List[str]) -> List[str]:
filtered_files = []
for path in changed_files:
_, ext = os.path.splitext(path)
if ext in (".cpp", ".c", ".h", ".hpp", ".hxx", ".cxx", ".inc", ".cppm"):
filtered_files.append(path)
elif ext == "" and self.should_include_extensionless_file(path):
filtered_files.append(path)
return filtered_files
@property
def clang_fmt_path(self) -> str:
if "CLANG_FORMAT_PATH" in os.environ:
return os.environ["CLANG_FORMAT_PATH"]
return "git-clang-format"
def has_tool(self) -> bool:
cmd = [self.clang_fmt_path, "-h"]
proc = None
try:
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
except:
return False
return proc.returncode == 0
def format_run(self, changed_files: List[str], args: FormatArgs) -> Optional[str]:
cpp_files = self.filter_changed_files(changed_files)
if not cpp_files:
return None
cf_cmd = [self.clang_fmt_path, "--diff"]
if args.start_rev and args.end_rev:
cf_cmd.append(args.start_rev)
cf_cmd.append(args.end_rev)
cf_cmd.append("--")
cf_cmd += cpp_files
if args.verbose:
print(f"Running: {' '.join(cf_cmd)}")
self.cf_cmd = cf_cmd
proc = subprocess.run(cf_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
sys.stdout.write(proc.stderr.decode("utf-8"))
if proc.returncode != 0:
# formatting needed, or the command otherwise failed
if args.verbose:
print(f"error: {self.name} exited with code {proc.returncode}")
# Print the diff in the log so that it is viewable there
print(proc.stdout.decode("utf-8"))
return proc.stdout.decode("utf-8")
else:
return None
class DarkerFormatHelper(FormatHelper):
name = "darker"
friendly_name = "Python code formatter"
@property
def instructions(self) -> str:
return " ".join(self.darker_cmd)
def filter_changed_files(self, changed_files: List[str]) -> List[str]:
filtered_files = []
for path in changed_files:
name, ext = os.path.splitext(path)
if ext == ".py":
filtered_files.append(path)
return filtered_files
@property
def darker_fmt_path(self) -> str:
if "DARKER_FORMAT_PATH" in os.environ:
return os.environ["DARKER_FORMAT_PATH"]
return "darker"
def has_tool(self) -> bool:
cmd = [self.darker_fmt_path, "--version"]
proc = None
try:
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
except:
return False
return proc.returncode == 0
def format_run(self, changed_files: List[str], args: FormatArgs) -> Optional[str]:
py_files = self.filter_changed_files(changed_files)
if not py_files:
return None
darker_cmd = [
self.darker_fmt_path,
"--check",
"--diff",
]
if args.start_rev and args.end_rev:
darker_cmd += ["-r", f"{args.start_rev}...{args.end_rev}"]
darker_cmd += py_files
if args.verbose:
print(f"Running: {' '.join(darker_cmd)}")
self.darker_cmd = darker_cmd
proc = subprocess.run(
darker_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE
)
if args.verbose:
sys.stdout.write(proc.stderr.decode("utf-8"))
if proc.returncode != 0:
# formatting needed, or the command otherwise failed
if args.verbose:
print(f"error: {self.name} exited with code {proc.returncode}")
# Print the diff in the log so that it is viewable there
print(proc.stdout.decode("utf-8"))
return proc.stdout.decode("utf-8")
else:
sys.stdout.write(proc.stdout.decode("utf-8"))
return None
ALL_FORMATTERS = (DarkerFormatHelper(), ClangFormatHelper())
def hook_main():
# fill out args
args = FormatArgs()
args.verbose = False
# find the changed files
cmd = ["git", "diff", "--cached", "--name-only", "--diff-filter=d"]
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
output = proc.stdout.decode("utf-8")
for line in output.splitlines():
args.changed_files.append(line)
failed_fmts = []
for fmt in ALL_FORMATTERS:
if fmt.has_tool():
if not fmt.run(args.changed_files, args):
failed_fmts.append(fmt.name)
if fmt.comment:
comments.append(fmt.comment)
else:
print(f"Couldn't find {fmt.name}, can't check " + fmt.friendly_name.lower())
if len(failed_fmts) > 0:
sys.exit(1)
sys.exit(0)
if __name__ == "__main__":
script_path = os.path.abspath(__file__)
if ".git/hooks" in script_path:
hook_main()
sys.exit(0)
parser = argparse.ArgumentParser()
parser.add_argument(
"--token", type=str, required=False, help="GitHub authentication token"
)
parser.add_argument(
"--repo",
type=str,
default=os.getenv("GITHUB_REPOSITORY", "llvm/llvm-project"),
help="The GitHub repository that we are working with in the form of <owner>/<repo> (e.g. llvm/llvm-project)",
)
parser.add_argument("--issue-number", type=int, required=True)
parser.add_argument(
"--start-rev",
type=str,
required=True,
help="Compute changes from this revision.",
)
parser.add_argument(
"--end-rev", type=str, required=True, help="Compute changes to this revision"
)
parser.add_argument(
"--changed-files",
type=str,
help="Comma separated list of files that has been changed",
)
parser.add_argument(
"--write-comment-to-file",
type=str,
help="Don't post comments on the PR, instead write the comments and metadata a file",
)
args = FormatArgs(parser.parse_args())
changed_files = []
if args.changed_files:
changed_files = args.changed_files.split(",")
failed_formatters = []
comments = []
for fmt in ALL_FORMATTERS:
if not fmt.run(changed_files, args):
failed_formatters.append(fmt.name)
if fmt.comment:
comments.append(fmt.comment)
if len(comments):
with open(args.write_comment_to_file, "w") as f:
import json
json.dump(comments, f)
if len(failed_formatters) > 0:
print(f"error: some formatters failed: {' '.join(failed_formatters)}")
sys.exit(1)
-52
View File
@@ -1,52 +0,0 @@
#
# This file is autogenerated by pip-compile with Python 3.11
# by the following command:
#
# pip-compile --output-file=llvm/utils/git/requirements_formatting.txt llvm/utils/git/requirements_formatting.txt.in
#
black==23.9.1
# via
# -r llvm/utils/git/requirements_formatting.txt.in
# darker
certifi==2023.7.22
# via requests
cffi==1.15.1
# via
# cryptography
# pynacl
charset-normalizer==3.2.0
# via requests
click==8.1.7
# via black
cryptography==41.0.3
# via pyjwt
darker==1.7.2
# via -r llvm/utils/git/requirements_formatting.txt.in
deprecated==1.2.14
# via pygithub
idna==3.4
# via requests
mypy-extensions==1.0.0
# via black
packaging==23.1
# via black
pathspec==0.11.2
# via black
platformdirs==3.10.0
# via black
pycparser==2.21
# via cffi
pygithub==1.59.1
# via -r llvm/utils/git/requirements_formatting.txt.in
pyjwt[crypto]==2.8.0
# via pygithub
pynacl==1.5.0
# via pygithub
requests==2.31.0
# via pygithub
toml==0.10.2
# via darker
urllib3==2.0.4
# via requests
wrapt==1.15.0
# via deprecated
+1 -1
+1 -1
+1 -1
+5 -9
View File
@@ -13,6 +13,8 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
set(_M_ARM_64 1)
endif()
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
include(CheckPIESupported)
@@ -28,21 +30,15 @@ set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
check_cxx_source_compiles(
"
__attribute__((preserve_all))
int Testy(int a, int b, int c, int d, int e, int f) {
return a + b + c + d + e + f;
void Testy() {
}
int main() {
return Testy(0, 1, 2, 3, 4, 5);
return 0;
}"
HAS_CLANG_PRESERVE_ALL)
unset(CMAKE_REQUIRED_FLAGS)
if (HAS_CLANG_PRESERVE_ALL)
if (MINGW_BUILD)
message(STATUS "Ignoring broken clang::preserve_all support")
set(HAS_CLANG_PRESERVE_ALL FALSE)
else()
message(STATUS "Has clang::preserve_all")
endif()
message(STATUS "Has clang::preserve_all")
endif ()
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
-21
View File
@@ -441,24 +441,6 @@ def print_parse_envloader_options(options):
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_jsonloader_options(options):
output_argloader.write("#ifdef JSONLOADER\n")
output_argloader.write("#undef JSONLOADER\n")
output_argloader.write("if (false) {}\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
value_type = op_vals["Type"]
if (value_type == "strenum"):
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
output_argloader.write("Set(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
output_argloader.write("}\n")
output_argloader.write("else {{\n".format(op_key))
output_argloader.write("Set(KeyOption, ConfigString);\n")
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_enum_options(options):
output_argloader.write("#ifdef ENUMDEFINES\n")
output_argloader.write("#undef ENUMDEFINES\n")
@@ -574,9 +556,6 @@ print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
# Generate json loader code
print_parse_jsonloader_options(options);
# Generate enum variable options
print_parse_enum_options(options);
+1 -1
View File
@@ -652,7 +652,7 @@ def print_ir_allocator_helpers():
# Save NZCV if needed before clobbering NZCV
if op.ImplicitFlagClobber:
output_file.write("\t\tSaveNZCV(IROps::OP_{});".format(op.Name.upper()))
output_file.write("\t\tSaveNZCV();")
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
+5 -5
View File
@@ -7,7 +7,6 @@ set (FEXCORE_BASE_SRCS
Utils/FileLoading.cpp
Utils/ForcedAssert.cpp
Utils/LogManager.cpp
Utils/SpinWaitLock.cpp
)
if (NOT MINGW_BUILD)
@@ -91,6 +90,7 @@ set (SRCS
Interface/Core/CPUBackend.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/GdbServer.cpp
Interface/Core/HostFeatures.cpp
Interface/Core/ObjectCache/JobHandling.cpp
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
@@ -101,7 +101,9 @@ set (SRCS
Interface/Core/OpcodeDispatcher/X87.cpp
Interface/Core/OpcodeDispatcher/X87F64.cpp
Interface/Core/OpcodeDispatcher.cpp
Interface/Core/SignalDelegator.cpp
Interface/Core/X86Tables.cpp
Interface/Core/X86DebugInfo.cpp
Interface/Core/X86HelperGen.cpp
Interface/Core/ArchHelpers/Arm64Emitter.cpp
Interface/Core/Dispatcher/Dispatcher.cpp
@@ -150,6 +152,7 @@ set (SRCS
Interface/IR/Passes/DeadStoreElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/InlineCallOptimization.cpp
Utils/NetStream.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/Profiler.cpp
@@ -194,15 +197,12 @@ endif()
# Some defines for the softfloat library
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils)
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
if (NOT MINGW_BUILD)
list (APPEND LIBS dl)
else()
list (APPEND LIBS synchronization)
if (_M_ARM_64EC)
list (APPEND LIBS kernelbase)
endif()
endif()
if (ENABLE_JEMALLOC)
+5 -4
View File
@@ -18,7 +18,7 @@ struct BitSet final {
constexpr static size_t MinimumSize = sizeof(ElementType);
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
ElementType* Memory;
ElementType *Memory;
void Allocate(size_t Elements) {
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
@@ -62,10 +62,11 @@ struct BitSetView final {
constexpr static size_t MinimumSize = sizeof(ElementType);
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
ElementType* Memory;
ElementType *Memory;
void GetView(BitSet<T>& Set, uint64_t ElementOffset) {
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0,
"Bitset view offset needs to be aligned to size of backing element");
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
}
+122 -131
View File
@@ -7,142 +7,133 @@
#include <unistd.h>
namespace FEXCore {
JITSymbols::JITSymbols() {}
JITSymbols::~JITSymbols() {
if (fd != -1) {
close(fd);
}
}
void JITSymbols::InitFile() {
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
#ifdef __ANDROID__
// Android simpleperf looks in /data/local/tmp instead of /tmp
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
#else
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
#endif
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
}
void JITSymbols::RegisterNamedRegion(const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) {
return;
JITSymbols::JITSymbols() {
}
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::RegisterJITSpace(const void* HostAddr, uint32_t CodeSize) {
if (fd == -1) {
return;
}
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
// Buffered JIT symbols.
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (fd == -1) {
return;
}
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, GuestAddr, CodeSize);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (fd == -1) {
return;
}
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult =
fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, CodeSize, Name, Offset);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) {
return;
}
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}\n", HostAddr, CodeSize, Name);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
RegisterNamedRegion(Buffer, HostAddr, CodeSize, Name);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite) {
auto Now = std::chrono::steady_clock::now();
if (!ForceWrite) {
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) && Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
// Still buffering, no need to write.
return;
JITSymbols::~JITSymbols() {
if (fd != -1) {
close(fd);
}
}
Buffer->LastWrite = Now;
auto Result = write(fd, Buffer->Buffer, Buffer->Offset);
if (Result == -1 && errno == EBADF) {
fd = -1;
void JITSymbols::InitFile() {
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
#ifdef __ANDROID__
// Android simpleperf looks in /data/local/tmp instead of /tmp
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
#else
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
#endif
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
}
Buffer->Offset = 0;
}
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
// Buffered JIT symbols.
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, GuestAddr, CodeSize);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, CodeSize, Name, Offset);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}\n", HostAddr, CodeSize, Name);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
RegisterNamedRegion(Buffer, HostAddr, CodeSize, Name);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite) {
auto Now = std::chrono::steady_clock::now();
if (!ForceWrite) {
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) &&
Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
// Still buffering, no need to write.
return;
}
}
Buffer->LastWrite = Now;
auto Result = write(fd, Buffer->Buffer, Buffer->Offset);
if (Result == -1 && errno == EBADF) {
fd = -1;
}
Buffer->Offset = 0;
}
} // namespace FEXCore
+8 -8
View File
@@ -17,20 +17,20 @@ public:
~JITSymbols();
void InitFile();
void RegisterNamedRegion(const void* HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void* HostAddr, uint32_t CodeSize);
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
// Allocate JIT buffer.
static fextl::unique_ptr<Core::JITSymbolBuffer> AllocateBuffer() {
return fextl::make_unique<Core::JITSymbolBuffer>();
}
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name);
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name);
private:
int fd {-1};
void WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite = false);
int fd{-1};
void WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite = false);
};
} // namespace FEXCore
}
+123 -93
View File
@@ -1,10 +1,10 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <cmath>
#include <cstring>
@@ -45,13 +45,13 @@ struct FEX_PACKED X80SoftFloat {
uint16_t Exponent : 15;
uint16_t Sign : 1;
X80SoftFloat() {
memset(this, 0, sizeof(*this));
}
X80SoftFloat() { memset(this, 0, sizeof(*this)); }
X80SoftFloat(uint16_t _Sign, uint16_t _Exponent, uint64_t _Significand)
: Significand {_Significand}
, Exponent {_Exponent}
, Sign {_Sign} {}
, Sign {_Sign}
{
}
fextl::string str() const {
fextl::ostringstream string;
@@ -63,19 +63,21 @@ struct FEX_PACKED X80SoftFloat {
}
// Ops
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FADD(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
faddp;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -83,19 +85,21 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSUB(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fsubp;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -103,19 +107,21 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FMUL(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fmulp;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -123,19 +129,21 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FDIV(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fdivp;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -143,10 +151,11 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
@@ -154,9 +163,10 @@ struct FEX_PACKED X80SoftFloat {
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -164,10 +174,11 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM1(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
@@ -175,9 +186,10 @@ struct FEX_PACKED X80SoftFloat {
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -185,27 +197,30 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(const X80SoftFloat& lhs, uint_fast8_t RoundMode) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
return extF80_roundToInt(lhs, RoundMode, false);
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FXTRACT_SIG(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fxtract;
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st", "st(1)");
return Result;
#else
@@ -216,19 +231,20 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FXTRACT_EXP(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fxtract;
ffreep %%st(0);
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st", "st(1)");
return Result;
#else
@@ -237,17 +253,19 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static void FCMP(const X80SoftFloat& lhs, const X80SoftFloat& rhs, bool* eq, bool* lt, bool* nan) {
FEXCORE_PRESERVE_ALL_ATTR
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
*eq = extF80_eq(lhs, rhs);
*lt = extF80_lt(lhs, rhs);
*nan = IsNan(lhs) || IsNan(rhs);
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSCALE(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
@@ -255,9 +273,10 @@ struct FEX_PACKED X80SoftFloat {
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -270,19 +289,20 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat F2XM1(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
f2xm1; # st0 = 2^st(0) - 1
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -293,20 +313,22 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FYL2X(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st(1)
fldt %[lhs]; # st(0)
fyl2x; # st(1) * log2l(st(0))
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -317,20 +339,22 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FATAN(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs];
fldt %[rhs];
fpatan;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -341,20 +365,21 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FTAN(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fptan;
ffreep %%st(0);
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -364,19 +389,20 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSIN(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fsin;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -386,19 +412,20 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FCOS(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fcos;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -408,18 +435,19 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSQRT(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fsqrt;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -443,7 +471,7 @@ struct FEX_PACKED X80SoftFloat {
const float128_t Result = extF80_to_f128(*this);
return FEXCore::BitCast<BIGFLOAT>(Result);
#else
BIGFLOAT result {};
BIGFLOAT result{};
memcpy(&result, this, sizeof(result));
return result;
#endif
@@ -542,17 +570,19 @@ struct FEX_PACKED X80SoftFloat {
}
operator extFloat80_t() const {
extFloat80_t Result {};
extFloat80_t Result{};
Result.signif = Significand;
Result.signExp = Exponent | (Sign << 15);
return Result;
}
static bool IsNan(const X80SoftFloat& lhs) {
return (lhs.Exponent == 0x7FFF) && (lhs.Significand & IntegerBit) && (lhs.Significand & Bottom62Significand);
static bool IsNan(X80SoftFloat const &lhs) {
return (lhs.Exponent == 0x7FFF) &&
(lhs.Significand & IntegerBit) &&
(lhs.Significand & Bottom62Significand);
}
static bool SignBit(const X80SoftFloat& lhs) {
static bool SignBit(X80SoftFloat const &lhs) {
return lhs.Sign;
}
+34 -41
View File
@@ -7,51 +7,44 @@
#include <optional>
namespace FEXCore::StrConv {
[[maybe_unused]]
static bool Conv(std::string_view Value, bool* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, bool *Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint8_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, uint8_t *Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint16_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, uint16_t *Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint32_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, uint32_t *Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, int32_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, int32_t *Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint64_t* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
[[maybe_unused]]
static bool Conv(std::string_view Value, T* Result) {
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, uint64_t *Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
template <typename T,
typename = std::enable_if<std::is_enum<T>::value, T>>
[[maybe_unused]] static bool Conv(std::string_view Value, T *Result) {
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, fextl::string* Result) {
*Result = Value;
return true;
[[maybe_unused]] static bool Conv(std::string_view Value, fextl::string *Result) {
*Result = Value;
return true;
}
}
} // namespace FEXCore::StrConv
+427 -409
View File
@@ -29,7 +29,7 @@
#include <utility>
namespace FEXCore::Context {
class Context;
class Context;
}
namespace FEXCore::Config {
@@ -40,464 +40,482 @@ namespace DefaultValues {
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
#include <FEXCore/Config/ConfigValues.inl>
} // namespace DefaultValues
enum Paths {
PATH_DATA_DIR = 0,
PATH_CONFIG_DIR_LOCAL,
PATH_CONFIG_DIR_GLOBAL,
PATH_CONFIG_FILE_LOCAL,
PATH_CONFIG_FILE_GLOBAL,
PATH_CONFIG_TELEMETRY_FOLDER,
PATH_LAST,
};
static std::array<fextl::string, Paths::PATH_LAST> Paths;
void SetDataDirectory(const std::string_view Path) {
Paths[PATH_DATA_DIR] = Path;
}
void SetConfigDirectory(const std::string_view Path, bool Global) {
Paths[PATH_CONFIG_DIR_LOCAL + Global] = Path;
}
enum Paths {
PATH_DATA_DIR = 0,
PATH_CONFIG_DIR_LOCAL,
PATH_CONFIG_DIR_GLOBAL,
PATH_CONFIG_FILE_LOCAL,
PATH_CONFIG_FILE_GLOBAL,
PATH_LAST,
};
static std::array<fextl::string, Paths::PATH_LAST> Paths;
void SetConfigFileLocation(const std::string_view Path, bool Global) {
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
}
void SetDataDirectory(const std::string_view Path) {
Paths[PATH_DATA_DIR] = Path;
}
const fextl::string& GetTelemetryDirectory() {
auto& Path = Paths[PATH_CONFIG_TELEMETRY_FOLDER];
if (Path.empty()) {
FEX_CONFIG_OPT(TelemetryDirectory, TELEMETRYDIRECTORY);
if (!TelemetryDirectory().empty()) {
Path = TelemetryDirectory;
Path += "/";
} else {
Path = Config::GetDataDirectory() + "Telemetry/";
void SetConfigDirectory(const std::string_view Path, bool Global) {
Paths[PATH_CONFIG_DIR_LOCAL + Global] = Path;
}
void SetConfigFileLocation(const std::string_view Path, bool Global) {
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
}
fextl::string const& GetDataDirectory() {
return Paths[PATH_DATA_DIR];
}
fextl::string const& GetConfigDirectory(bool Global) {
return Paths[PATH_CONFIG_DIR_LOCAL + Global];
}
fextl::string const& GetConfigFileLocation(bool Global) {
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
}
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
fextl::string ConfigFile = GetConfigDirectory(Global);
if (!Global &&
!FHU::Filesystem::Exists(ConfigFile) &&
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
// Let's go local in this case
return fextl::fmt::format("./{}.json", Program);
}
}
return Path;
}
ConfigFile += "AppConfig/";
const fextl::string& GetDataDirectory() {
return Paths[PATH_DATA_DIR];
}
const fextl::string& GetConfigDirectory(bool Global) {
return Paths[PATH_CONFIG_DIR_LOCAL + Global];
}
const fextl::string& GetConfigFileLocation(bool Global) {
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
}
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
fextl::string ConfigFile = GetConfigDirectory(Global);
if (!Global && !FHU::Filesystem::Exists(ConfigFile) && !FHU::Filesystem::CreateDirectories(ConfigFile)) {
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
// Let's go local in this case
return fextl::fmt::format("./{}.json", Program);
}
ConfigFile += "AppConfig/";
// Attempt to create the local folder if it doesn't exist
if (!Global && !FHU::Filesystem::Exists(ConfigFile) && !FHU::Filesystem::CreateDirectories(ConfigFile)) {
// Let's go local in this case
return fextl::fmt::format("./{}.json", Program);
}
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
}
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, uint64_t Config) {}
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, const fextl::string& Config) {}
uint64_t GetConfig(FEXCore::Context::Context* CTX, ConfigOption Option) {
return 0;
}
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
static FEXCore::Config::Layer* Meta {};
constexpr std::array<FEXCore::Config::LayerType, 10> LoadOrder = {
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN, FEXCore::Config::LayerType::LAYER_MAIN,
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP, FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP, FEXCore::Config::LayerType::LAYER_LOCAL_APP,
FEXCore::Config::LayerType::LAYER_ARGUMENTS, FEXCore::Config::LayerType::LAYER_USER_OVERRIDE,
FEXCore::Config::LayerType::LAYER_ENVIRONMENT, FEXCore::Config::LayerType::LAYER_TOP};
Layer::Layer(const LayerType _Type)
: Type {_Type} {}
Layer::~Layer() {}
class MetaLayer final : public FEXCore::Config::Layer {
public:
MetaLayer(const LayerType _Type)
: FEXCore::Config::Layer(_Type) {}
~MetaLayer() {}
void Load();
private:
void MergeConfigMap(const LayerOptions& Options);
void MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value);
};
void MetaLayer::Load() {
OptionMap.clear();
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
auto it = ConfigLayers.find(*CurrentLayer);
if (it != ConfigLayers.end() && *CurrentLayer != Type) {
// Merge this layer's options to this layer
MergeConfigMap(it->second->GetOptionMap());
// Attempt to create the local folder if it doesn't exist
if (!Global &&
!FHU::Filesystem::Exists(ConfigFile) &&
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
// Let's go local in this case
return fextl::fmt::format("./{}.json", Program);
}
}
}
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value) {
// Environment variables need a bit of additional work
// We want to merge the arrays rather than overwrite entirely
auto MetaEnvironment = OptionMap.find(Option);
if (MetaEnvironment == OptionMap.end()) {
// Doesn't exist, just insert
OptionMap.insert_or_assign(Option, Value);
return;
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
}
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
const auto AddToMap = [&LookupMap](const FEXCore::Config::LayerValue& Value) {
for (const auto& EnvVar : Value) {
const auto ItEq = EnvVar.find_first_of('=');
if (ItEq == fextl::string::npos) {
// Broken environment variable
// Skip
continue;
}
auto Key = fextl::string(EnvVar.begin(), EnvVar.begin() + ItEq);
auto Value = fextl::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
}
// Add the key to the map, overwriting whatever previous value was there
LookupMap.insert_or_assign(std::move(Key), std::move(Value));
}
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, fextl::string const &Config) {
}
uint64_t GetConfig(FEXCore::Context::Context *CTX, ConfigOption Option) {
return 0;
}
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
static FEXCore::Config::Layer *Meta{};
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
FEXCore::Config::LayerType::LAYER_MAIN,
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP,
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
FEXCore::Config::LayerType::LAYER_ARGUMENTS,
FEXCore::Config::LayerType::LAYER_ENVIRONMENT,
FEXCore::Config::LayerType::LAYER_TOP
};
AddToMap(MetaEnvironment->second);
AddToMap(Value);
// Now with the two layers merged in the map
// Add all the values to the option
Erase(Option);
for (auto& Val : LookupMap) {
// Set will emplace multiple options in to its list
Set(Option, Val.first + "=" + Val.second);
Layer::Layer(const LayerType _Type)
: Type {_Type} {
}
}
void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
// Insert this layer's options, overlaying previous options that exist here
for (auto& it : Options) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
MergeEnvironmentVariables(it.first, it.second);
} else {
OptionMap.insert_or_assign(it.first, it.second);
Layer::~Layer() {
}
class MetaLayer final : public FEXCore::Config::Layer {
public:
MetaLayer(const LayerType _Type)
: FEXCore::Config::Layer (_Type) {
}
}
}
void Initialize() {
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
Meta = ConfigLayers.begin()->second.get();
}
void Shutdown() {
ConfigLayers.clear();
Meta = nullptr;
}
void Load() {
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
auto it = ConfigLayers.find(*CurrentLayer);
if (it != ConfigLayers.end()) {
it->second->Load();
~MetaLayer() {
}
}
}
void Load();
fextl::string ExpandPath(const fextl::string& ContainerPrefix, fextl::string PathName) {
if (PathName.empty()) {
return {};
private:
void MergeConfigMap(const LayerOptions &Options);
void MergeEnvironmentVariables(ConfigOption const &Option, LayerValue const &Value);
};
void MetaLayer::Load() {
OptionMap.clear();
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
auto it = ConfigLayers.find(*CurrentLayer);
if (it != ConfigLayers.end() && *CurrentLayer != Type) {
// Merge this layer's options to this layer
MergeConfigMap(it->second->GetOptionMap());
}
}
}
// Expand home if it exists
if (FHU::Filesystem::IsRelative(PathName)) {
fextl::string Home = getenv("HOME") ?: "";
// Home expansion only works if it is the first character
// This matches bash behaviour
if (PathName.at(0) == '~') {
PathName.replace(0, 1, Home);
return PathName;
void MetaLayer::MergeEnvironmentVariables(ConfigOption const &Option, LayerValue const &Value) {
// Environment variables need a bit of additional work
// We want to merge the arrays rather than overwrite entirely
auto MetaEnvironment = OptionMap.find(Option);
if (MetaEnvironment == OptionMap.end()) {
// Doesn't exist, just insert
OptionMap.insert_or_assign(Option, Value);
return;
}
// Expand relative path to absolute
char ExistsTempPath[PATH_MAX];
char* RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
if (RealPath) {
PathName = RealPath;
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
const auto AddToMap = [&LookupMap](FEXCore::Config::LayerValue const &Value) {
for (const auto &EnvVar : Value) {
const auto ItEq = EnvVar.find_first_of('=');
if (ItEq == fextl::string::npos) {
// Broken environment variable
// Skip
continue;
}
auto Key = fextl::string(EnvVar.begin(), EnvVar.begin() + ItEq);
auto Value = fextl::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
// Add the key to the map, overwriting whatever previous value was there
LookupMap.insert_or_assign(std::move(Key), std::move(Value));
}
};
AddToMap(MetaEnvironment->second);
AddToMap(Value);
// Now with the two layers merged in the map
// Add all the values to the option
Erase(Option);
for (auto &Val : LookupMap) {
// Set will emplace multiple options in to its list
Set(Option, Val.first + "=" + Val.second);
}
}
void MetaLayer::MergeConfigMap(const LayerOptions &Options) {
// Insert this layer's options, overlaying previous options that exist here
for (auto &it : Options) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV ||
it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
MergeEnvironmentVariables(it.first, it.second);
}
else {
OptionMap.insert_or_assign(it.first, it.second);
}
}
}
void Initialize() {
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
Meta = ConfigLayers.begin()->second.get();
}
void Shutdown() {
ConfigLayers.clear();
Meta = nullptr;
}
void Load() {
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
auto it = ConfigLayers.find(*CurrentLayer);
if (it != ConfigLayers.end()) {
it->second->Load();
}
}
}
fextl::string ExpandPath(fextl::string const &ContainerPrefix, fextl::string PathName) {
if (PathName.empty()) {
return {};
}
// Only return if it exists
if (FHU::Filesystem::Exists(PathName)) {
return PathName;
// Expand home if it exists
if (FHU::Filesystem::IsRelative(PathName)) {
fextl::string Home = getenv("HOME") ?: "";
// Home expansion only works if it is the first character
// This matches bash behaviour
if (PathName.at(0) == '~') {
PathName.replace(0, 1, Home);
return PathName;
}
// Expand relative path to absolute
char ExistsTempPath[PATH_MAX];
char *RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
if (RealPath) {
PathName = RealPath;
}
// Only return if it exists
if (FHU::Filesystem::Exists(PathName)) {
return PathName;
}
}
} else {
// If the containerprefix and pathname isn't empty
// Then we check if the pathname exists in our current namespace
// If the path DOESN'T exist but DOES exist with the prefix applied
// then redirect to the prefix
//
// This might not be expected behaviour for some edge cases but since
// all paths aren't mounted inside the container, then it'll be fine
//
// Main catch case for this is the default thunk install folders
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
if (!ContainerPrefix.empty() && !PathName.empty()) {
if (!FHU::Filesystem::Exists(PathName)) {
auto ContainerPath = ContainerPrefix + PathName;
if (FHU::Filesystem::Exists(ContainerPath)) {
return ContainerPath;
else {
// If the containerprefix and pathname isn't empty
// Then we check if the pathname exists in our current namespace
// If the path DOESN'T exist but DOES exist with the prefix applied
// then redirect to the prefix
//
// This might not be expected behaviour for some edge cases but since
// all paths aren't mounted inside the container, then it'll be fine
//
// Main catch case for this is the default thunk install folders
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
if (!ContainerPrefix.empty() && !PathName.empty()) {
if (!FHU::Filesystem::Exists(PathName)) {
auto ContainerPath = ContainerPrefix + PathName;
if (FHU::Filesystem::Exists(ContainerPath)) {
return ContainerPath;
}
}
}
}
return {};
}
return {};
}
constexpr char ContainerManager[] = "/run/host/container-manager";
constexpr char ContainerManager[] = "/run/host/container-manager";
fextl::string FindContainer() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::vector<char> Manager {};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
return ManagerStr;
}
}
return {};
}
fextl::string FindContainerPrefix() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::vector<char> Manager {};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
// We are running inside of pressure vessel
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
return "/run/host/";
fextl::string FindContainer() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::vector<char> Manager{};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
return ManagerStr;
}
}
return {};
}
return {};
}
void ReloadMetaLayer() {
Meta->Load();
fextl::string FindContainerPrefix() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::vector<char> Manager{};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
// We are running inside of pressure vessel
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
return "/run/host/";
}
}
}
return {};
}
// Do configuration option fix ups after everything is reloaded
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
// Sanitize Core option
FEX_CONFIG_OPT(Core, CORE);
void ReloadMetaLayer() {
Meta->Load();
// Do configuration option fix ups after everything is reloaded
{
// Always fix up the number of threads and create the configuration
// Otherwise the application could receive zero as the number of threads
FEX_CONFIG_OPT(Cores, THREADS);
if (Cores == 0) {
// When the number of emulated CPU cores is zero then auto detect
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, fextl::fmt::format("{}", FEXCore::CPUInfo::CalculateNumberOfCPUs()));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
// Sanitize Core option
FEX_CONFIG_OPT(Core, CORE);
#if (_M_X86_64)
constexpr uint32_t MaxCoreNumber = 1;
constexpr uint32_t MaxCoreNumber = 1;
#else
constexpr uint32_t MaxCoreNumber = 0;
constexpr uint32_t MaxCoreNumber = 0;
#endif
if (Core > MaxCoreNumber) {
// Sanitize the core option by setting the core to the JIT if invalid
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(Core, CORE);
}
fextl::string ContainerPrefix {FindContainerPrefix()};
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
auto NewPath = ExpandPath(ContainerPrefix, PathName);
if (!NewPath.empty()) {
FEXCore::Config::EraseSet(Config, NewPath);
}
};
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
FEX_CONFIG_OPT(PathName, ROOTFS);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
} else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
if (FHU::Filesystem::Exists(NamedRootFS)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
if (Core > MaxCoreNumber) {
// Sanitize the core option by setting the core to the JIT if invalid
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
}
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKHOSTLIBS)) {
FEX_CONFIG_OPT(PathName, THUNKHOSTLIBS);
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKGUESTLIBS)) {
FEX_CONFIG_OPT(PathName, THUNKGUESTLIBS);
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
} else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
if (FHU::Filesystem::Exists(NamedConfig)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(Core, CORE);
}
fextl::string ContainerPrefix { FindContainerPrefix() };
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
auto NewPath = ExpandPath(ContainerPrefix, PathName);
if (!NewPath.empty()) {
FEXCore::Config::EraseSet(Config, NewPath);
}
};
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
FEX_CONFIG_OPT(PathName, ROOTFS);
auto ExpandedString = ExpandPath(ContainerPrefix,PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
}
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
if (FHU::Filesystem::Exists(NamedRootFS)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
}
}
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
if (PathName() != "stdout" && PathName() != "stderr" && PathName() != "server") {
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, PathName());
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKHOSTLIBS)) {
FEX_CONFIG_OPT(PathName, THUNKHOSTLIBS);
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKGUESTLIBS)) {
FEX_CONFIG_OPT(PathName, THUNKGUESTLIBS);
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
}
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
if (FHU::Filesystem::Exists(NamedConfig)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
}
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
if (PathName() != "stdout" && PathName() != "stderr" && PathName() != "server") {
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, PathName());
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_DUMPIR) &&
!FEXCore::Config::Exists(FEXCore::Config::CONFIG_PASSMANAGERDUMPIR)) {
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
FEX_CONFIG_OPT(PathName, DUMPIR);
if (PathName() != "no") {
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR, fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
// Single stepping also enforces single instruction size blocks
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_DUMPIR) && !FEXCore::Config::Exists(FEXCore::Config::CONFIG_PASSMANAGERDUMPIR)) {
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
FEX_CONFIG_OPT(PathName, DUMPIR);
if (PathName() != "no") {
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer) {
ConfigLayers.emplace(_Layer->GetLayerType(), std::move(_Layer));
}
bool Exists(ConfigOption Option) {
return Meta->OptionExists(Option);
}
std::optional<LayerValue*> All(ConfigOption Option) {
return Meta->All(Option);
}
std::optional<fextl::string*> Get(ConfigOption Option) {
return Meta->Get(Option);
}
void Set(ConfigOption Option, std::string_view Data) {
Meta->Set(Option, Data);
}
void Erase(ConfigOption Option) {
Meta->Erase(Option);
}
void EraseSet(ConfigOption Option, std::string_view Data) {
Meta->EraseSet(Option, Data);
}
template<typename T>
T Value<T>::Get(FEXCore::Config::ConfigOption Option) {
T Result;
auto Value = FEXCore::Config::Get(Option);
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
// Single stepping also enforces single instruction size blocks
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
}
}
void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer) {
ConfigLayers.emplace(_Layer->GetLayerType(), std::move(_Layer));
}
bool Exists(ConfigOption Option) {
return Meta->OptionExists(Option);
}
std::optional<LayerValue*> All(ConfigOption Option) {
return Meta->All(Option);
}
std::optional<fextl::string*> Get(ConfigOption Option) {
return Meta->Get(Option);
}
void Set(ConfigOption Option, std::string_view Data) {
Meta->Set(Option, Data);
}
void Erase(ConfigOption Option) {
Meta->Erase(Option);
}
void EraseSet(ConfigOption Option, std::string_view Data) {
Meta->EraseSet(Option, Data);
}
template<typename T>
T Value<T>::Get(FEXCore::Config::ConfigOption Option) {
T Result;
auto Value = FEXCore::Config::Get(Option);
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
}
return Result;
}
template<typename T>
T Value<T>::GetIfExists(FEXCore::Config::ConfigOption Option, T Default) {
T Result;
auto Value = FEXCore::Config::Get(Option);
if (Value && FEXCore::StrConv::Conv(**Value, &Result)) {
return Result;
} else {
return Default;
}
template<typename T>
T Value<T>::GetIfExists(FEXCore::Config::ConfigOption Option, T Default) {
T Result;
auto Value = FEXCore::Config::Get(Option);
if (Value && FEXCore::StrConv::Conv(**Value, &Result)) {
return Result;
}
else {
return Default;
}
}
template<>
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, fextl::string Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
}
else {
return Default;
}
}
template<>
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
}
else {
return fextl::string(Default);
}
}
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
template int16_t Value<int16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int16_t Default);
template uint16_t Value<uint16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint16_t Default);
template int32_t Value<int32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int32_t Default);
template uint32_t Value<uint32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint32_t Default);
template int64_t Value<int64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int64_t Default);
template uint64_t Value<uint64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint64_t Default);
// Constructor
template Value<fextl::string>::Value(FEXCore::Config::ConfigOption _Option, fextl::string Default);
template Value<bool>::Value(FEXCore::Config::ConfigOption _Option, bool Default);
template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t Default);
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
template<typename T>
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List) {
auto Value = FEXCore::Config::All(Option);
List->clear();
if (Value) {
*List = **Value;
}
}
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
}
template<>
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, fextl::string Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
} else {
return Default;
}
}
template<>
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
} else {
return fextl::string(Default);
}
}
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
template int16_t Value<int16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int16_t Default);
template uint16_t Value<uint16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint16_t Default);
template int32_t Value<int32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int32_t Default);
template uint32_t Value<uint32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint32_t Default);
template int64_t Value<int64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int64_t Default);
template uint64_t Value<uint64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint64_t Default);
// Constructor
template Value<fextl::string>::Value(FEXCore::Config::ConfigOption _Option, fextl::string Default);
template Value<bool>::Value(FEXCore::Config::ConfigOption _Option, bool Default);
template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t Default);
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
template<typename T>
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List) {
auto Value = FEXCore::Config::All(Option);
List->clear();
if (Value) {
*List = **Value;
}
}
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List);
} // namespace FEXCore::Config
+31 -66
View File
@@ -31,6 +31,15 @@
"Maximum number of instruction to store in a block"
]
},
"Threads": {
"Type": "uint32",
"Default": "0",
"ShortArg": "T",
"Desc": [
"Number of physical hardware threads to tell the process we have.",
"0 will auto detect."
]
},
"CacheObjectCodeCompilation": {
"Type": "uint32",
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
@@ -77,9 +86,7 @@
"ENABLECRYPTO": "enablecrypto",
"DISABLECRYPTO": "disablecrypto",
"ENABLERPRES": "enablerpres",
"DISABLERPRES": "disablerpres",
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
"DISABLEPRESERVEALLABI": "disablepreserveallabi"
"DISABLERPRES": "disablerpres"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
@@ -99,28 +106,7 @@
"\t{enable,disable}flagm: Will force enable or disable flagm even if the host doesn't support it",
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it",
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
]
},
"CPUID": {
"Type": "strenum",
"Default": "FEXCore::Config::CPUID::OFF",
"Enums": {
"ENABLESHA": "enablesha",
"DISABLESHA": "disablesha"
},
"Desc": [
"Allows controlling of the CPU features are exposed in CPUID.",
"\toff: Default CPU features queried from CPU features",
"\t{enable,disable}sha: Will force enable or disable sha even if the host doesn't support it"
]
},
"SmallTSCScale": {
"Type": "bool",
"Default": "true",
"Desc": [
"Scales the cycle counter on systems that have low frequencies."
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it"
]
}
},
@@ -270,6 +256,23 @@
"Disables optimizations passes for debugging."
]
},
"SRA": {
"Type": "bool",
"Default": "true",
"Desc": [
"Set to false to disable Static Register Allocation"
]
},
"Force32BitAllocator": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces use of the 32-bit allocator on 32-bit applications",
"Used to work around ulimit problems of CI runner",
"Potentially useful for debugging memory problems",
"32-bit allocator is always used if your host kernel is older than 4.17"
]
},
"GlobalJITNaming": {
"Type": "bool",
"Default": "false",
@@ -368,14 +371,6 @@
"File to write FEX output to.",
"[stdout, stderr, server, <Filename>]"
]
},
"TelemetryDirectory": {
"Type": "str",
"Default": "",
"Desc": [
"Redirects the telemetry folder that FEX usually writes to.",
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
]
}
},
"Hacks": {
@@ -387,8 +382,9 @@
"Desc": [
"Checks code for modification before execution.",
"\tnone: No checks",
"\tmtrack: Page tracking based invalidation (default)",
"\tfull: Validate code before every run (slow)"
"\tmtrack: Page tracking based invalidation",
"\tfull: Validate code before every run (slow)",
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
]
},
"TSOEnabled": {
@@ -399,29 +395,6 @@
"Highly likely to break any multithreaded application if disabled."
]
},
"VectorTSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
]
},
"MemcpySetTSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
"Only affects REP MOVS and REP STOS instructions"
]
},
"HalfBarrierTSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"When TSO emulation is enabled, controls if unaligned loads and stores should be backpatched to half-barrier atomics.",
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
@@ -469,14 +442,6 @@
"Hides the hypervisor CPUID bit when set.",
"Should only be used for applications that have issues with this set."
]
},
"StartupSleep": {
"Type": "uint32",
"Default": "0",
"Desc": [
"Sleeps the process at startup for a duration of seconds.",
"Useful if an application crashes too quickly to attach a debugger."
]
}
},
"Misc": {
+70 -44
View File
@@ -12,62 +12,88 @@
#include <string.h>
#include <utility>
namespace FEXCore::HLE {
class SyscallVisitor;
}
namespace FEXCore::Context {
void InitializeStaticTables(OperatingMode Mode) {
X86Tables::InitializeInfoTables(Mode);
IR::InstallOpcodeHandlers(Mode);
}
void InitializeStaticTables(OperatingMode Mode) {
X86Tables::InitializeInfoTables(Mode);
IR::InstallOpcodeHandlers(Mode);
}
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext() {
return fextl::make_unique<FEXCore::Context::ContextImpl>();
}
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext() {
return fextl::make_unique<FEXCore::Context::ContextImpl>();
}
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
CustomExitHandler = std::move(handler);
}
bool FEXCore::Context::ContextImpl::InitializeContext() {
// This should be used for generating things that are shared between threads
CPUID.Init(this);
return true;
}
ExitHandler FEXCore::Context::ContextImpl::GetExitHandler() const {
return CustomExitHandler;
}
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
CustomExitHandler = std::move(handler);
}
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
CompileBlock(Thread->CurrentFrame, GuestRIP);
}
ExitHandler FEXCore::Context::ContextImpl::GetExitHandler() const {
return CustomExitHandler;
}
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
}
void FEXCore::Context::ContextImpl::Stop() {
Stop(false);
}
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
CustomCPUFactory = std::move(Factory);
}
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
CompileBlock(Thread->CurrentFrame, GuestRIP);
}
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
return HostFeatures;
}
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
}
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator* _SignalDelegation) {
SignalDelegation = _SignalDelegation;
}
FEXCore::Context::ExitReason FEXCore::Context::ContextImpl::GetExitReason() {
return ParentThread->ExitReason;
}
void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) {
SyscallHandler = Handler;
SourcecodeResolver = Handler->GetSourcecodeResolver();
}
bool FEXCore::Context::ContextImpl::IsDone() const {
return IsPaused();
}
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
return CPUID.RunFunction(Function, Leaf);
}
void FEXCore::Context::ContextImpl::GetCPUState(FEXCore::Core::CPUState *State) const {
memcpy(State, ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
}
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
return CPUID.RunXCRFunction(Function);
}
void FEXCore::Context::ContextImpl::SetCPUState(const FEXCore::Core::CPUState *State) {
memcpy(ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
}
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
return CPUID.RunFunctionName(Function, Leaf, CPU);
}
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
CustomCPUFactory = std::move(Factory);
}
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const {
return Thread->CPUBackend->IsAddressInCodeBuffer(Address);
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
return HostFeatures;
}
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator *_SignalDelegation) {
SignalDelegation = _SignalDelegation;
}
void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) {
SyscallHandler = Handler;
SourcecodeResolver = Handler->GetSourcecodeResolver();
}
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
return CPUID.RunFunction(Function, Leaf);
}
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
return CPUID.RunXCRFunction(Function);
}
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
return CPUID.RunFunctionName(Function, Leaf, CPU);
}
}
} // namespace FEXCore::Context
+365 -315
View File
@@ -13,10 +13,9 @@
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/DeferredSignalMutex.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/set.h>
#include <FEXCore/fextl/string.h>
@@ -38,6 +37,7 @@
namespace FEXCore {
class CodeLoader;
class ThunkHandler;
class GdbServer;
namespace CodeSerialize {
class CodeObjectSerializeService;
@@ -45,377 +45,427 @@ namespace CodeSerialize {
namespace CPU {
class Arm64JITCore;
class X86JITCore;
class Dispatcher;
} // namespace CPU
}
namespace HLE {
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
} // namespace HLE
} // namespace FEXCore
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
}
}
namespace FEXCore::IR {
class RegisterAllocationData;
class IRListView;
class RegisterAllocationData;
class IRListView;
namespace Validation {
class IRValidation;
}
} // namespace FEXCore::IR
}
namespace FEXCore::Context {
enum CoreRunningMode {
MODE_RUN = 0,
MODE_SINGLESTEP = 1,
};
enum CoreRunningMode {
MODE_RUN = 0,
MODE_SINGLESTEP = 1,
};
struct ExitFunctionLinkData {
uint64_t HostBranch;
uint64_t GuestRIP;
};
class ContextImpl final : public FEXCore::Context::Context {
public:
// Context base class implementation.
bool InitializeContext() override;
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
constexpr uint32_t TSC_SCALE = 128;
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
class ContextImpl final : public FEXCore::Context::Context {
public:
// Context base class implementation.
bool InitCore() override;
void SetExitHandler(ExitHandler handler) override;
ExitHandler GetExitHandler() const override;
void SetExitHandler(ExitHandler handler) override;
ExitHandler GetExitHandler() const override;
void Pause() override;
void Run() override;
void Stop() override;
void Step() override;
ExitReason RunUntilExit(FEXCore::Core::InternalThreadState* Thread) override;
ExitReason RunUntilExit() override;
void ExecuteThread(FEXCore::Core::InternalThreadState* Thread) override;
void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) override;
void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) override;
void CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
int GetProgramStatus() const override;
HostFeatures GetHostFeatures() const override;
ExitReason GetExitReason() override;
void HandleCallback(FEXCore::Core::InternalThreadState* Thread, uint64_t RIP) override;
bool IsDone() const override;
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) override;
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) override;
void GetCPUState(FEXCore::Core::CPUState *State) const override;
void SetCPUState(const FEXCore::Core::CPUState *State) override;
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param InitialRIP The starting RIP of this thread
* @param StackPointer The starting RSP of this thread
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* Parent thread Creation:
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
* - CTX->RunUntilExit(Thread);
* OS thread Creation:
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
* - ThreadHandler calls `CTX->ExecutionThread(Thread)`
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(0, 0, NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
FEXCore::Core::InternalThreadState*
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
HostFeatures GetHostFeatures() const override;
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState* Thread) override;
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
/**
* @brief Destroys this FEX thread object and stops tracking it internally
*
* @param Thread The internal FEX thread state object
*/
void DestroyThread(FEXCore::Core::InternalThreadState* Thread, bool NeedsTLSUninstall) override;
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) override;
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) override;
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param NewThreadState The initial thread state to setup for our state
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* OS thread Creation:
* - Thread = CreateThread(NewState, PPID);
* - InitializeThread(Thread);
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
*
* @param Thread The internal FEX thread state object
*
* The OS thread will wait until RunThread is executed
*/
void InitializeThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Starts the OS thread object to start executing guest code
*
* @param Thread The internal FEX thread state object
*/
void RunThread(FEXCore::Core::InternalThreadState *Thread) override;
void StopThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Destroys this FEX thread object and stops tracking it internally
*
* @param Thread The internal FEX thread state object
*/
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
#ifndef _WIN32
void LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) override;
void UnlockAfterFork(FEXCore::Core::InternalThreadState* Thread, bool Child) override;
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override;
#endif
void SetSignalDelegator(FEXCore::SignalDelegator* SignalDelegation) override;
void SetSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) override;
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) override;
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) override;
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
}
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
}
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
}
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
}
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
}
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
}
void FinalizeAOTIRCache() override {
IRCaptureCache.FinalizeAOTIRCache();
}
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void FinalizeAOTIRCache() override {
IRCaptureCache.FinalizeAOTIRCache();
}
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
void MarkMemoryShared() override;
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
return CodeInvalidationMutex;
}
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
// returns false if a handler was already registered
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) override;
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
public:
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
#ifdef JIT_X86_64
friend class FEXCore::CPU::X86JITCore;
#endif
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
friend class FEXCore::IR::Validation::IRValidation;
// returns false if a handler was already registered
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator = nullptr, void* Data = nullptr);
struct {
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
uint64_t VirtualMemSize{1ULL << 36};
void AppendThunkDefinitions(const fextl::vector<FEXCore::IR::ThunkDefinition>& Definitions) override;
// this is for internal use
bool ValidateIRarser { false };
public:
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
// Used if the JIT needs to have its interrupt fault code emitted.
bool NeedsPendingInterruptFaultCheck { false };
friend class FEXCore::IR::Validation::IRValidation;
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(Core, CORE);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
} Config;
struct {
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
uint64_t VirtualMemSize {1ULL << 36};
FEXCore::HostFeatures HostFeatures;
// this is for internal use
bool ValidateIRarser {false};
std::mutex ThreadCreationMutex;
FEXCore::Core::InternalThreadState* ParentThread{};
fextl::vector<FEXCore::Core::InternalThreadState*> Threads;
std::atomic_bool CoreShuttingDown{false};
bool NeedToCheckXID{true};
// Used if the JIT needs to have its interrupt fault code emitted.
bool NeedsPendingInterruptFaultCheck {false};
std::mutex IdleWaitMutex;
std::condition_variable IdleWaitCV;
std::atomic<uint32_t> IdleWaitRefCount{};
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(Core, CORE);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
} Config;
Event PauseWait;
bool Running{};
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
std::atomic_bool CoreShuttingDown {false};
FEXCore::CPUIDEmu CPUID;
FEXCore::HLE::SyscallHandler *SyscallHandler{};
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
FEXCore::HostFeatures HostFeatures;
// CPUID depends on HostFeatures so needs to be initialized after that.
FEXCore::CPUIDEmu CPUID;
FEXCore::HLE::SyscallHandler* SyscallHandler {};
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CustomCPUFactoryType CustomCPUFactory;
FEXCore::Context::ExitHandler CustomExitHandler;
CustomCPUFactoryType CustomCPUFactory;
FEXCore::Context::ExitHandler CustomExitHandler;
#ifdef BLOCKSTATS
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
#endif
SignalDelegator* SignalDelegation {};
X86GeneratedCode X86CodeGen;
SignalDelegator *SignalDelegation{};
X86GeneratedCode X86CodeGen;
ContextImpl();
~ContextImpl();
ContextImpl();
~ContextImpl();
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestDestination,
FEXCore::Context::ExitFunctionLinkData* HostLink, const BlockDelinkerFunc& delinker);
bool IsPaused() const { return !Running; }
void WaitForThreadsToRun() override;
void Stop(bool IgnoreCurrentThread);
void WaitForIdle() override;
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, ExitFunctionLinkData* Record) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
void StartGdbServer();
void StopGdbServer();
return Fn(Frame, Record);
}
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
// Must be called from owning thread
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
LOGMAN_THROW_A_FMT(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}",
Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
return Fn(Frame, record);
}
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
// Must be called from owning thread
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
struct GenerateIRResult {
FEXCore::IR::IRListView* IRList;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
uint64_t TotalInstructions;
uint64_t TotalInstructionsLength;
uint64_t StartAddr;
uint64_t Length;
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
struct GenerateIRResult {
FEXCore::IR::IRListView* IRList;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
uint64_t TotalInstructions;
uint64_t TotalInstructionsLength;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
struct CompileCodeResult {
void* CompiledCode;
FEXCore::IR::IRListView* IRData;
FEXCore::Core::DebugData* DebugData;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
bool GeneratedIR;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
// same as CompileBlock, but aborts on failure
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
// Used for thread creation from syscalls
/**
* @brief Initializes TID, PID and TLS data for a thread
*
* @param Thread The internal FEX thread state object
*/
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread);
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
FEXCore::JITSymbols Symbols;
void GetVDSOSigReturn(VDSOSigReturn *VDSOPointers) override {
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
}
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
}
}
void IncrementIdleRefCount() override {
++IdleWaitRefCount;
}
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
// If Atomic-based TSO emulation is enabled or not.
bool IsAtomicTSOEnabled() const { return AtomicTSOEmulationEnabled; }
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
SupportsHardwareTSO = HardwareTSOSupported;
UpdateAtomicTSOEmulationConfig();
}
// Returns if Software TSO emulation is required.
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
// This will still return true if on a single thread and TSO is currently disabled.
//
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
// we return consistent results.
//
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
bool SoftwareTSORequired() const {
if (SupportsHardwareTSO) return false;
return Config.TSOEnabled;
}
void EnableExitOnHLT() override { ExitOnHLT = true; }
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
ThreadsState GetThreads() override {
return ThreadsState {
.ParentThread = ParentThread,
.Threads = &Threads,
};
}
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
protected:
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
// If the hardware supports TSO then we don't need to emulate it through atomics.
AtomicTSOEmulationEnabled = false;
}
else {
// Atomic TSO emulation only enabled if the config option is enabled.
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
}
}
private:
/**
* @brief Does some final thread initialization
*
* @param Thread The internal FEX thread state object
*
* InitCore and CreateThread both call this to finish up thread object initialization
*/
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
/**
* @brief Initializes the JIT compilers for the thread
*
* @param State The internal FEX thread state object
*
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
void WaitForIdleWithTimeout();
void NotifyPause();
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
// Entry Cache
std::mutex ExitMutex;
fextl::unique_ptr<GdbServer> DebugServer;
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
bool StartPaused = false;
bool IsMemoryShared = false;
bool SupportsHardwareTSO = false;
bool AtomicTSOEmulationEnabled = true;
bool ExitOnHLT = false;
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
std::shared_mutex CustomIRMutex;
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void *, void *>> CustomIRHandlers;
FEXCore::CPU::DispatcherConfig DispatcherConfig;
};
[[nodiscard]]
GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
struct CompileCodeResult {
void* CompiledCode;
FEXCore::IR::IRListView* IRData;
FEXCore::Core::DebugData* DebugData;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
bool GeneratedIR;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]]
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
// Used for thread creation from syscalls
/**
* @brief Initializes TID, PID and TLS data for a thread
*
* @param Thread The internal FEX thread state object
*/
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState* Thread);
void CopyMemoryMapping(FEXCore::Core::InternalThreadState* ParentThread, FEXCore::Core::InternalThreadState* ChildThread);
uint8_t GetGPRSize() const {
return Config.Is64BitMode ? 8 : 4;
}
FEXCore::JITSymbols Symbols;
void GetVDSOSigReturn(VDSOSigReturn* VDSOPointers) override {
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
}
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
}
}
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
// If Atomic-based TSO emulation is enabled or not.
bool IsAtomicTSOEnabled() const {
return AtomicTSOEmulationEnabled;
}
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
SupportsHardwareTSO = HardwareTSOSupported;
UpdateAtomicTSOEmulationConfig();
}
// Returns if Software TSO emulation is required.
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
// This will still return true if on a single thread and TSO is currently disabled.
//
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
// we return consistent results.
//
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
bool SoftwareTSORequired() const {
if (SupportsHardwareTSO) {
return false;
}
return Config.TSOEnabled;
}
void EnableExitOnHLT() override {
ExitOnHLT = true;
}
bool ExitOnHLTEnabled() const {
return ExitOnHLT;
}
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
protected:
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
// If the hardware supports TSO then we don't need to emulate it through atomics.
AtomicTSOEmulationEnabled = false;
} else {
// Atomic TSO emulation only enabled if the config option is enabled.
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
}
}
private:
/**
* @brief Initializes the JIT compilers for the thread
*
* @param State The internal FEX thread state object
*
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* Ptr);
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
bool StartPaused = false;
bool IsMemoryShared = false;
bool SupportsHardwareTSO = false;
bool AtomicTSOEmulationEnabled = true;
bool ExitOnHLT = false;
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
std::shared_mutex CustomIRMutex;
std::atomic<bool> HasCustomIRHandlers {};
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void*, void*>> CustomIRHandlers;
};
} // namespace FEXCore::Context
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
}
@@ -1,6 +1,5 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "FEXCore/Core/X86Enums.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
@@ -9,11 +8,10 @@
#include "Interface/HLE/Thunks/Thunks.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <aarch64/cpu-aarch64.h>
#include <aarch64/instructions-aarch64.h>
#include <cpu-features.h>
@@ -29,227 +27,183 @@ namespace FEXCore::CPU {
// TODO: Allow x18 register allocation on Linux in the future to gain one more register.
namespace x64 {
#ifndef _M_ARM_64EC
// All but x19 and x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
FEXCore::ARMEmitter::Reg::r4,
FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6,
FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8,
FEXCore::ARMEmitter::Reg::r9,
FEXCore::ARMEmitter::Reg::r10,
FEXCore::ARMEmitter::Reg::r11,
FEXCore::ARMEmitter::Reg::r12,
FEXCore::ARMEmitter::Reg::r13,
FEXCore::ARMEmitter::Reg::r14,
FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16,
FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r19,
FEXCore::ARMEmitter::Reg::r29,
// PF/AF must be last.
REG_PF,
REG_AF,
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
};
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
// All these callee saved
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r30,
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
FEXCore::ARMEmitter::Reg::r30,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
}};
// All are caller saved
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
};
// v8..v15 = (lower 64bits) Callee saved
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
// v0 ~ v1 are used as temps.
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
};
#else
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
FEXCore::ARMEmitter::Reg::r8,
FEXCore::ARMEmitter::Reg::r0,
FEXCore::ARMEmitter::Reg::r1,
FEXCore::ARMEmitter::Reg::r27,
// SP's register location isn't specified by the ARM64EC ABI, we choose to use r23
FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r29,
FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r26,
FEXCore::ARMEmitter::Reg::r2,
FEXCore::ARMEmitter::Reg::r3,
FEXCore::ARMEmitter::Reg::r4,
FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r19,
FEXCore::ARMEmitter::Reg::r20,
FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22,
REG_PF,
REG_AF,
};
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r30,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
{FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7},
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
}};
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1, FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9, FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13, FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
};
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19, FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23, FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27, FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
#endif
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 7> PreserveAll_SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
};
constexpr uint32_t PreserveAll_SRAMask = {[]() -> uint32_t {
uint32_t Mask {};
for (auto Reg : PreserveAll_SRA) {
switch (Reg.Idx()) {
case 0:
case 1:
case 2:
case 3:
case 4:
case 5:
case 6:
case 7:
case 8:
case 16:
case 17: Mask |= (1U << Reg.Idx()); break;
default: break;
constexpr uint32_t PreserveAll_SRAMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRA) {
switch (Reg.Idx()) {
case 0:
case 1:
case 2:
case 3:
case 4:
case 5:
case 6:
case 7:
case 8:
case 16:
case 17:
Mask |= (1U << Reg.Idx());
break;
default: break;
}
}
}
return Mask;
}()};
return Mask;
}()
};
// Dynamic GPRs
constexpr std::array<FEXCore::ARMEmitter::Register, 1> PreserveAll_Dynamic = {
// Only LR needs to get saved.
FEXCore::ARMEmitter::Reg::r30};
FEXCore::ARMEmitter::Reg::r30
};
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
// None.
};
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
uint32_t Mask {};
for (auto Reg : PreserveAll_SRAFPR) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()};
constexpr uint32_t PreserveAll_SRAFPRMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPR) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs
// - v0-v7
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
// v0 ~ v1 are temps
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4,
FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
};
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
// This is /all/ of the SRA registers
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> PreserveAll_SRAFPRSVE = SRAFPR;
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {[]() -> uint32_t {
uint32_t Mask {};
for (auto Reg : PreserveAll_SRAFPRSVE) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()};
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPRSVE) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs when the host supports SVE-256bit.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> PreserveAll_DynamicFPRSVE = {
// v0 ~ v1 are used as temps.
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
};
} // namespace x64
}
namespace x32 {
// All but x19 and x29 are caller saved
constexpr std::array<FEXCore::ARMEmitter::Register, 10> SRA = {
FEXCore::ARMEmitter::Reg::r4,
FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6,
FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8,
FEXCore::ARMEmitter::Reg::r9,
FEXCore::ARMEmitter::Reg::r10,
FEXCore::ARMEmitter::Reg::r11,
// PF/AF must be last.
REG_PF,
REG_AF,
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
};
constexpr std::array<FEXCore::ARMEmitter::Register, 15> RA = {
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
// All these callee saved
FEXCore::ARMEmitter::Reg::r20,
FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22,
FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24,
FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
// Registers only available on 32-bit
// All these are caller saved (except for r19).
FEXCore::ARMEmitter::Reg::r12,
FEXCore::ARMEmitter::Reg::r13,
FEXCore::ARMEmitter::Reg::r14,
FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16,
FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r29,
FEXCore::ARMEmitter::Reg::r30,
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30,
FEXCore::ARMEmitter::Reg::r19,
};
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 7> RAPair = {{
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
@@ -259,8 +213,10 @@ namespace x32 {
// All are caller saved
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> SRAFPR = {
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
};
// v8..v15 = (lower 64bits) Callee saved
@@ -268,98 +224,122 @@ namespace x32 {
// v0 ~ v1 are used as temps.
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
};
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 5> PreserveAll_SRA = {
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6,
FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r8,
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
FEXCore::ARMEmitter::Reg::r8,
};
constexpr uint32_t PreserveAll_SRAMask = {[]() -> uint32_t {
uint32_t Mask {};
for (auto Reg : PreserveAll_SRA) {
switch (Reg.Idx()) {
case 0:
case 1:
case 2:
case 3:
case 4:
case 5:
case 6:
case 7:
case 8:
case 16:
case 17: Mask |= (1U << Reg.Idx()); break;
default: break;
constexpr uint32_t PreserveAll_SRAMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRA) {
switch (Reg.Idx()) {
case 0:
case 1:
case 2:
case 3:
case 4:
case 5:
case 6:
case 7:
case 8:
case 16:
case 17:
Mask |= (1U << Reg.Idx());
break;
default: break;
}
}
}
return Mask;
}()};
return Mask;
}()
};
// Dynamic GPRs
constexpr std::array<FEXCore::ARMEmitter::Register, 3> PreserveAll_Dynamic = {
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r30};
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
FEXCore::ARMEmitter::Reg::r30
};
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
// None.
};
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
uint32_t Mask {};
for (auto Reg : PreserveAll_SRAFPR) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()};
constexpr uint32_t PreserveAll_SRAFPRMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPR) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs
// - v0-v7
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
// v0 ~ v1 are temps
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4,
FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
};
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
// This is /all/ of the SRA registers
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> PreserveAll_SRAFPRSVE = SRAFPR;
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {[]() -> uint32_t {
uint32_t Mask {};
for (auto Reg : PreserveAll_SRAFPRSVE) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()};
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {
[]() -> uint32_t {
uint32_t Mask{};
for (auto Reg : PreserveAll_SRAFPRSVE) {
Mask |= (1U << Reg.Idx());
}
return Mask;
}()
};
// Dynamic FPRs when the host supports SVE-256bit.
constexpr std::array<FEXCore::ARMEmitter::VRegister, 22> PreserveAll_DynamicFPRSVE = {
// v0 ~ v1 are used as temps.
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
} // namespace x32
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
};
}
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr, size_t size)
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr, size_t size)
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
, EmitterCTX {ctx}
#ifdef VIXL_SIMULATOR
, Simulator {&SimDecoder, stdout, vixl::aarch64::SimStack(SimulatorStackSize).Allocate()}
, Simulator {&SimDecoder}
#endif
{
#ifdef VIXL_SIMULATOR
@@ -372,10 +352,8 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
// Only setup the disassembler if enabled.
// vixl's decoder is expensive to setup.
if (Disassemble()) {
DisasmBuffer.resize(DISASM_BUFFER_SIZE);
Disasm = fextl::make_unique<vixl::aarch64::Disassembler>(DisasmBuffer.data(), DISASM_BUFFER_SIZE);
DisasmDecoder = fextl::make_unique<vixl::aarch64::Decoder>();
DisasmDecoder->AppendVisitor(Disasm.get());
DisasmDecoder->AppendVisitor(&Disasm);
}
#endif
@@ -388,11 +366,9 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
GeneralPairRegisters = x64::RAPair;
StaticFPRegisters = x64::SRAFPR;
GeneralFPRegisters = x64::RAFPR;
#ifdef _M_ARM_64EC
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
#endif
} else {
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
}
else {
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
StaticRegisters = x32::SRA;
GeneralRegisters = x32::RA;
@@ -407,13 +383,11 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
int Segments = Is64Bit ? 4 : 2;
if (Is64Bit && ((~Constant) >> 16) == 0) {
if (Is64Bit && ((~Constant)>> 16) == 0) {
movn(s, Reg, (~Constant) & 0xFFFF);
if (NOPPad) {
nop();
nop();
nop();
nop(); nop(); nop();
}
return;
}
@@ -425,18 +399,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
Segments = 2;
}
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
movn(s, Reg.W(), (~Constant) & 0xFFFF);
if (NOPPad) {
nop();
nop();
nop();
}
return;
}
int RequiredMoveSegments {};
int RequiredMoveSegments{};
// Count the number of move segments
// We only want to use ADRP+ADD if we have more than 1 segment
@@ -455,9 +418,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
if (IsImm) {
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
if (NOPPad) {
nop();
nop();
nop();
nop(); nop(); nop();
}
return;
}
@@ -481,20 +442,23 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
// If this is 4k page aligned then we only need ADRP
if ((AlignedOffset & 0xFFF) == 0) {
adrp(Reg, AlignedOffset >> 12);
} else {
}
else {
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
// 21-bit signed integer here
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
if (vixl::IsInt21(SmallOffset)) {
adr(Reg, SmallOffset);
} else {
}
else {
// Need to use ADRP + ADD
adrp(Reg, AlignedOffset >> 12);
add(s, Reg, Reg, Constant & 0xFFF);
NumMoves = 2;
}
}
} else {
}
else {
int CurrentSegment = 0;
for (; CurrentSegment < Segments; ++CurrentSegment) {
uint16_t Part = (Constant >> (CurrentSegment * 16)) & 0xFFFF;
@@ -540,14 +504,18 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
}};
for (auto& RegPair : CalleeSaved) {
for (auto &RegPair : CalleeSaved) {
stp<ARMEmitter::IndexType::PRE>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, -16);
}
// Additionally we need to store the lower 64bits of v8-v15
// Here's a fun thing, we can use two ST4 instructions to store everything
// We just need a single sub to sp before that
const std::array< std::tuple<ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister>, 2> FPRs = {{
const std::array<
std::tuple<ARMEmitter::DRegister,
ARMEmitter::DRegister,
ARMEmitter::DRegister,
ARMEmitter::DRegister>, 2> FPRs = {{
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
}};
@@ -558,21 +526,37 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
// We just saved x19 so it is safe
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r19, ARMEmitter::Reg::rsp, 0);
for (auto& RegQuad : FPRs) {
st4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
ARMEmitter::Reg::r19, 32);
for (auto &RegQuad : FPRs) {
st4(ARMEmitter::SubRegSize::i64Bit,
std::get<0>(RegQuad),
std::get<1>(RegQuad),
std::get<2>(RegQuad),
std::get<3>(RegQuad),
0,
ARMEmitter::Reg::r19,
32);
}
}
void Arm64Emitter::PopCalleeSavedRegisters() {
const std::array< std::tuple<ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister>, 2> FPRs = {{
const std::array<
std::tuple<ARMEmitter::DRegister,
ARMEmitter::DRegister,
ARMEmitter::DRegister,
ARMEmitter::DRegister>, 2> FPRs = {{
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
}};
for (auto& RegQuad : FPRs) {
ld4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
ARMEmitter::Reg::rsp, 32);
for (auto &RegQuad : FPRs) {
ld4(ARMEmitter::SubRegSize::i64Bit,
std::get<0>(RegQuad),
std::get<1>(RegQuad),
std::get<2>(RegQuad),
std::get<3>(RegQuad),
0,
ARMEmitter::Reg::rsp,
32);
}
const fextl::vector<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>> CalleeSaved = {{
@@ -584,7 +568,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
}};
for (auto& RegPair : CalleeSaved) {
for (auto &RegPair : CalleeSaved) {
ldp<ARMEmitter::IndexType::POST>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, 16);
}
}
@@ -597,49 +581,34 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
// Disable FPCR.NEP and FPCR.AH
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
// Also interacts with RPRES to change reciprocal/rsqrt precision from 8-bit mantissa to 12-bit.
//
// Additional interesting AFP bits:
// FIZ(0): Flush Inputs to Zero
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
(1U << 2) | // NEP
(1U << 1)); // AH
(1U << 2) | // NEP
(1U << 1)); // AH
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
}
#endif
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
// is always static and almost certainly clobbered by the subsequent code.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
GPRSpillMask &= ~PFAFSpillMask;
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i + 1];
if (((1U << Reg1.Idx()) & GPRSpillMask) && ((1U << Reg2.Idx()) & GPRSpillMask)) {
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
} else if (((1U << Reg1.Idx()) & GPRSpillMask)) {
str(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
} else if (((1U << Reg2.Idx()) & GPRSpillMask)) {
str(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i + 1]));
}
if (!StaticRegisterAllocation()) {
return;
}
// Now handle PF/AF
if (PFAFSpillMask) {
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i+1];
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
((1U << Reg2.Idx()) & GPRSpillMask)) {
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
}
else if (((1U << Reg1.Idx()) & GPRSpillMask)) {
str(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
}
else if (((1U << Reg2.Idx()) & GPRSpillMask)) {
str(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
}
}
if (FPRs) {
@@ -664,17 +633,21 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
const auto Reg4 = StaticFPRegisters[i + 3];
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
}
} else {
}
else {
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
const auto Reg1 = StaticFPRegisters[i];
const auto Reg2 = StaticFPRegisters[i + 1];
if (((1U << Reg1.Idx()) & FPRSpillMask) && ((1U << Reg2.Idx()) & FPRSpillMask)) {
if (((1U << Reg1.Idx()) & FPRSpillMask) &&
((1U << Reg2.Idx()) & FPRSpillMask)) {
stp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
} else if (((1U << Reg1.Idx()) & FPRSpillMask)) {
}
else if (((1U << Reg1.Idx()) & FPRSpillMask)) {
str(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
} else if (((1U << Reg2.Idx()) & FPRSpillMask)) {
str(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i + 1][0]));
}
else if (((1U << Reg2.Idx()) & FPRSpillMask)) {
str(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0]));
}
}
}
@@ -685,7 +658,7 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
[[maybe_unused]] bool FoundRegister {};
bool FoundRegister{};
for (auto Reg : StaticRegisters) {
if (((1U << Reg.Idx()) & GPRFillMask)) {
TmpReg = Reg;
@@ -709,19 +682,15 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
// Additional interesting AFP bits:
// FIZ(0): Flush Inputs to Zero
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
(1U << 2) | // NEP
(1U << 1)); // AH
(1U << 2) | // NEP
(1U << 1)); // AH
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
}
#endif
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
// is always static and was almost certainly clobbered.
//
// TODO: Can we prove that NZCV is not used across a call in some cases and
// omit this? Might help x87 perf? Future idea.
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
if (!StaticRegisterAllocation()) {
return;
}
if (FPRs) {
// Set up predicate registers.
@@ -755,46 +724,40 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
const auto Reg4 = StaticFPRegisters[i + 3];
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
}
} else {
}
else {
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
const auto Reg1 = StaticFPRegisters[i];
const auto Reg2 = StaticFPRegisters[i + 1];
if (((1U << Reg1.Idx()) & FPRFillMask) && ((1U << Reg2.Idx()) & FPRFillMask)) {
if (((1U << Reg1.Idx()) & FPRFillMask) &&
((1U << Reg2.Idx()) & FPRFillMask)) {
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
} else if (((1U << Reg1.Idx()) & FPRFillMask)) {
}
else if (((1U << Reg1.Idx()) & FPRFillMask)) {
ldr(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
} else if (((1U << Reg2.Idx()) & FPRFillMask)) {
ldr(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i + 1][0]));
}
else if (((1U << Reg2.Idx()) & FPRFillMask)) {
ldr(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0]));
}
}
}
}
}
// PF/AF are special, remove them from the mask
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
GPRFillMask &= ~PFAFMask;
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
auto Reg1 = StaticRegisters[i];
auto Reg2 = StaticRegisters[i + 1];
if (((1U << Reg1.Idx()) & GPRFillMask) && ((1U << Reg2.Idx()) & GPRFillMask)) {
auto Reg2 = StaticRegisters[i+1];
if (((1U << Reg1.Idx()) & GPRFillMask) &&
((1U << Reg2.Idx()) & GPRFillMask)) {
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
} else if ((1U << Reg1.Idx()) & GPRFillMask) {
ldr(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
} else if ((1U << Reg2.Idx()) & GPRFillMask) {
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i + 1]));
}
}
// Now handle PF/AF
if (PFAFFillMask) {
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
else if ((1U << Reg1.Idx()) & GPRFillMask) {
ldr(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
}
else if ((1U << Reg2.Idx()) & GPRFillMask) {
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
}
}
}
@@ -817,7 +780,8 @@ void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, boo
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
}
} else {
}
else {
size_t i = 0;
for (; i < (VRegs.size() % 4); i += 2) {
const auto Reg1 = VRegs[i];
@@ -901,7 +865,8 @@ void Arm64Emitter::PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Regi
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
: Core::CPUState::XMM_SSE_REG_SIZE;
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
@@ -918,9 +883,7 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
// Push the general registers.
PushGeneralRegisters(TmpReg, ConfiguredDynamicRegisterBase);
#ifndef _M_ARM_64EC
str(ARMEmitter::XReg::lr, TmpReg, 0);
#endif
}
void Arm64Emitter::PopDynamicRegsAndLR() {
@@ -932,19 +895,18 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
// Pop GPRs second
PopGeneralRegisters(ConfiguredDynamicRegisterBase);
#ifndef _M_ARM_64EC
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
#endif
}
void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs) {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
: Core::CPUState::XMM_SSE_REG_SIZE;
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs {};
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs {};
uint32_t PreserveSRAMask {};
uint32_t PreserveSRAFPRMask {};
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs{};
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs{};
uint32_t PreserveSRAMask{};
uint32_t PreserveSRAFPRMask{};
if (EmitterCTX->Config.Is64BitMode()) {
DynamicGPRs = x64::PreserveAll_Dynamic;
DynamicFPRs = x64::PreserveAll_DynamicFPR;
@@ -955,7 +917,8 @@ void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpR
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
}
} else {
}
else {
DynamicGPRs = x32::PreserveAll_Dynamic;
DynamicFPRs = x32::PreserveAll_DynamicFPR;
PreserveSRAMask = x32::PreserveAll_SRAMask;
@@ -989,10 +952,10 @@ void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpR
void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs {};
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs {};
uint32_t PreserveSRAMask {};
uint32_t PreserveSRAFPRMask {};
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs{};
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs{};
uint32_t PreserveSRAMask{};
uint32_t PreserveSRAFPRMask{};
if (EmitterCTX->Config.Is64BitMode()) {
DynamicGPRs = x64::PreserveAll_Dynamic;
@@ -1004,7 +967,8 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
}
} else {
}
else {
DynamicGPRs = x32::PreserveAll_Dynamic;
DynamicFPRs = x32::PreserveAll_DynamicFPR;
PreserveSRAMask = x32::PreserveAll_SRAMask;
@@ -1033,4 +997,4 @@ void Arm64Emitter::Align16B() {
}
}
} // namespace FEXCore::CPU
}
@@ -21,7 +21,6 @@
#endif
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/vector.h>
#include <array>
#include <cstddef>
@@ -37,37 +36,16 @@ namespace FEXCore::CPU {
// Contains the address to the currently available CPU state
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
#ifndef _M_ARM_64EC
// GPR temporaries. Only x3 can be used across spill boundaries
// so if these ever need to change, be very careful about that.
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x0;
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x1;
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x2;
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x3;
constexpr bool TMP_ABIARGS = true;
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
// Vector temporaries
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v0;
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
#else
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x10;
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x11;
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x12;
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x13;
constexpr bool TMP_ABIARGS = false;
// We pin r11/r12 as PF/AF respectively for arm64ec, as r26/r27 are used for SRA.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r9;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r24;
// Vector temporaries
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v16;
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v17;
#endif
// Predicate register temporaries (used when AVX support is enabled)
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
@@ -75,22 +53,21 @@ constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v17;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
protected:
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr = nullptr, size_t size = 0);
FEXCore::Context::ContextImpl* EmitterCTX;
FEXCore::Context::ContextImpl *EmitterCTX;
vixl::aarch64::CPU CPU;
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters {};
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters {};
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters {};
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters {};
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters {};
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase{};
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters{};
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters{};
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters{};
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters{};
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters{};
/**
* @name Register Allocation
@@ -151,19 +128,21 @@ protected:
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
if (SupportsPreserveAllABI) {
SpillForPreserveAllABICall(TmpReg, FPRs);
} else {
SpillStaticRegs(TmpReg, FPRs);
PushDynamicRegsAndLR(TmpReg);
SpillForPreserveAllABICall(TMP1, true);
}
else {
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
}
}
void FillForABICall(bool SupportsPreserveAllABI, bool FPRs = true) {
if (SupportsPreserveAllABI) {
FillForPreserveAllABICall(FPRs);
} else {
FillForPreserveAllABICall(true);
}
else {
PopDynamicRegsAndLR();
FillStaticRegs(FPRs);
FillStaticRegs();
}
}
@@ -183,7 +162,8 @@ protected:
template<typename R, typename... P>
void GenerateRuntimeCall(R (*Function)(P...)) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
uintptr_t FunctionAddress = reinterpret_cast<uintptr_t>(Function);
@@ -201,7 +181,8 @@ protected:
template<typename R, typename... P>
void GenerateIndirectRuntimeCall(ARMEmitter::Register Reg) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
@@ -217,8 +198,8 @@ protected:
template<>
void GenerateIndirectRuntimeCall<float, __uint128_t>(ARMEmitter::Register Reg) {
uintptr_t SimulatorWrapperAddress =
reinterpret_cast<uintptr_t>(&(vixl::aarch64::Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
@@ -245,17 +226,15 @@ protected:
#ifdef VIXL_SIMULATOR
vixl::aarch64::Decoder SimDecoder;
vixl::aarch64::Simulator Simulator;
constexpr static size_t SimulatorStackSize = 8 * 1024 * 1024;
#endif
#ifdef VIXL_DISASSEMBLER
fextl::vector<char> DisasmBuffer;
constexpr static int DISASM_BUFFER_SIZE {256};
fextl::unique_ptr<vixl::aarch64::Disassembler> Disasm;
vixl::aarch64::Disassembler Disasm;
fextl::unique_ptr<vixl::aarch64::Decoder> DisasmDecoder;
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
#endif
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
};
} // namespace FEXCore::CPU
}
@@ -35,10 +35,8 @@ public:
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void adr(FEXCore::ARMEmitter::Register rd, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADR });
void adr(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADR });
constexpr uint32_t Op = 0b0001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
}
@@ -64,10 +62,8 @@ public:
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, Imm);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void adrp(FEXCore::ARMEmitter::Register rd, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADRP });
void adrp(FEXCore::ARMEmitter::Register rd, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::ADRP });
constexpr uint32_t Op = 0b1001'0000 << 24;
DataProcessing_PCRel_Imm(Op, rd, 0);
}
@@ -109,7 +105,7 @@ public:
}
}
void LongAddressGen(FEXCore::ARMEmitter::Register rd, ForwardLabel* Label) {
Label->Insts.emplace_back(SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN });
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN });
// Emit a register index and a nop. These will be backpatched.
dc32(rd.Idx());
nop();
@@ -775,16 +771,6 @@ public:
dc32(Op);
}
void axflag() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0101'1111;
dc32(Op);
}
void xaflag() {
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0011'1111;
dc32(Op);
}
// Conditional compare - register
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
constexpr uint32_t Op = 0b0011'1010'010 << 21;
@@ -60,7 +60,7 @@ public:
}
void sha256su1(FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn, FEXCore::ARMEmitter::VRegister rm) {
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
Crypto3RegSHA(Op, 0b110, rd, rn, rm);
Crypto3RegSHA(Op, 0b100, rd, rn, rm);
}
// Cryptographic two-register SHA
@@ -18,10 +18,8 @@ public:
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void b(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
}
@@ -47,10 +45,8 @@ public:
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bc(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void bc(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
}
@@ -106,10 +102,8 @@ public:
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
void b(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
@@ -137,10 +131,8 @@ public:
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bl(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
void bl(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
@@ -171,10 +163,8 @@ public:
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0100 << 24;
@@ -205,10 +195,8 @@ public:
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0101 << 24;
@@ -238,11 +226,8 @@ public:
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0110 << 24;
@@ -271,11 +256,8 @@ public:
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
@@ -5,102 +5,102 @@
#include <cstring>
namespace FEXCore::ARMEmitter {
class Buffer {
public:
Buffer() {
SetBuffer(nullptr, 0);
}
class Buffer {
public:
Buffer() {
SetBuffer(nullptr, 0);
}
Buffer(uint8_t* Base, uint64_t BaseSize) {
SetBuffer(Base, BaseSize);
}
Buffer(uint8_t* Base, uint64_t BaseSize) {
SetBuffer(Base, BaseSize);
}
void SetBuffer(uint8_t* Base, uint64_t BaseSize) {
BufferBase = Base;
CurrentOffset = BufferBase;
Size = BaseSize;
}
void SetBuffer(uint8_t* Base, uint64_t BaseSize) {
BufferBase = Base;
CurrentOffset = BufferBase;
Size = BaseSize;
}
void dc8(uint8_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc8(uint8_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc16(uint16_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc16(uint16_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc32(uint32_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc32(uint32_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc64(uint64_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void EmitString(const char* String) {
const auto StringLength = strlen(String);
memcpy(CurrentOffset, String, StringLength);
CurrentOffset += StringLength;
}
void dc64(uint64_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void EmitString(const char *String) {
const auto StringLength = strlen(String);
memcpy(CurrentOffset, String, StringLength);
CurrentOffset += StringLength;
}
void Align() {
// Align the buffer to instruction size
auto CurrentAlignment = reinterpret_cast<uint64_t>(CurrentOffset) & 0b11;
if (!CurrentAlignment) {
return;
}
CurrentOffset += 4 - CurrentAlignment;
}
void Align() {
// Align the buffer to instruction size
auto CurrentAlignment = reinterpret_cast<uint64_t>(CurrentOffset) & 0b11;
if (!CurrentAlignment) {
return;
}
CurrentOffset += 4 - CurrentAlignment;
}
template<typename T>
T GetCursorAddress() const {
return reinterpret_cast<T>(CurrentOffset);
}
template<typename T>
T GetCursorAddress() const {
return reinterpret_cast<T>(CurrentOffset);
}
static void ClearICache(void* Begin, std::size_t Length) {
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
}
static void ClearICache(void* Begin, std::size_t Length) {
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
}
size_t GetCursorOffset() const {
return static_cast<size_t>(CurrentOffset - BufferBase);
}
size_t GetCursorOffset() const {
return static_cast<size_t>(CurrentOffset - BufferBase);
}
uint8_t* GetBufferBase() const {
return BufferBase;
}
uint8_t *GetBufferBase() const {
return BufferBase;
}
void CursorIncrement(size_t Size) {
CurrentOffset += Size;
}
void CursorIncrement(size_t Size) {
CurrentOffset += Size;
}
void SetCursorOffset(size_t Offset) {
CurrentOffset = BufferBase + Offset;
}
void SetCursorOffset(size_t Offset) {
CurrentOffset = BufferBase + Offset;
}
uint64_t GetBufferSize() const {
return Size;
}
uint64_t GetBufferSize() const {
return Size;
}
template<typename T>
size_t GetCursorOffsetFromAddress(const T* Address) const {
return static_cast<size_t>(reinterpret_cast<const uint8_t*>(Address) - BufferBase);
}
template<typename T>
size_t GetCursorOffsetFromAddress(const T* Address) const {
return static_cast<size_t>(reinterpret_cast<const uint8_t*>(Address) - BufferBase);
}
protected:
protected:
void ResetBuffer() {
CurrentOffset = BufferBase;
}
void ResetBuffer() {
CurrentOffset = BufferBase;
}
uint8_t* BufferBase;
uint8_t* CurrentOffset;
uint64_t Size;
};
} // namespace FEXCore::ARMEmitter
uint8_t* BufferBase;
uint8_t* CurrentOffset;
uint64_t Size;
};
}
File diff suppressed because it is too large. Load diff
@@ -2121,58 +2121,38 @@ public:
LoadStoreLiteral(Op, prfop, static_cast<uint32_t>(Imm >> 2) & 0x7'FFFF);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::WRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::WRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b0001'1000 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::SRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::SRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b0001'1100 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::XRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::XRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b0101'1000 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::DRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::DRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b0101'1100 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldrsw(FEXCore::ARMEmitter::XRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldrsw(FEXCore::ARMEmitter::XRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b1001'1000 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void ldr(FEXCore::ARMEmitter::QRegister rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void ldr(FEXCore::ARMEmitter::QRegister rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b1001'1100 << 24;
LoadStoreLiteral(Op, rt, 0);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void prfm(FEXCore::ARMEmitter::Prefetch prfop, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::RELATIVE_LOAD });
void prfm(FEXCore::ARMEmitter::Prefetch prfop, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::RELATIVE_LOAD });
constexpr uint32_t Op = 0b1101'1000 << 24;
LoadStoreLiteral(Op, prfop, 0);
}
@@ -3762,12 +3742,7 @@ public:
}
else {
if (MemSrc.MetaType.ImmType.Index == ARMEmitter::IndexType::OFFSET) {
if ((MemSrc.MetaType.ImmType.Imm & 0b111) || MemSrc.MetaType.ImmType.Imm < 0) {
prfum<IndexType::OFFSET>(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
}
else {
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
}
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
}
else {
LOGMAN_MSG_A_FMT("Unexpected loadstore index type");
File diff suppressed because it is too large. Load diff
@@ -6,46 +6,48 @@
#include <utility>
namespace FEXCore {
void BlockSamplingData::DumpBlockData() {
std::fstream Output;
Output.open("output.csv", std::fstream::out | std::fstream::binary);
void BlockSamplingData::DumpBlockData() {
std::fstream Output;
Output.open("output.csv", std::fstream::out | std::fstream::binary);
if (!Output.is_open()) {
return;
}
if (!Output.is_open())
return;
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
for (auto it : SamplingMap) {
if (!it.second->TotalCalls) {
continue;
for (auto it : SamplingMap) {
if (!it.second->TotalCalls)
continue;
Output << "0x" << std::hex << it.first
<< ", " << std::dec << it.second->Min
<< ", " << std::dec << it.second->Max
<< ", " << std::dec << it.second->TotalTime
<< ", " << std::dec << it.second->TotalCalls
<< ", " << std::dec << ((double)it.second->TotalTime / (double)it.second->TotalCalls)
<< std::endl;
}
Output << "0x" << std::hex << it.first << ", " << std::dec << it.second->Min << ", " << std::dec << it.second->Max << ", " << std::dec
<< it.second->TotalTime << ", " << std::dec << it.second->TotalCalls << ", " << std::dec
<< ((double)it.second->TotalTime / (double)it.second->TotalCalls) << std::endl;
Output.close();
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
}
Output.close();
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
}
BlockSamplingData::BlockData* BlockSamplingData::GetBlockData(uint64_t RIP) {
auto it = SamplingMap.find(RIP);
if (it != SamplingMap.end()) {
return it->second;
BlockSamplingData::BlockData *BlockSamplingData::GetBlockData(uint64_t RIP) {
auto it = SamplingMap.find(RIP);
if (it != SamplingMap.end()) {
return it->second;
}
BlockData *NewData = new BlockData{};
memset(NewData, 0, sizeof(BlockData));
NewData->Min = ~0ULL;
SamplingMap[RIP] = NewData;
return NewData;
}
BlockData* NewData = new BlockData {};
memset(NewData, 0, sizeof(BlockData));
NewData->Min = ~0ULL;
SamplingMap[RIP] = NewData;
return NewData;
}
BlockSamplingData::~BlockSamplingData() {
DumpBlockData();
for (auto it : SamplingMap) {
delete it.second;
BlockSamplingData::~BlockSamplingData() {
DumpBlockData();
for (auto it : SamplingMap) {
delete it.second;
}
SamplingMap.clear();
}
SamplingMap.clear();
}
} // namespace FEXCore
@@ -14,7 +14,7 @@ public:
uint64_t TotalCalls;
};
BlockData* GetBlockData(uint64_t RIP);
BlockData *GetBlockData(uint64_t RIP);
~BlockSamplingData();
void DumpBlockData();
@@ -22,4 +22,4 @@ public:
private:
std::unordered_map<uint64_t, BlockData*> SamplingMap;
};
} // namespace FEXCore
}
+345 -352
View File
@@ -2,388 +2,381 @@
#include "FEXCore/IR/IR.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#ifndef _WIN32
#include <sys/prctl.h>
#endif
#include <FEXCore/Core/CPUBackend.h>
namespace FEXCore {
namespace CPU {
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
{0x0706'0504'0302'0100ULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0111B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1001B
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB_UPPER
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_ONE
{0xD49A'784B'CD1B'8AFEULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_LOG2_10
{0xB8AA'3B29'5C17'F0BCULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_LOG2_E
{0xC90F'DAA2'2168'C235ULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_PI
{0x9A20'9A84'FBCF'F799ULL, 0x0000'0000'0000'3FFDULL}, // NAMED_VECTOR_X87_LOG10_2
{0xB172'17F7'D1CF'79ACULL, 0x0000'0000'0000'3FFEULL}, // NAMED_VECTOR_X87_LOG_2
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
{0x0706'0504'0302'0100ULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0111B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1001B
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
};
constexpr static auto PSHUFLW_LUT {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFLW with ARM's TBL (single register) instruction
// PSHUFLW behaviour:
// 16-bit words in [63:48], [47:32], [31:16], [15:0] are selected using the 8-bit Index.
// For 128-bit PSHUFLW, bits [127:64] are identity copied.
constexpr uint64_t IdentityCopyUpper = 0x0f'0e'0d'0c'0b'0a'09'08;
std::array<LUTType, 256> TotalLUT{};
uint64_t WordSelection[4] = {
0x01'00,
0x03'02,
0x05'04,
0x07'06,
};
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] =
(WordSelection[Word0] << 0) |
(WordSelection[Word1] << 16) |
(WordSelection[Word2] << 32) |
(WordSelection[Word3] << 48);
LUT.Val[1] = IdentityCopyUpper;
}
return TotalLUT;
}()
};
constexpr static auto PSHUFHW_LUT {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFHW with ARM's TBL (single register) instruction
// PSHUFHW behaviour:
// 16-bit words in [127:112], [111:96], [95:80], [79:64] are selected using the 8-bit Index.
// Incoming words come from bits [127:64] of the source.
// Bits [63:0] are identity copied.
constexpr uint64_t IdentityCopyLower = 0x07'06'05'04'03'02'01'00;
std::array<LUTType, 256> TotalLUT{};
uint64_t WordSelection[4] = {
0x09'08,
0x0b'0a,
0x0d'0c,
0x0f'0e,
};
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] = IdentityCopyLower;
LUT.Val[1] =
(WordSelection[Word0] << 0) |
(WordSelection[Word1] << 16) |
(WordSelection[Word2] << 32) |
(WordSelection[Word3] << 48);
}
return TotalLUT;
}()
};
constexpr static auto PSHUFD_LUT {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFD with ARM's TBL (single register) instruction
// PSHUFD behaviour:
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
std::array<LUTType, 256> TotalLUT{};
uint64_t WordSelection[4] = {
0x03'02'01'00,
0x07'06'05'04,
0x0b'0a'09'08,
0x0f'0e'0d'0c,
};
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] =
(WordSelection[Word0] << 0) |
(WordSelection[Word1] << 32);
LUT.Val[1] =
(WordSelection[Word2] << 0) |
(WordSelection[Word3] << 32);
}
return TotalLUT;
}()
};
constexpr static auto SHUFPS_LUT {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
// Expectation for this LUT is to simulate SHUFPS with ARM's TBL (two register) instruction.
// SHUFPS behaviour:
// Two 32-bits words from each source are selected from each source in the lower and upper halves of the 128-bit destination.
// Dest[31:0] = Src1[<Word0>]
// Dest[63:32] = Src1[<Word1>]
// Dest[95:64] = Src2[<Word2>]
// Dest[127:96] = Src2[<Word3>]
std::array<LUTType, 256> TotalLUT{};
const uint64_t WordSelectionSrc1[4] = {
0x03'02'01'00,
0x07'06'05'04,
0x0b'0a'09'08,
0x0f'0e'0d'0c,
};
constexpr static auto PSHUFLW_LUT {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFLW with ARM's TBL (single register) instruction
// PSHUFLW behaviour:
// 16-bit words in [63:48], [47:32], [31:16], [15:0] are selected using the 8-bit Index.
// For 128-bit PSHUFLW, bits [127:64] are identity copied.
constexpr uint64_t IdentityCopyUpper = 0x0f'0e'0d'0c'0b'0a'09'08;
std::array<LUTType, 256> TotalLUT {};
uint64_t WordSelection[4] = {
0x01'00,
0x03'02,
0x05'04,
0x07'06,
};
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
// Src2 needs to offset each byte index by 16-bytes to pull from the second source.
const uint64_t WordSelectionSrc2[4] = {
0x03'02'01'00 + (0x10101010),
0x07'06'05'04 + (0x10101010),
0x0b'0a'09'08 + (0x10101010),
0x0f'0e'0d'0c + (0x10101010),
};
LUT.Val[0] = (WordSelection[Word0] << 0) | (WordSelection[Word1] << 16) | (WordSelection[Word2] << 32) | (WordSelection[Word3] << 48);
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[1] = IdentityCopyUpper;
}
return TotalLUT;
}()};
LUT.Val[0] =
(WordSelectionSrc1[Word0] << 0) |
(WordSelectionSrc1[Word1] << 32);
constexpr static auto PSHUFHW_LUT {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFHW with ARM's TBL (single register) instruction
// PSHUFHW behaviour:
// 16-bit words in [127:112], [111:96], [95:80], [79:64] are selected using the 8-bit Index.
// Incoming words come from bits [127:64] of the source.
// Bits [63:0] are identity copied.
constexpr uint64_t IdentityCopyLower = 0x07'06'05'04'03'02'01'00;
std::array<LUTType, 256> TotalLUT {};
uint64_t WordSelection[4] = {
0x09'08,
0x0b'0a,
0x0d'0c,
0x0f'0e,
};
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[1] =
(WordSelectionSrc2[Word2] << 0) |
(WordSelectionSrc2[Word3] << 32);
}
return TotalLUT;
}()
};
LUT.Val[0] = IdentityCopyLower;
constexpr static auto DPPS_MASK {
[]() consteval {
struct LUTType {
uint32_t Val[4];
};
LUT.Val[1] = (WordSelection[Word0] << 0) | (WordSelection[Word1] << 16) | (WordSelection[Word2] << 32) | (WordSelection[Word3] << 48);
}
return TotalLUT;
}()};
constexpr static auto PSHUFD_LUT {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFD with ARM's TBL (single register) instruction
// PSHUFD behaviour:
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
std::array<LUTType, 256> TotalLUT {};
uint64_t WordSelection[4] = {
0x03'02'01'00,
0x07'06'05'04,
0x0b'0a'09'08,
0x0f'0e'0d'0c,
};
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] = (WordSelection[Word0] << 0) | (WordSelection[Word1] << 32);
LUT.Val[1] = (WordSelection[Word2] << 0) | (WordSelection[Word3] << 32);
}
return TotalLUT;
}()};
constexpr static auto SHUFPS_LUT {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
// Expectation for this LUT is to simulate SHUFPS with ARM's TBL (two register) instruction.
// SHUFPS behaviour:
// Two 32-bits words from each source are selected from each source in the lower and upper halves of the 128-bit destination.
// Dest[31:0] = Src1[<Word0>]
// Dest[63:32] = Src1[<Word1>]
// Dest[95:64] = Src2[<Word2>]
// Dest[127:96] = Src2[<Word3>]
std::array<LUTType, 256> TotalLUT {};
const uint64_t WordSelectionSrc1[4] = {
0x03'02'01'00,
0x07'06'05'04,
0x0b'0a'09'08,
0x0f'0e'0d'0c,
};
// Src2 needs to offset each byte index by 16-bytes to pull from the second source.
const uint64_t WordSelectionSrc2[4] = {
0x03'02'01'00 + (0x10101010),
0x07'06'05'04 + (0x10101010),
0x0b'0a'09'08 + (0x10101010),
0x0f'0e'0d'0c + (0x10101010),
};
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] = (WordSelectionSrc1[Word0] << 0) | (WordSelectionSrc1[Word1] << 32);
LUT.Val[1] = (WordSelectionSrc2[Word2] << 0) | (WordSelectionSrc2[Word3] << 32);
}
return TotalLUT;
}()};
constexpr static auto DPPS_MASK {[]() consteval {
struct LUTType {
uint32_t Val[4];
};
std::array<LUTType, 16> TotalLUT {};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto& LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1U;
}
return 0U;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
LUT.Val[2] = GetLUT(i, 2);
LUT.Val[3] = GetLUT(i, 3);
}
return TotalLUT;
}()};
constexpr static auto DPPD_MASK {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
std::array<LUTType, 4> TotalLUT {};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto& LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1ULL;
}
return 0ULL;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
}
return TotalLUT;
}()};
constexpr static auto PBLENDW_LUT {[]() consteval {
struct LUTType {
uint16_t Val[8];
};
// 16-bit words in [127:112], [111:96], [95:80], [79:64], [63:48], [47:32], [31:16], [15:0] are selected using 8-bit swizzle.
// Expectation for this LUT is to simulate PBLENDW with ARM's TBX (one register) instruction.
// PBLENDW behaviour:
// 16-bit words from the source is moved in to the destination based on the bit in the swizzle.
// Dest[15:0] = Swizzle[0] ? Src[15:0] : Dest[15:0]
// Dest[31:16] = Swizzle[1] ? Src[31:16] : Dest[31:16]
// Dest[47:32] = Swizzle[2] ? Src[47:32] : Dest[47:32]
// Dest[63:48] = Swizzle[3] ? Src[63:48] : Dest[63:48]
// Dest[79:64] = Swizzle[4] ? Src[79:64] : Dest[79:64]
// Dest[95:80] = Swizzle[5] ? Src[95:80] : Dest[95:80]
// Dest[111:96] = Swizzle[6] ? Src[111:96] : Dest[111:96]
// Dest[127:112] = Swizzle[7] ? Src[127:112] : Dest[127:112]
std::array<LUTType, 256> TotalLUT {};
const uint16_t WordSelectionSrc[8] = {
0x01'00, 0x03'02, 0x05'04, 0x07'06, 0x09'08, 0x0B'0A, 0x0D'0C, 0x0F'0E,
};
constexpr uint16_t OriginalDest = 0xFF'FF;
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
for (size_t j = 0; j < 8; ++j) {
LUT.Val[j] = ((i >> j) & 1) ? WordSelectionSrc[j] : OriginalDest;
std::array<LUTType, 16> TotalLUT{};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto &LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1U;
}
return 0U;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
LUT.Val[2] = GetLUT(i, 2);
LUT.Val[3] = GetLUT(i, 3);
}
return TotalLUT;
}()
};
constexpr static auto DPPD_MASK {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
std::array<LUTType, 4> TotalLUT{};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto &LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1ULL;
}
return 0ULL;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
}
return TotalLUT;
}()
};
constexpr static auto PBLENDW_LUT {
[]() consteval {
struct LUTType {
uint16_t Val[8];
};
// 16-bit words in [127:112], [111:96], [95:80], [79:64], [63:48], [47:32], [31:16], [15:0] are selected using 8-bit swizzle.
// Expectation for this LUT is to simulate PBLENDW with ARM's TBX (one register) instruction.
// PBLENDW behaviour:
// 16-bit words from the source is moved in to the destination based on the bit in the swizzle.
// Dest[15:0] = Swizzle[0] ? Src[15:0] : Dest[15:0]
// Dest[31:16] = Swizzle[1] ? Src[31:16] : Dest[31:16]
// Dest[47:32] = Swizzle[2] ? Src[47:32] : Dest[47:32]
// Dest[63:48] = Swizzle[3] ? Src[63:48] : Dest[63:48]
// Dest[79:64] = Swizzle[4] ? Src[79:64] : Dest[79:64]
// Dest[95:80] = Swizzle[5] ? Src[95:80] : Dest[95:80]
// Dest[111:96] = Swizzle[6] ? Src[111:96] : Dest[111:96]
// Dest[127:112] = Swizzle[7] ? Src[127:112] : Dest[127:112]
std::array<LUTType, 256> TotalLUT{};
const uint16_t WordSelectionSrc[8] = {
0x01'00,
0x03'02,
0x05'04,
0x07'06,
0x09'08,
0x0B'0A,
0x0D'0C,
0x0F'0E,
};
constexpr uint16_t OriginalDest = 0xFF'FF;
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
for (size_t j = 0; j < 8; ++j) {
LUT.Val[j] = ((i >> j) & 1) ? WordSelectionSrc[j] : OriginalDest;
}
return TotalLUT;
}()};
}
return TotalLUT;
}()
};
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
: ThreadState(ThreadState)
, InitialCodeSize(InitialCodeSize)
, MaxCodeSize(MaxCodeSize) {
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
// Initialize named vector constants.
for (size_t i = 0; i < FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX; ++i) {
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
}
// Initialize named vector constants.
for (size_t i = 0; i < FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX; ++i) {
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
}
// Copy named vector constants.
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
// Copy named vector constants.
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
// Initialize Indexed named vector constants.
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] =
reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] =
reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] =
reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] =
reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] =
reinterpret_cast<uint64_t>(DPPS_MASK.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] =
reinterpret_cast<uint64_t>(DPPD_MASK.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] =
reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
// Initialize Indexed named vector constants.
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] = reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] = reinterpret_cast<uint64_t>(DPPS_MASK.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] = reinterpret_cast<uint64_t>(DPPD_MASK.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] = reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
#ifndef FEX_DISABLE_TELEMETRY
// Fill in telemetry values
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
auto& Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(Telem.GetAddr());
}
// Fill in telemetry values
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
auto &Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(Telem.GetAddr());
}
#endif
}
CPUBackend::~CPUBackend() {
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
}
CPUBackend::~CPUBackend() {
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
}
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
if (CodeBuffers.empty()) {
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
} else {
if (CodeBuffers.size() > 1) {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
for (size_t i = 1; i < CodeBuffers.size(); i++) {
FreeCodeBuffer(CodeBuffers[i]);
}
CodeBuffers.resize(1);
}
// Set the current code buffer to the initial
CurrentCodeBuffer = &CodeBuffers[0];
if (CurrentCodeBuffer->Size != MaxCodeSize) {
FreeCodeBuffer(*CurrentCodeBuffer);
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer->Size *= 1.5;
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
}
}
} else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
if (CodeBuffers.empty()) {
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
}
} else {
if (CodeBuffers.size() > 1) {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
for (size_t i = 1; i < CodeBuffers.size(); i++) {
FreeCodeBuffer(CodeBuffers[i]);
}
CodeBuffers.resize(1);
}
// Set the current code buffer to the initial
CurrentCodeBuffer = &CodeBuffers[0];
return CurrentCodeBuffer;
}
if (CurrentCodeBuffer->Size != MaxCodeSize) {
FreeCodeBuffer(*CurrentCodeBuffer);
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
#ifndef _WIN32
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
//
// MDWE prevents applications from creating RWX memory mappings.
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
//
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
//
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
//
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
// -1: The kernel doesn't support MDWE
// 0: MDWE is supported but disabled
// >0: MDWE is enabled, hence prohibiting RWX mappings
#ifndef PR_GET_MDWE
#define PR_GET_MDWE 66
#endif
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
if (MDWE != -1 && MDWE != 0) {
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
}
#endif
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer->Size *= 1.5;
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
for (auto& Buffer : CodeBuffers) {
auto start = (uintptr_t)Buffer.Ptr;
auto end = start + Buffer.Size;
if (Address >= start && Address < end) {
return true;
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
}
}
return false;
} else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
}
} // namespace CPU
} // namespace FEXCore
return CurrentCodeBuffer;
}
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t *>(
FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
for (auto &Buffer: CodeBuffers) {
auto start = (uintptr_t)Buffer.Ptr;
auto end = start + Buffer.Size;
if (Address >= start && Address < end) {
return true;
}
}
return false;
}
}
}
File diff suppressed because it is too large. Load diff
+82 -94
View File
@@ -14,8 +14,6 @@ namespace Context {
class ContextImpl;
}
uint32_t GetCycleCounterFrequency();
// Debugging define to switch what family of CPU we execute as.
// Might be useful if an application makes an assumption about a CPU.
// #define CPUID_AMD
@@ -30,12 +28,12 @@ private:
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
public:
CPUIDEmu(const FEXCore::Context::ContextImpl* ctx);
// X86 cacheline size effectively has to be hardcoded to 64
// if we report anything differently then applications are likely to break
constexpr static uint64_t CACHELINE_SIZE = 64;
void Init(FEXCore::Context::ContextImpl *ctx);
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) const {
if (Function < Primary.size()) {
const auto Handler = Primary[Function];
@@ -58,13 +56,12 @@ public:
}
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) const {
if (Function == 0x8000'0002U) {
if (Function == 0x8000'0002U)
return Function_8000_0002h(Leaf, CPU % PerCPUData.size());
} else if (Function == 0x8000'0003U) {
else if (Function == 0x8000'0003U)
return Function_8000_0003h(Leaf, CPU % PerCPUData.size());
} else {
else
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
}
}
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) const {
@@ -114,12 +111,10 @@ public:
}
private:
const FEXCore::Context::ContextImpl* CTX;
bool Hybrid {};
uint32_t Cores {};
FEXCore::Context::ContextImpl *CTX;
bool Hybrid{};
FEX_CONFIG_OPT(Cores, THREADS);
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
// XFEATURE_ENABLED_MASK
// Mask that configures what features are enabled on the CPU.
@@ -141,17 +136,11 @@ private:
constexpr static uint64_t XCR0_SSE = 1ULL << 1;
constexpr static uint64_t XCR0_AVX = 1ULL << 2;
struct FeaturesConfig {
uint64_t SHA : 1;
uint64_t _pad : 63;
uint64_t XCR0 {
XCR0_X87 |
XCR0_SSE
};
FeaturesConfig Features {
.SHA = 1,
};
uint64_t XCR0 {XCR0_X87 | XCR0_SSE};
uint32_t SupportsAVX() const {
return (XCR0 & XCR0_AVX) ? 1 : 0;
}
@@ -159,13 +148,13 @@ private:
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf) const;
struct CPUData {
const char* ProductName {};
const char *ProductName{};
#ifdef _M_ARM_64
uint32_t MIDR {};
uint32_t MIDR{};
#endif
bool IsBig {};
bool IsBig{};
};
fextl::vector<CPUData> PerCPUData {};
fextl::vector<CPUData> PerCPUData{};
// Functions
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf) const;
@@ -200,7 +189,6 @@ private:
FEXCore::CPUID::XCRResults XCRFunction_0h() const;
void SetupHostHybridFlag();
void SetupFeatures();
static constexpr size_t PRIMARY_FUNCTION_COUNT = 27;
static constexpr size_t HYPERVISOR_FUNCTION_COUNT = 2;
static constexpr size_t EXTENDED_FUNCTION_COUNT = 32;
@@ -276,74 +264,74 @@ private:
static constexpr std::array<FunctionConstant, PRIMARY_FUNCTION_COUNT> Primary_Constant = {{
// 0: Highest function parameter and ID
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
// 1: Processor info
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 2: Cache and TLB info
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 3: Serial Number(previously), now reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifndef CPUID_AMD
// 4: Deterministic cache parameters for each level
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
#else
// 4: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
// 5: Monitor/mwait
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 6: Thermal and power management
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 7: Extended feature flags
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
// 0x08: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 9: Direct Cache Access information
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0A: Architectural performance monitoring
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0B: Extended topology enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0C: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0D: Processor extended state enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
// 0x0E: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0F: Intel RDT monitoring
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x10: Intel RDT allocation enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x12: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x12: Intel SGX capability enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x13: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x14: Intel Processor trace
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifndef CPUID_AMD
// 0x15: Timestamp counter information
// Doesn't exist on AMD hardware
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#else
// 0x15: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
// 0x16: Processor frequency information
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x17: SoC vendor attribute enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x18: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x19: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifndef CPUID_AMD
// 0x1A: Hybrid Information Sub-leaf
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#else
// 0x1A: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
}};
@@ -356,9 +344,9 @@ private:
static constexpr std::array<FunctionConstant, HYPERVISOR_FUNCTION_COUNT> Hypervisor_Constant = {{
// Hypervisor CPUID information leaf
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// FEX-Emu specific leaf
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
}};
static constexpr std::array<FunctionHandler, EXTENDED_FUNCTION_COUNT> Extended = {
@@ -438,79 +426,79 @@ private:
static constexpr std::array<FunctionConstant, EXTENDED_FUNCTION_COUNT> Extended_Constant = {{
// Largest extended function number
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// Processor vendor
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// Processor brand string
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// Processor brand string continued
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// Processor brand string continued
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifdef CPUID_AMD
// 0x8000'0005: L1 Cache and TLB identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#else
// 0x8000'0005: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
// 0x8000'0006: L2 Cache identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0007: Advanced power management information
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0008: Virtual and physical address sizes
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0009: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000A: SVM Revision
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000B: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000C: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000D: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000E: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000F: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0010: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0011: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0012: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0013: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0014: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0015: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0016: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0017: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0018: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0019: TLB 1GB page identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'001A: Performance optimization identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'001B: Instruction based sampling identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'001C: Lightweight profiling capabilities
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifdef CPUID_AMD
// 0x8000'001D: Cache properties
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
#else
// 0x8000'001D: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
// 0x8000'001E: Extended APIC ID
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'001F: AMD Secure Encryption
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
}};
};
} // namespace FEXCore
}
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,6 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <stdint.h>
namespace FEXCore::CPU {
}
@@ -5,14 +5,12 @@
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/X86HelperGen.h"
#include "Utils/MemberFunctionToPointer.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
@@ -25,15 +23,42 @@
namespace FEXCore::CPU {
static void SleepThread(FEXCore::Context::ContextImpl* CTX, FEXCore::Core::CpuStateFrame* Frame) {
CTX->SyscallHandler->SleepThread(CTX, Frame);
void Dispatcher::SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame) {
auto Thread = Frame->Thread;
--ctx->IdleWaitRefCount;
ctx->IdleWaitCV.notify_all();
Thread->RunningEvents.ThreadSleeping = true;
// Go to sleep
Thread->StartRunning.Wait();
Thread->RunningEvents.Running = true;
++ctx->IdleWaitRefCount;
Thread->RunningEvents.ThreadSleeping = false;
ctx->IdleWaitCV.notify_all();
}
uint64_t Dispatcher::GetCompileBlockPtr() {
using ClassPtrType = void (FEXCore::Context::ContextImpl::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
PtrCast CompileBlockPtr;
CompileBlockPtr.ClassPtr = &FEXCore::Context::ContextImpl::CompileBlockJit;
return CompileBlockPtr.Data;
}
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl* ctx)
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
, CTX {ctx} {
, CTX {ctx}
, config {config} {
EmitDispatcher();
}
@@ -60,11 +85,8 @@ void Dispatcher::EmitDispatcher() {
// }
ARMEmitter::ForwardLabel l_CTX;
ARMEmitter::SingleUseForwardLabel l_Sleep;
#ifdef _M_ARM_64EC
ARMEmitter::SingleUseForwardLabel ExitEC;
#endif
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
ARMEmitter::ForwardLabel l_Sleep;
ARMEmitter::ForwardLabel l_CompileBlock;
// Push all the register we need to save
PushCalleeSavedRegisters();
@@ -81,111 +103,101 @@ void Dispatcher::EmitDispatcher() {
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
FillStaticRegs();
if (config.StaticRegisterAllocation) {
FillStaticRegs();
}
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
ARMEmitter::BiDirectionalLabel FullLookup {};
ARMEmitter::BiDirectionalLabel CallBlock {};
ARMEmitter::BackwardLabel LoopTop {};
ARMEmitter::BiDirectionalLabel FullLookup{};
ARMEmitter::BiDirectionalLabel CallBlock{};
ARMEmitter::BackwardLabel LoopTop{};
Bind(&LoopTop);
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
// Load in our RIP
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
// IMPORTANT: Pointers.Common.ExitFunctionEC callsites/implementations need to be
// adjusted accordingly if this changes.
auto RipReg = TMP3;
// Don't modify x2 since it contains our RIP once the block doesn't exist
auto RipReg = ARMEmitter::XReg::x2;
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
// L1 Cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP1, TMP1, 0);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
sub(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, RipReg);
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &FullLookup);
br(TMP4);
br(ARMEmitter::Reg::r3);
// L1C check failed, do a full lookup
Bind(&FullLookup);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), VirtualMemorySize - 1);
}
else {
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
}
ARMEmitter::ForwardLabel NoBlock;
{
// Offset the address and add to our page pointer
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
lsr(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 12);
// Load the pointer from the offset
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ExtendedType::LSL_64, 3);
// If page pointer is zero then we have no block
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
#ifdef _M_ARM_64EC
// The LSB of an L2 page entry indicates if this page contains EC code
tbnz(TMP1, 0, &ExitEC);
#endif
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 0x0FFF);
// Shift the offset by the size of the block cache entry
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
// The the full LookupCacheEntry with a single LDP.
// Check the guest address first to ensure it maps to the address we are currently at.
// This fixes aliasing problems
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
// If the guest address doesn't match, Compile the block.
sub(TMP2, TMP2, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
sub(ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, RipReg);
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, &NoBlock);
// Check the host address to see if it matches, else compile the block.
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
stp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP3, TMP1);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, 4);
stp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x2, ARMEmitter::Reg::r0);
// Jump to the block
br(TMP4);
br(ARMEmitter::Reg::r3);
}
}
#ifdef _M_ARM_64EC
{
Bind(&ExitEC);
// Target PC is already loaded into TMP3 at the start of the dispatcher
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
}
#endif
{
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
@@ -198,7 +210,8 @@ void Dispatcher::EmitDispatcher() {
{
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
@@ -209,37 +222,32 @@ void Dispatcher::EmitDispatcher() {
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
} else {
GenerateIndirectRuntimeCall<uintptr_t, void *, void *>(ARMEmitter::Reg::r2);
}
else {
blr(ARMEmitter::Reg::r2);
}
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
}
if (config.StaticRegisterAllocation)
FillStaticRegs();
FillStaticRegs();
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
// Trigger segfault if any deferred signals are pending
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
str(ARMEmitter::XReg::zr, TMP2, 0);
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
str(ARMEmitter::XReg::zr, ARMEmitter::XReg::x1, 0);
br(TMP1);
br(ARMEmitter::Reg::r0);
}
// Need to create the block
{
Bind(&NoBlock);
SpillStaticRegs(TMP1);
if (!TMP_ABIARGS) {
mov(ARMEmitter::XReg::x2, TMP3);
}
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
@@ -247,21 +255,22 @@ void Dispatcher::EmitDispatcher() {
ldr(ARMEmitter::XReg::x0, &l_CTX);
mov(ARMEmitter::XReg::x1, STATE);
// x2 contains guest RIP
mov(ARMEmitter::XReg::x3, 0);
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
ldr(ARMEmitter::XReg::x3, &l_CompileBlock);
// X2 contains our guest RIP
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
} else {
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
GenerateIndirectRuntimeCall<void, void *, uint64_t, void *>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
}
FillStaticRegs();
if (config.StaticRegisterAllocation)
FillStaticRegs();
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
// Trigger segfault if any deferred signals are pending
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
@@ -291,7 +300,8 @@ void Dispatcher::EmitDispatcher() {
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
hlt(0);
}
@@ -299,9 +309,10 @@ void Dispatcher::EmitDispatcher() {
{
// Guest SIGTRAP handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
brk(0);
}
@@ -311,7 +322,8 @@ void Dispatcher::EmitDispatcher() {
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
// hlt/udf = SIGILL
// brk = SIGTRAP
@@ -322,7 +334,8 @@ void Dispatcher::EmitDispatcher() {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::r0, 0);
PopCalleeSavedRegisters();
ret();
} else {
}
else {
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
}
@@ -330,7 +343,8 @@ void Dispatcher::EmitDispatcher() {
{
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
// We are pausing, this means the frontend should be waiting for this thread to idle
@@ -341,8 +355,9 @@ void Dispatcher::EmitDispatcher() {
mov(ARMEmitter::XReg::x1, STATE);
ldr(ARMEmitter::XReg::x2, &l_Sleep);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
GenerateIndirectRuntimeCall<void, void *, void *>(ARMEmitter::Reg::r2);
}
else {
blr(ARMEmitter::Reg::r2);
}
@@ -396,58 +411,112 @@ void Dispatcher::EmitDispatcher() {
str(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, State.rip));
// load static regs
FillStaticRegs();
if (config.StaticRegisterAllocation)
FillStaticRegs();
// Now go back to the regular dispatcher loop
b(&LoopTop);
}
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
auto Address = GetCursorAddress<uint64_t>();
{
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
if (!TMP_ABIARGS) {
mov(ARMEmitter::XReg::x0, TMP1);
mov(ARMEmitter::XReg::x1, TMP2);
mov(ARMEmitter::XReg::x2, TMP3);
}
ldr(ARMEmitter::XReg::x3, R, Offset);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
} else {
}
else {
blr(ARMEmitter::Reg::r3);
}
// Result is now in x0
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
return Address;
};
}
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
LUREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
LREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
{
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
Bind(&l_CTX);
dc64(reinterpret_cast<uintptr_t>(CTX));
Bind(&l_Sleep);
dc64(reinterpret_cast<uint64_t>(SleepThread));
Bind(&l_CompileBlock);
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::Context::ContextImpl::CompileBlock);
dc64(PMF.GetConvertedPointer());
dc64(GetCompileBlockPtr());
Start = reinterpret_cast<uint64_t>(DispatchPtr);
End = GetCursorAddress<uint64_t>();
@@ -466,7 +535,7 @@ void Dispatcher::EmitDispatcher() {
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
for (auto PCToDecode = DisasmBegin; PCToDecode < DisasmEnd; PCToDecode += 4) {
DisasmDecoder->Decode(PCToDecode);
auto Output = Disasm->GetOutput();
auto Output = Disasm.GetOutput();
LogMan::Msg::IFmt("{}", Output);
}
}
@@ -474,23 +543,23 @@ void Dispatcher::EmitDispatcher() {
}
#ifdef VIXL_SIMULATOR
void Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
void Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.RunFrom(reinterpret_cast< const vixl::aarch64::Instruction*>(DispatchPtr));
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(DispatchPtr));
}
void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.WriteXRegister(1, RIP);
Simulator.RunFrom(reinterpret_cast< const vixl::aarch64::Instruction*>(CallbackPtr));
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(CallbackPtr));
}
#endif
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread) {
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto& Common = Thread->CurrentFrame->Pointers.Common;
auto &Common = Thread->CurrentFrame->Pointers.Common;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
@@ -503,7 +572,7 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
Common.SignalReturnHandler = SignalHandlerReturnAddress;
Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
auto& AArch64 = Thread->CurrentFrame->Pointers.AArch64;
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
AArch64.LUDIVHandler = LUDIVHandlerAddress;
AArch64.LDIVHandler = LDIVHandlerAddress;
AArch64.LUREMHandler = LUREMHandlerAddress;
@@ -511,8 +580,8 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
}
}
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl* CTX) {
return fextl::make_unique<Dispatcher>(CTX);
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
return fextl::make_unique<Dispatcher>(CTX, Config);
}
} // namespace FEXCore::CPU
}
@@ -2,8 +2,8 @@
#pragma once
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/fextl/memory.h>
#ifdef VIXL_SIMULATOR
@@ -23,7 +23,7 @@ struct GuestSigAction;
namespace FEXCore::Core {
struct CpuStateFrame;
struct InternalThreadState;
} // namespace FEXCore::Core
}
namespace FEXCore::Context {
class ContextImpl;
@@ -31,58 +31,61 @@ class ContextImpl;
namespace FEXCore::CPU {
#define STATE_PTR(STATE_TYPE, FIELD) STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
struct DispatcherConfig {
bool StaticRegisterAllocation = false;
};
#define STATE_PTR(STATE_TYPE, FIELD) \
STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
class Dispatcher final : public Arm64Emitter {
public:
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl* CTX);
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
Dispatcher(FEXCore::Context::ContextImpl* ctx);
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config);
~Dispatcher();
/**
* @name Dispatch Helper functions
* @{ */
uint64_t ThreadStopHandlerAddress {};
uint64_t ThreadStopHandlerAddressSpillSRA {};
uint64_t AbsoluteLoopTopAddress {};
uint64_t AbsoluteLoopTopAddressFillSRA {};
uint64_t ThreadPauseHandlerAddress {};
uint64_t ThreadPauseHandlerAddressSpillSRA {};
uint64_t ExitFunctionLinkerAddress {};
uint64_t SignalHandlerReturnAddress {};
uint64_t SignalHandlerReturnAddressRT {};
uint64_t GuestSignal_SIGILL {};
uint64_t GuestSignal_SIGTRAP {};
uint64_t GuestSignal_SIGSEGV {};
uint64_t IntCallbackReturnAddress {};
uint64_t ThreadStopHandlerAddress{};
uint64_t ThreadStopHandlerAddressSpillSRA{};
uint64_t AbsoluteLoopTopAddress{};
uint64_t AbsoluteLoopTopAddressFillSRA{};
uint64_t ThreadPauseHandlerAddress{};
uint64_t ThreadPauseHandlerAddressSpillSRA{};
uint64_t ExitFunctionLinkerAddress{};
uint64_t SignalHandlerReturnAddress{};
uint64_t SignalHandlerReturnAddressRT{};
uint64_t GuestSignal_SIGILL{};
uint64_t GuestSignal_SIGTRAP{};
uint64_t GuestSignal_SIGSEGV{};
uint64_t IntCallbackReturnAddress{};
uint64_t PauseReturnInstruction {};
uint64_t PauseReturnInstruction{};
/** @} */
uint64_t Start {};
uint64_t End {};
uint64_t Start{};
uint64_t End{};
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread);
#ifdef VIXL_SIMULATOR
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) ;
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
#else
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
DispatchPtr(Frame);
}
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
CallbackPtr(Frame, RIP);
}
#endif
uint16_t GetSRAGPRCount() const {
// PF/AF are the final two SRA registers.
// Only return the SRA for GPRs.
return StaticRegisters.size() - 2;
return StaticRegisters.size();
}
uint16_t GetSRAFPRCount() const {
@@ -90,7 +93,7 @@ public:
}
void GetSRAGPRMapping(uint8_t Mapping[16]) const {
for (size_t i = 0; i < StaticRegisters.size() - 2; ++i) {
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
Mapping[i] = StaticRegisters[i].Idx();
}
}
@@ -101,22 +104,29 @@ public:
}
}
protected:
FEXCore::Context::ContextImpl* CTX;
const DispatcherConfig& GetConfig() const { return config; }
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame);
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
protected:
FEXCore::Context::ContextImpl *CTX;
DispatcherConfig config;
static void SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame);
static uint64_t GetCompileBlockPtr();
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
AsmDispatch DispatchPtr;
JITCallback CallbackPtr;
private:
// Long division helpers
uint64_t LUDIVHandlerAddress {};
uint64_t LDIVHandlerAddress {};
uint64_t LUREMHandlerAddress {};
uint64_t LREMHandlerAddress {};
uint64_t LUDIVHandlerAddress{};
uint64_t LDIVHandlerAddress{};
uint64_t LUREMHandlerAddress{};
uint64_t LREMHandlerAddress{};
void EmitDispatcher();
};
} // namespace FEXCore::CPU
}
File diff suppressed because it is too large. Load diff
+24 -31
View File
@@ -21,10 +21,10 @@ class Decoder final {
public:
// New Frontend decoding
struct DecodedBlocks final {
uint64_t Entry {};
uint64_t NumInstructions {};
FEXCore::X86Tables::DecodedInst* DecodedInstructions;
bool HasInvalidInstruction {};
uint64_t Entry{};
uint64_t NumInstructions{};
FEXCore::X86Tables::DecodedInst *DecodedInstructions;
bool HasInvalidInstruction{};
};
struct DecodedBlockInformation final {
@@ -32,24 +32,19 @@ public:
fextl::vector<DecodedBlocks> Blocks;
};
Decoder(FEXCore::Context::ContextImpl* ctx);
Decoder(FEXCore::Context::ContextImpl *ctx);
~Decoder();
void DecodeInstructionsAtEntry(const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst,
std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, uint64_t MaxInst, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
const DecodedBlockInformation* GetDecodedBlockInfo() const {
DecodedBlockInformation const *GetDecodedBlockInfo() const {
return &BlockInfo;
}
uint64_t DecodedMinAddress {};
uint64_t DecodedMaxAddress {~0ULL};
void SetSectionMaxAddress(uint64_t v) {
SectionMaxAddress = v;
}
void SetExternalBranches(fextl::set<uint64_t>* v) {
ExternalBranches = v;
}
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
void SetExternalBranches(fextl::set<uint64_t> *v) { ExternalBranches = v; }
void DelayedDisownBuffer() {
PoolObject.DelayedDisownBuffer();
@@ -64,8 +59,8 @@ private:
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
};
FEXCore::Context::ContextImpl* CTX;
const FEXCore::HLE::SyscallOSABI OSABI {};
FEXCore::Context::ContextImpl *CTX;
const FEXCore::HLE::SyscallOSABI OSABI{};
bool DecodeInstruction(uint64_t PC);
@@ -75,24 +70,22 @@ private:
uint8_t ReadByte();
uint8_t PeekByte(uint8_t Offset) const;
uint64_t ReadData(uint8_t Size);
void SkipBytes(uint8_t Size) {
InstructionSize += Size;
}
void SkipBytes(uint8_t Size) { InstructionSize += Size; }
bool NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
size_t DecodedSize {};
const uint8_t* InstStream;
uint8_t const *InstStream;
static constexpr size_t MAX_INST_SIZE = 15;
uint8_t InstructionSize;
std::array<uint8_t, MAX_INST_SIZE> Instruction;
FEXCore::X86Tables::DecodedInst* DecodeInst;
FEXCore::X86Tables::DecodedInst *DecodeInst;
// This is for multiblock data tracking
bool SymbolAvailable {false};
@@ -106,21 +99,21 @@ private:
DecodedBlockInformation BlockInfo;
fextl::set<uint64_t> BlocksToDecode;
fextl::set<uint64_t> HasBlocks;
fextl::set<uint64_t>* ExternalBranches {nullptr};
fextl::set<uint64_t> *ExternalBranches {nullptr};
// ModRM rm decoding
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp {
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp{
&FEXCore::Frontend::Decoder::DecodeModRM_64,
&FEXCore::Frontend::Decoder::DecodeModRM_16,
};
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
const uint8_t *AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP);
FEXCORE_TELEMETRY_INIT(VEXOpTelem, TYPE_USES_VEX_OPS);
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
};
} // namespace FEXCore::Frontend
}
+103
View File
@@ -0,0 +1,103 @@
// SPDX-License-Identifier: MIT
/*
$info$
tags: glue|gdbserver
$end_info$
*/
#pragma once
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/Threads.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <atomic>
#include <istream>
#include <memory>
#include <mutex>
#include <stdint.h>
namespace FEXCore {
class GdbServer {
public:
GdbServer(FEXCore::Context::Context *ctx, SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler);
// Public for threading
void GdbServerLoop();
void AlertLibrariesChanged() {
LibraryMapChanged = true;
}
private:
void Break(int signal);
void OpenListenSocket();
fextl::unique_ptr<std::iostream> OpenSocket();
void StartThread();
fextl::string ReadPacket(std::iostream &stream);
void SendPacket(std::ostream &stream, const fextl::string& packet);
void SendACK(std::ostream &stream, bool NACK);
Event ThreadBreakEvent{};
void WaitForThreadWakeup();
struct HandledPacketType {
fextl::string Response{};
enum ResponseType {
TYPE_NONE,
TYPE_UNKNOWN,
TYPE_ACK,
TYPE_NACK,
TYPE_ONLYACK,
TYPE_ONLYNACK,
};
ResponseType TypeResponse{};
};
void SendPacketPair(const HandledPacketType& packetPair);
HandledPacketType ProcessPacket(const fextl::string &packet);
HandledPacketType handleQuery(const fextl::string &packet);
HandledPacketType handleXfer(const fextl::string &packet);
HandledPacketType handleMemory(const fextl::string &packet);
HandledPacketType handleV(const fextl::string& packet);
HandledPacketType handleThreadOp(const fextl::string &packet);
HandledPacketType handleBreakpoint(const fextl::string &packet);
HandledPacketType handleProgramOffsets();
HandledPacketType ThreadAction(char action, uint32_t tid);
fextl::string readRegs();
HandledPacketType readReg(const fextl::string& packet);
FEXCore::Context::Context *CTX;
FEXCore::HLE::SyscallHandler *const SyscallHandler;
fextl::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
fextl::unique_ptr<std::iostream> CommsStream;
std::mutex sendMutex;
bool SettingNoAckMode{false};
bool NoAckMode{false};
bool NonStopMode{false};
fextl::string ThreadString{};
fextl::string OSDataString{};
void buildLibraryMap();
std::atomic<bool> LibraryMapChanged = true;
std::atomic<bool> CoreShuttingDown{};
fextl::string LibraryMapString{};
// Used to keep track of which signals to pass to the guest
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
uint32_t CurrentDebuggingThread{};
int ListenSocket{};
FEX_CONFIG_OPT(Filename, APP_FILENAME);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
};
}
+143 -58
View File
@@ -9,6 +9,14 @@
#ifdef _M_X86_64
#define XBYAK64
#define XBYAK_CUSTOM_ALLOC
#define XBYAK_CUSTOM_MALLOC FEXCore::Allocator::malloc
#define XBYAK_CUSTOM_FREE FEXCore::Allocator::free
#define XBYAK_CUSTOM_SETS
#define XBYAK_STD_UNORDERED_SET fextl::unordered_set
#define XBYAK_STD_UNORDERED_MAP fextl::unordered_map
#define XBYAK_STD_UNORDERED_MULTIMAP fextl::unordered_multimap
#define XBYAK_STD_LIST fextl::list
#define XBYAK_NO_EXCEPTION
#include <FEXCore/fextl/list.h>
#include <FEXCore/fextl/unordered_map.h>
@@ -28,21 +36,23 @@ namespace FEXCore {
[[maybe_unused]] constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
#ifdef _M_ARM_64
[[maybe_unused]]
static uint32_t GetDCZID() {
uint64_t Result {};
__asm("mrs %[Res], DCZID_EL0" : [Res] "=r"(Result));
[[maybe_unused]] static uint32_t GetDCZID() {
uint64_t Result{};
__asm("mrs %[Res], DCZID_EL0"
: [Res] "=r" (Result));
return Result;
}
static uint32_t GetFPCR() {
uint64_t Result {};
__asm("mrs %[Res], FPCR" : [Res] "=r"(Result));
uint64_t Result{};
__asm ("mrs %[Res], FPCR"
: [Res] "=r" (Result));
return Result;
}
static void SetFPCR(uint64_t Value) {
__asm("msr FPCR, %[Value]" ::[Value] "r"(Value));
__asm ("msr FPCR, %[Value]"
:: [Value] "r" (Value));
}
#else
static uint32_t GetDCZID() {
@@ -51,7 +61,7 @@ static uint32_t GetDCZID() {
}
#endif
static void OverrideFeatures(HostFeatures* Features) {
static void OverrideFeatures(HostFeatures *Features) {
// Override features if the user has specifically called for it.
FEX_CONFIG_OPT(HostFeatures, HOSTFEATURES);
if (!HostFeatures()) {
@@ -59,53 +69,130 @@ static void OverrideFeatures(HostFeatures* Features) {
return;
}
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
do { \
#define ENABLE_DISABLE_OPTION(name, enum_name) \
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
const bool AlreadyEnabled = Features->FeatureName; \
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
Features->FeatureName = Result; \
} while (0)
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
#define GET_SINGLE_OPTION(name, enum_name) \
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
ENABLE_DISABLE_OPTION(SupportsAVX, AVX, AVX);
ENABLE_DISABLE_OPTION(SupportsAVX2, AVX2, AVX2);
ENABLE_DISABLE_OPTION(SupportsSVE, SVE, SVE);
ENABLE_DISABLE_OPTION(SupportsAFP, AFP, AFP);
ENABLE_DISABLE_OPTION(SupportsRCPC, LRCPC, LRCPC);
ENABLE_DISABLE_OPTION(SupportsTSOImm9, LRCPC2, LRCPC2);
ENABLE_DISABLE_OPTION(SupportsCSSC, CSSC, CSSC);
ENABLE_DISABLE_OPTION(SupportsPMULL_128Bit, PMULL128, PMULL128);
ENABLE_DISABLE_OPTION(SupportsRAND, RNG, RNG);
ENABLE_DISABLE_OPTION(SupportsCLZERO, CLZERO, CLZERO);
ENABLE_DISABLE_OPTION(SupportsAtomics, Atomics, ATOMICS);
ENABLE_DISABLE_OPTION(SupportsFCMA, FCMA, FCMA);
ENABLE_DISABLE_OPTION(SupportsFlagM, FlagM, FLAGM);
ENABLE_DISABLE_OPTION(SupportsFlagM2, FlagM2, FLAGM2);
ENABLE_DISABLE_OPTION(SupportsRPRES, RPRES, RPRES);
ENABLE_DISABLE_OPTION(SupportsPreserveAllABI, PRESERVEALLABI, PRESERVEALLABI);
GET_SINGLE_OPTION(Crypto, CRYPTO);
ENABLE_DISABLE_OPTION(AVX, AVX);
ENABLE_DISABLE_OPTION(AVX2, AVX2);
ENABLE_DISABLE_OPTION(SVE, SVE);
ENABLE_DISABLE_OPTION(AFP, AFP);
ENABLE_DISABLE_OPTION(LRCPC, LRCPC);
ENABLE_DISABLE_OPTION(LRCPC2, LRCPC2);
ENABLE_DISABLE_OPTION(CSSC, CSSC);
ENABLE_DISABLE_OPTION(PMULL128, PMULL128);
ENABLE_DISABLE_OPTION(RNG, RNG);
ENABLE_DISABLE_OPTION(CLZERO, CLZERO);
ENABLE_DISABLE_OPTION(Atomics, ATOMICS);
ENABLE_DISABLE_OPTION(FCMA, FCMA);
ENABLE_DISABLE_OPTION(FlagM, FLAGM);
ENABLE_DISABLE_OPTION(FlagM2, FLAGM2);
ENABLE_DISABLE_OPTION(Crypto, CRYPTO);
ENABLE_DISABLE_OPTION(RPRES, RPRES);
#undef ENABLE_DISABLE_OPTION
#undef GET_SINGLE_OPTION
if (EnableAVX) {
Features->SupportsAVX = true;
}
else if (DisableAVX) {
Features->SupportsAVX = false;
}
if (EnableAVX2) {
Features->SupportsAVX2 = true;
}
else if (DisableAVX2) {
Features->SupportsAVX2 = false;
}
if (EnableSVE) {
Features->SupportsSVE = true;
}
else if (DisableSVE) {
Features->SupportsSVE = false;
}
if (EnableAFP) {
Features->SupportsAFP = true;
}
else if (DisableAFP) {
Features->SupportsAFP = false;
}
if (EnableLRCPC) {
Features->SupportsRCPC = true;
}
else if (DisableLRCPC) {
Features->SupportsRCPC = false;
}
if (EnableLRCPC2) {
Features->SupportsTSOImm9 = true;
}
else if (DisableLRCPC2) {
Features->SupportsTSOImm9 = false;
}
if (EnableCSSC) {
Features->SupportsCSSC = true;
}
else if (DisableCSSC) {
Features->SupportsCSSC = false;
}
if (EnablePMULL128) {
Features->SupportsPMULL_128Bit = true;
}
else if (DisablePMULL128) {
Features->SupportsPMULL_128Bit = false;
}
if (EnableRNG) {
Features->SupportsRAND = true;
}
else if (DisableRNG) {
Features->SupportsRAND = false;
}
if (EnableCLZERO) {
Features->SupportsCLZERO = true;
}
else if (DisableCLZERO) {
Features->SupportsCLZERO = false;
}
if (EnableAtomics) {
Features->SupportsAtomics = true;
}
else if (DisableAtomics) {
Features->SupportsAtomics = false;
}
if (EnableFCMA) {
Features->SupportsFCMA = true;
}
else if (DisableFCMA) {
Features->SupportsFCMA = false;
}
if (EnableFlagM) {
Features->SupportsFlagM = true;
}
else if (DisableFlagM) {
Features->SupportsFlagM = false;
}
if (EnableFlagM2) {
Features->SupportsFlagM2 = true;
}
else if (DisableFlagM2) {
Features->SupportsFlagM2 = false;
}
if (EnableCrypto) {
Features->SupportsAES = true;
Features->SupportsCRC = true;
Features->SupportsSHA = true;
Features->SupportsPMULL_128Bit = true;
} else if (DisableCrypto) {
}
else if (DisableCrypto) {
Features->SupportsAES = false;
Features->SupportsCRC = false;
Features->SupportsSHA = false;
Features->SupportsPMULL_128Bit = false;
}
if (EnableRPRES) {
Features->SupportsRPRES = true;
}
else if (DisableRPRES) {
Features->SupportsRPRES = false;
}
}
HostFeatures::HostFeatures() {
@@ -124,7 +211,6 @@ HostFeatures::HostFeatures() {
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) && Features.Has(vixl::CPUFeatures::Feature::kSHA2);
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
@@ -147,18 +233,18 @@ HostFeatures::HostFeatures() {
SupportsAVX = true;
#else
SupportsSVE = Features.Has(vixl::CPUFeatures::Feature::kSVE);
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) && vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
#endif
// TODO: AVX2 is currently unsupported. Disable until the remaining features are implemented.
SupportsAVX2 = false;
SupportsSHA = true;
SupportsBMI1 = true;
SupportsBMI2 = true;
SupportsCLWB = true;
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
SupportsAFP = false;
// RPRES has a dependency on AFP. Disable it until AFP is enabled.
SupportsRPRES = false;
if (!SupportsAtomics) {
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
@@ -168,19 +254,21 @@ HostFeatures::HostFeatures() {
// We need to get the CPU's cache line size
// We expect sane targets that have correct cacheline sizes across clusters
uint64_t CTR;
__asm volatile("mrs %[ctr], ctr_el0" : [ctr] "=r"(CTR));
__asm volatile ("mrs %[ctr], ctr_el0"
: [ctr] "=r"(CTR));
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
ICacheLineSize = 4 << (CTR & 0xF);
// Test if this CPU supports float exception trapping by attempting to enable
// On unsupported these bits are architecturally defined as RAZ/WI
constexpr uint32_t ExceptionEnableTraps = (1U << 8) | // Invalid Operation float exception trap enable
(1U << 9) | // Divide by zero float exception trap enable
(1U << 10) | // Overflow float exception trap enable
(1U << 11) | // Underflow float exception trap enable
(1U << 12) | // Inexact float exception trap enable
(1U << 15); // Input Denormal float exception trap enable
constexpr uint32_t ExceptionEnableTraps =
(1U << 8) | // Invalid Operation float exception trap enable
(1U << 9) | // Divide by zero float exception trap enable
(1U << 10) | // Overflow float exception trap enable
(1U << 11) | // Underflow float exception trap enable
(1U << 12) | // Inexact float exception trap enable
(1U << 15); // Input Denormal float exception trap enable
uint32_t OriginalFPCR = GetFPCR();
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
@@ -195,8 +283,6 @@ HostFeatures::HostFeatures() {
#ifdef VIXL_SIMULATOR
// simulator doesn't support dc(ZVA)
SupportsCLZERO = false;
// Simulator doesn't support SHA
SupportsSHA = false;
#else
// Check if we can support cacheline clears
uint32_t DCZID = GetDCZID();
@@ -215,7 +301,7 @@ HostFeatures::HostFeatures() {
ICacheLineSize = 64U;
#if !defined(VIXL_SIMULATOR)
Xbyak::util::Cpu X86Features {};
Xbyak::util::Cpu X86Features{};
SupportsAES = X86Features.has(Xbyak::util::Cpu::tAESNI);
SupportsCRC = X86Features.has(Xbyak::util::Cpu::tSSE42);
SupportsRAND = X86Features.has(Xbyak::util::Cpu::tRDRAND) && X86Features.has(Xbyak::util::Cpu::tRDSEED);
@@ -246,7 +332,6 @@ HostFeatures::HostFeatures() {
SupportsFloatExceptions = true;
#endif
#endif
SupportsPreserveAllABI = FEXCORE_HAS_PRESERVE_ALL_ATTR;
OverrideFeatures(this);
}
} // namespace FEXCore
}
@@ -0,0 +1,2 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/Debug/InternalThreadState.h>
@@ -2,36 +2,48 @@
#include "Common/SoftFloat.h"
#include "Common/SoftFloat-3e/softfloat.h"
#include <FEXCore/IR/IR.h>
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
#include "Interface/IR/IR.h"
namespace FEXCore::CPU {
FEXCORE_PRESERVE_ALL_ATTR static void LoadDeferredFCW(uint16_t NewFCW) {
FEXCORE_PRESERVE_ALL_ATTR
static void LoadDeferredFCW(uint16_t NewFCW) {
auto PC = (NewFCW >> 8) & 3;
switch (PC) {
case 0: extF80_roundingPrecision = 32; break;
case 2: extF80_roundingPrecision = 64; break;
case 3: extF80_roundingPrecision = 80; break;
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
switch(PC) {
case 0: extF80_roundingPrecision = 32; break;
case 2: extF80_roundingPrecision = 64; break;
case 3: extF80_roundingPrecision = 80; break;
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
}
auto RC = (NewFCW >> 10) & 3;
switch (RC) {
case 0: softfloat_roundingMode = softfloat_round_near_even; break;
case 1: softfloat_roundingMode = softfloat_round_min; break;
case 2: softfloat_roundingMode = softfloat_round_max; break;
case 3: softfloat_roundingMode = softfloat_round_minMag; break;
switch(RC) {
case 0:
softfloat_roundingMode = softfloat_round_near_even;
break;
case 1:
softfloat_roundingMode = softfloat_round_min;
break;
case 2:
softfloat_roundingMode = softfloat_round_max;
break;
case 3:
softfloat_roundingMode = softfloat_round_minMag;
break;
}
}
template<>
struct OpHandlers<IR::OP_F80CVTTO> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t NewFCW, float src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle4(uint16_t NewFCW, float src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle8(uint16_t NewFCW, double src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle8(uint16_t NewFCW, double src) {
LoadDeferredFCW(NewFCW);
return src;
}
@@ -40,20 +52,24 @@ struct OpHandlers<IR::OP_F80CVTTO> {
template<>
struct OpHandlers<IR::OP_F80CMP> {
template<uint32_t Flags>
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static uint64_t handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
bool eq, lt, nan;
uint64_t ResultFlags = 0;
X80SoftFloat::FCMP(Src1, Src2, &eq, &lt, &nan);
if (Flags & (1 << IR::FCMP_FLAG_LT) && lt) {
if (Flags & (1 << IR::FCMP_FLAG_LT) &&
lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) && nan) {
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
nan) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
if (Flags & (1 << IR::FCMP_FLAG_EQ) && eq) {
if (Flags & (1 << IR::FCMP_FLAG_EQ) &&
eq) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
return ResultFlags;
@@ -62,12 +78,14 @@ struct OpHandlers<IR::OP_F80CMP> {
template<>
struct OpHandlers<IR::OP_F80CVT> {
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static float handle4(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static double handle8(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
@@ -75,22 +93,26 @@ struct OpHandlers<IR::OP_F80CVT> {
template<>
struct OpHandlers<IR::OP_F80CVTINT> {
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int16_t handle2(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t handle4(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int64_t handle8(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int16_t handle2t(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
@@ -103,12 +125,14 @@ struct OpHandlers<IR::OP_F80CVTINT> {
}
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t handle4t(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return extF80_to_i32(src, softfloat_round_minMag, false);
}
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int64_t handle8t(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return extF80_to_i64(src, softfloat_round_minMag, false);
}
@@ -116,12 +140,14 @@ struct OpHandlers<IR::OP_F80CVTINT> {
template<>
struct OpHandlers<IR::OP_F80CVTTOINT> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle2(uint16_t NewFCW, int16_t src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle2(uint16_t NewFCW, int16_t src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t NewFCW, int32_t src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle4(uint16_t NewFCW, int32_t src) {
LoadDeferredFCW(NewFCW);
return src;
}
@@ -129,7 +155,8 @@ struct OpHandlers<IR::OP_F80CVTTOINT> {
template<>
struct OpHandlers<IR::OP_F80ROUND> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FRNDINT(Src1);
}
@@ -137,7 +164,8 @@ struct OpHandlers<IR::OP_F80ROUND> {
template<>
struct OpHandlers<IR::OP_F80F2XM1> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::F2XM1(Src1);
}
@@ -145,7 +173,8 @@ struct OpHandlers<IR::OP_F80F2XM1> {
template<>
struct OpHandlers<IR::OP_F80TAN> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FTAN(Src1);
}
@@ -153,7 +182,8 @@ struct OpHandlers<IR::OP_F80TAN> {
template<>
struct OpHandlers<IR::OP_F80SQRT> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSQRT(Src1);
}
@@ -161,7 +191,8 @@ struct OpHandlers<IR::OP_F80SQRT> {
template<>
struct OpHandlers<IR::OP_F80SIN> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSIN(Src1);
}
@@ -169,7 +200,8 @@ struct OpHandlers<IR::OP_F80SIN> {
template<>
struct OpHandlers<IR::OP_F80COS> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FCOS(Src1);
}
@@ -177,7 +209,8 @@ struct OpHandlers<IR::OP_F80COS> {
template<>
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FXTRACT_EXP(Src1);
}
@@ -185,7 +218,8 @@ struct OpHandlers<IR::OP_F80XTRACT_EXP> {
template<>
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FXTRACT_SIG(Src1);
}
@@ -193,7 +227,8 @@ struct OpHandlers<IR::OP_F80XTRACT_SIG> {
template<>
struct OpHandlers<IR::OP_F80ADD> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FADD(Src1, Src2);
}
@@ -201,7 +236,8 @@ struct OpHandlers<IR::OP_F80ADD> {
template<>
struct OpHandlers<IR::OP_F80SUB> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSUB(Src1, Src2);
}
@@ -209,7 +245,8 @@ struct OpHandlers<IR::OP_F80SUB> {
template<>
struct OpHandlers<IR::OP_F80MUL> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FMUL(Src1, Src2);
}
@@ -217,7 +254,8 @@ struct OpHandlers<IR::OP_F80MUL> {
template<>
struct OpHandlers<IR::OP_F80DIV> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FDIV(Src1, Src2);
}
@@ -225,7 +263,8 @@ struct OpHandlers<IR::OP_F80DIV> {
template<>
struct OpHandlers<IR::OP_F80FYL2X> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FYL2X(Src1, Src2);
}
@@ -233,7 +272,8 @@ struct OpHandlers<IR::OP_F80FYL2X> {
template<>
struct OpHandlers<IR::OP_F80ATAN> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FATAN(Src1, Src2);
}
@@ -241,7 +281,8 @@ struct OpHandlers<IR::OP_F80ATAN> {
template<>
struct OpHandlers<IR::OP_F80FPREM1> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FREM1(Src1, Src2);
}
@@ -249,7 +290,8 @@ struct OpHandlers<IR::OP_F80FPREM1> {
template<>
struct OpHandlers<IR::OP_F80FPREM> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FREM(Src1, Src2);
}
@@ -257,7 +299,8 @@ struct OpHandlers<IR::OP_F80FPREM> {
template<>
struct OpHandlers<IR::OP_F80SCALE> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSCALE(Src1, Src2);
}
@@ -331,14 +374,15 @@ template<>
struct OpHandlers<IR::OP_F64SCALE> {
static double handle(uint16_t NewFCW, double src1, double src2) {
LoadDeferredFCW(NewFCW);
double trunc = (double)(int64_t)(src2); // truncate
double trunc = (double)(int64_t)(src2); //truncate
return src1 * exp2(trunc);
}
};
template<>
struct OpHandlers<IR::OP_F80BCDSTORE> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
bool Negative = Src1.Sign;
@@ -349,7 +393,7 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
uint64_t Tmp = Src1;
X80SoftFloat Rv;
uint8_t* BCD = reinterpret_cast<uint8_t*>(&Rv);
uint8_t *BCD = reinterpret_cast<uint8_t*>(&Rv);
memset(BCD, 0, 10);
for (size_t i = 0; i < 9; ++i) {
@@ -379,10 +423,11 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
template<>
struct OpHandlers<IR::OP_F80BCDLOAD> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src) {
LoadDeferredFCW(NewFCW);
uint8_t* Src1 = reinterpret_cast<uint8_t*>(&Src);
uint64_t BCD {};
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
// Each 4bit split is a digit
// Only 0-9 is supported, A-F results in undefined data
@@ -68,7 +68,8 @@ namespace FEXCore::CPU {
//
// 5. Done.
//
template<IR::IROps Op>
struct OpHandlers {};
template <IR::IROps Op>
struct OpHandlers {
};
} // namespace FEXCore::CPU
@@ -10,23 +10,23 @@
namespace FEXCore::CPU {
template<typename R, typename... Args>
static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_UNKNOWN, (void*)fn, HandlerIndex, false};
}
template<>
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_I16_F64, (void*)fn, HandlerIndex, false};
}
template<>
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_I16_F64_F64, (void*)fn, HandlerIndex, false};
}
void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4);
Info[Core::OPINDEX_F80CVT_8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8);
Info[Core::OPINDEX_F80CVTINT_2] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2);
@@ -55,8 +55,8 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
Info[Core::OPINDEX_F80COS] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80COS>::handle);
Info[Core::OPINDEX_F80XTRACT_EXP] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle);
Info[Core::OPINDEX_F80XTRACT_SIG] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle);
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle);
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle);
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle);
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle);
// Binary
Info[Core::OPINDEX_F80ADD] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80ADD>::handle);
@@ -85,123 +85,126 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
Info[Core::OPINDEX_VPCMPISTRX] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle);
}
bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::IROp_Header* IROp, FallbackInfo* Info) {
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
uint8_t OpSize = IROp->Size;
switch (IROp->Op) {
case IR::OP_F80CVTTO: {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch(IROp->Op) {
case IR::OP_F80CVTTO: {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch (Op->SrcSize) {
case 4: {
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
return true;
}
case 8: {
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CVT: {
switch (OpSize) {
case 4: {
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
return true;
}
case 8: {
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CVTINT: {
auto Op = IROp->C<IR::IROp_F80CVTInt>();
switch (OpSize) {
case 2: {
if (Op->Truncate) {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
SupportsPreserveAllABI};
} else {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, SupportsPreserveAllABI};
switch (Op->SrcSize) {
case 4: {
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
case 8: {
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
return true;
break;
}
case 4: {
if (Op->Truncate) {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
SupportsPreserveAllABI};
} else {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, SupportsPreserveAllABI};
case IR::OP_F80CVT: {
switch (OpSize) {
case 4: {
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
case 8: {
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
return true;
break;
}
case 8: {
if (Op->Truncate) {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
SupportsPreserveAllABI};
} else {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, SupportsPreserveAllABI};
case IR::OP_F80CVTINT: {
auto Op = IROp->C<IR::IROp_F80CVTInt>();
switch (OpSize) {
case 2: {
if (Op->Truncate) {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
else {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
return true;
}
case 4: {
if (Op->Truncate) {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
else {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
return true;
}
case 8: {
if (Op->Truncate) {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
else {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CMP: {
auto Op = IROp->C<IR::IROp_F80Cmp>();
static constexpr std::array handlers{
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
};
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags), FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CMP: {
auto Op = IROp->C<IR::IROp_F80Cmp>();
static constexpr std::array handlers {
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
};
case IR::OP_F80CVTTOINT: {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags),
SupportsPreserveAllABI};
return true;
}
case IR::OP_F80CVTTOINT: {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->SrcSize) {
case 2: {
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
return true;
switch (Op->SrcSize) {
case 2: {
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
case 4: {
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case 4: {
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
#define COMMON_UNARY_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
return true; \
}
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, FEXCORE_HAS_PRESERVE_ALL_ATTR}; \
return true; \
}
#define COMMON_BINARY_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
return true; \
}
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, FEXCORE_HAS_PRESERVE_ALL_ATTR}; \
return true; \
}
#define COMMON_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
return true; \
}
case IR::OP_F64##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
return true; \
}
// Unary
COMMON_UNARY_X87_OP(ROUND)
@@ -239,20 +242,20 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
COMMON_F64_OP(FPREM)
COMMON_F64_OP(SCALE)
// SSE4.2 Fallbacks
case IR::OP_VPCMPESTRX:
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX,
SupportsPreserveAllABI};
return true;
case IR::OP_VPCMPISTRX:
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
return true;
// SSE4.2 Fallbacks
case IR::OP_VPCMPESTRX:
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
case IR::OP_VPCMPISTRX:
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
default: break;
default:
break;
}
return false;
}
} // namespace FEXCore::CPU
}
@@ -15,9 +15,9 @@ namespace FEXCore::CPU {
template<>
struct OpHandlers<IR::OP_VPCMPESTRX> {
enum class AggregationOp {
EqualAny = 0b00,
Ranges = 0b01,
EqualEach = 0b10,
EqualAny = 0b00,
Ranges = 0b01,
EqualEach = 0b10,
EqualOrdered = 0b11,
};
@@ -35,7 +35,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
NegativeMasked,
};
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(uint64_t RAX, uint64_t RDX, __uint128_t lhs, __uint128_t rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t handle(uint64_t RAX, uint64_t RDX, __uint128_t lhs, __uint128_t rhs, uint16_t control) {
// Subtract by 1 in order to make validity limits 0-based
const auto valid_lhs = GetExplicitLength(RAX, control) - 1;
const auto valid_rhs = GetExplicitLength(RDX, control) - 1;
@@ -44,7 +45,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
}
// Main PCMPXSTRX algorithm body. Allows for reuse with both implicit and explicit length variants.
FEXCORE_PRESERVE_ALL_ATTR static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
const uint32_t aggregation = PerformAggregation(lhs, valid_lhs, rhs, valid_rhs, control);
const int32_t upper_limit = (16 >> (control & 1)) - 1;
@@ -68,7 +70,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
return result | (flags << 16);
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetExplicitLength(uint64_t reg, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t GetExplicitLength(uint64_t reg, uint16_t control) {
// Bit 8 controls whether or not the reg value is 64-bit or 32-bit.
int64_t value = 0;
if (((control >> 8) & 1) != 0) {
@@ -91,50 +94,62 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
return std::abs(static_cast<int>(value));
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetElement(const __uint128_t& vec, int32_t index, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t GetElement(const __uint128_t& vec, int32_t index, uint16_t control) {
const auto* vec_ptr = reinterpret_cast<const uint8_t*>(&vec);
// Control bits [1:0] define the data type being dealt with.
switch (static_cast<SourceData>(control & 0b11)) {
case SourceData::U8: return static_cast<int32_t>(vec_ptr[index]);
case SourceData::U8:
return static_cast<int32_t>(vec_ptr[index]);
case SourceData::U16: {
uint16_t value {};
uint16_t value{};
std::memcpy(&value, vec_ptr + (sizeof(uint16_t) * static_cast<size_t>(index)), sizeof(value));
return value;
}
case SourceData::S8: return static_cast<int8_t>(vec_ptr[index]);
case SourceData::S8:
return static_cast<int8_t>(vec_ptr[index]);
case SourceData::S16:
default: {
int16_t value {};
int16_t value{};
std::memcpy(&value, vec_ptr + (sizeof(int16_t) * static_cast<size_t>(index)), sizeof(value));
return value;
}
}
}
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
PerformAggregation(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t PerformAggregation(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
switch (static_cast<AggregationOp>((control >> 2) & 0b11)) {
case AggregationOp::EqualAny: return HandleEqualAny(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::Ranges: return HandleRanges(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::EqualEach: return HandleEqualEach(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::EqualAny:
return HandleEqualAny(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::Ranges:
return HandleRanges(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::EqualEach:
return HandleEqualEach(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::EqualOrdered:
default: return HandleEqualOrdered(lhs, valid_lhs, rhs, valid_rhs, control);
default:
return HandleEqualOrdered(lhs, valid_lhs, rhs, valid_rhs, control);
}
}
FEXCORE_PRESERVE_ALL_ATTR static uint32_t HandlePolarity(uint32_t value, uint16_t control, int upper_limit, int valid_rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandlePolarity(uint32_t value, uint16_t control, int upper_limit, int valid_rhs) {
switch (static_cast<Polarity>((control >> 4) & 0b11)) {
case Polarity::Negative: return value ^ ((2U << upper_limit) - 1);
case Polarity::NegativeMasked: return value ^ ((1U << (valid_rhs + 1)) - 1);
case Polarity::Positive:
case Polarity::PositiveMasked:
default:
// Both positive masking and positive polarity are documented
// as both being equivalent to "IntRes2 = IntRes1", where IntRes1
// is our 'value' parameter, so we don't need to do anything in
// these cases except return the same value.
return value;
case Polarity::Negative:
return value ^ ((2U << upper_limit) - 1);
case Polarity::NegativeMasked:
return value ^ ((1U << (valid_rhs + 1)) - 1);
case Polarity::Positive:
case Polarity::PositiveMasked:
default:
// Both positive masking and positive polarity are documented
// as both being equivalent to "IntRes2 = IntRes1", where IntRes1
// is our 'value' parameter, so we don't need to do anything in
// these cases except return the same value.
return value;
}
}
@@ -160,8 +175,10 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
// │
// 'c' match ────────┘
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
HandleEqualAny(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandleEqualAny(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
uint32_t result = 0;
for (int j = valid_rhs; j >= 0; j--) {
@@ -205,8 +222,10 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
// │
// 'Z' >= 'z' && 'A' <= 'z' ──────────┘
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
HandleRanges(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandleRanges(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
uint32_t result = 0;
for (int j = valid_rhs; j >= 0; j--) {
@@ -256,8 +275,10 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
// │
// 'a' == 'a' ──────────┘
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
HandleEqualEach(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandleEqualEach(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
const auto upper_limit = (16 >> (control & 1)) - 1;
const auto max_valid = std::max(valid_lhs, valid_rhs);
const auto min_valid = std::min(valid_lhs, valid_rhs);
@@ -309,8 +330,10 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
// │
// At index 0 ──────────┘
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
HandleEqualOrdered(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandleEqualOrdered(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
const auto upper_limit = (16 >> (control & 1)) - 1;
// Edge case!
@@ -322,7 +345,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
}
uint32_t result = 0;
const int initial = valid_rhs == upper_limit ? valid_rhs : valid_rhs - valid_lhs;
const int initial = valid_rhs == upper_limit ? valid_rhs
: valid_rhs - valid_lhs;
for (int j = initial; j >= 0; j--) {
result <<= 1;
@@ -355,7 +379,8 @@ struct OpHandlers<IR::OP_VPCMPISTRX> {
// to be the max length possible for the given character size specified
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
// Subtract by 1 in order to make validity limits 0-based
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
@@ -363,7 +388,8 @@ struct OpHandlers<IR::OP_VPCMPISTRX> {
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
const auto is_using_words = (control & 1) != 0;
@@ -373,7 +399,7 @@ struct OpHandlers<IR::OP_VPCMPISTRX> {
const auto get_word = [data_u8](int32_t index) {
const auto* src = data_u8 + (index * sizeof(uint16_t));
uint16_t element {};
uint16_t element{};
std::memcpy(&element, src, sizeof(uint16_t));
return element;
};
@@ -7,43 +7,44 @@
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::IR {
class IRListView;
struct IROp_Header;
} // namespace FEXCore::IR
class IRListView;
struct IROp_Header;
}
namespace FEXCore::CPU {
enum FallbackABI {
FABI_UNKNOWN,
FABI_F80_I16_F32,
FABI_F80_I16_F64,
FABI_F80_I16_I16,
FABI_F80_I16_I32,
FABI_F32_I16_F80,
FABI_F64_I16_F80,
FABI_F64_I16_F64,
FABI_F64_I16_F64_F64,
FABI_I16_I16_F80,
FABI_I32_I16_F80,
FABI_I64_I16_F80,
FABI_I64_I16_F80_F80,
FABI_F80_I16_F80,
FABI_F80_I16_F80_F80,
FABI_I32_I64_I64_I128_I128_I16,
FABI_I32_I128_I128_I16,
};
enum FallbackABI {
FABI_UNKNOWN,
FABI_F80_I16_F32,
FABI_F80_I16_F64,
FABI_F80_I16_I16,
FABI_F80_I16_I32,
FABI_F32_I16_F80,
FABI_F64_I16_F80,
FABI_F64_I16_F64,
FABI_F64_I16_F64_F64,
FABI_I16_I16_F80,
FABI_I32_I16_F80,
FABI_I64_I16_F80,
FABI_I64_I16_F80_F80,
FABI_F80_I16_F80,
FABI_F80_I16_F80_F80,
FABI_I32_I64_I64_I128_I128_I16,
FABI_I32_I128_I128_I16,
};
struct FallbackInfo {
FallbackABI ABI;
void* fn;
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
bool SupportsPreserveAllABI;
};
struct FallbackInfo {
FallbackABI ABI;
void *fn;
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
bool SupportsPreserveAllABI;
};
class InterpreterOps {
public:
static void FillFallbackIndexPointers(uint64_t* Info);
static bool GetFallbackHandler(bool SupportsPreserveAllABI, const IR::IROp_Header* IROp, FallbackInfo* Info);
};
class InterpreterOps {
public:
static void FillFallbackIndexPointers(uint64_t *Info);
static bool GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info);
};
} // namespace FEXCore::CPU
File diff suppressed because it is too large. Load diff
@@ -13,19 +13,21 @@ namespace FEXCore::CPU {
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
switch (Op) {
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
break;
default:
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
break;
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op)); break;
}
return ~0ULL;
}
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum) {
Relocation MoveABI {};
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum &Sum) {
Relocation MoveABI{};
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
auto CurrentCursor = GetCursorAddress<uint8_t *>();
MoveABI.NamedThunkMove.Offset = CurrentCursor - CodeData.BlockBegin;
MoveABI.NamedThunkMove.Symbol = Sum;
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
@@ -41,25 +43,22 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
Arm64JITCore::NamedSymbolLiteralPair Lit {
.Lit = Pointer,
.MoveABI =
{
.NamedSymbolLiteral =
{
.Header =
{
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
},
.Symbol = Op,
.Offset = 0,
},
.MoveABI = {
.NamedSymbolLiteral = {
.Header = {
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
},
.Symbol = Op,
.Offset = 0,
},
},
};
return Lit;
}
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
auto CurrentCursor = GetCursorAddress<uint8_t *>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
Bind(&Lit.Loc);
@@ -68,10 +67,10 @@ void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
}
void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant) {
Relocation MoveABI {};
Relocation MoveABI{};
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
auto CurrentCursor = GetCursorAddress<uint8_t *>();
MoveABI.GuestRIPMove.Offset = CurrentCursor - CodeData.BlockBegin;
MoveABI.GuestRIPMove.GuestRIP = Constant;
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
@@ -80,54 +79,54 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
Relocations.emplace_back(MoveABI);
}
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
const char* EntryRelocations) {
size_t DataIndex {};
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
size_t DataIndex{};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
dc64(Pointer);
// Generate a literal so we can place it
dc64(Pointer);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
// XXX: Reenable once the JIT Object Cache is upstream
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
return false;
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
// XXX: Reenable once the JIT Object Cache is upstream
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
}
}
return true;
}
} // namespace FEXCore::CPU
}
@@ -11,7 +11,7 @@ $end_info$
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CASPair>();
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Wrong element size");
@@ -29,20 +29,13 @@ DEF_OP(CASPair) {
caspal(EmitSize, TMP3, TMP4, Desired.first, Desired.second, MemSrc);
mov(EmitSize, Dst.first, TMP3.R());
mov(EmitSize, Dst.second, TMP4.R());
} else {
// Save NZCV so we don't have to mark this op as clobbering NZCV (the
// SupportsAtomics does not clobber atomics and this !SupportsAtomics path
// is so slow it's not worth the complexity of splitting the IR op.). We
// clobber NZCV inside the hot loop and we can't replace cmp/ccmp/b.ne with
// something NZCV-preserving without requiring an extra instruction.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
}
else {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
ARMEmitter::SingleUseForwardLabel LoopExpected;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
Bind(&LoopTop);
// This instruction sequence must be synced with HandleCASPAL_Armv8.
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
cmp(EmitSize, TMP2, Expected.first);
ccmp(EmitSize, TMP3, Expected.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
@@ -54,16 +47,13 @@ DEF_OP(CASPair) {
b(&LoopExpected);
Bind(&LoopNotExpected);
mov(EmitSize, Dst.first, TMP2.R());
mov(EmitSize, Dst.second, TMP3.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopNotExpected);
mov(EmitSize, Dst.first, TMP2.R());
mov(EmitSize, Dst.second, TMP3.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopExpected);
// Restore
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
}
}
@@ -81,26 +71,28 @@ DEF_OP(CAS) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
mov(EmitSize, TMP2, Expected);
casal(SubEmitSize, TMP2, Desired, MemSrc);
mov(EmitSize, GetReg(Node), TMP2.R());
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
ARMEmitter::SingleUseForwardLabel LoopExpected;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
if (OpSize == 1) {
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
} else if (OpSize == 2) {
}
else if (OpSize == 2) {
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTH, 0);
} else {
}
else {
cmp(EmitSize, TMP2, Expected);
}
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
@@ -109,11 +101,11 @@ DEF_OP(CAS) {
mov(EmitSize, GetReg(Node), Expected);
b(&LoopExpected);
Bind(&LoopNotExpected);
mov(EmitSize, GetReg(Node), TMP2.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopNotExpected);
mov(EmitSize, GetReg(Node), TMP2.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopExpected);
}
}
@@ -128,14 +120,14 @@ DEF_OP(AtomicAdd) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
staddl(SubEmitSize, Src, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -155,15 +147,15 @@ DEF_OP(AtomicSub) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
neg(EmitSize, TMP2, Src);
staddl(SubEmitSize, TMP2, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -183,15 +175,15 @@ DEF_OP(AtomicAnd) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
mvn(EmitSize, TMP2, Src);
stclrl(SubEmitSize, TMP2, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -211,14 +203,14 @@ DEF_OP(AtomicCLR) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
stclrl(SubEmitSize, Src, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -238,14 +230,14 @@ DEF_OP(AtomicOr) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
stsetl(SubEmitSize, Src, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -265,14 +257,14 @@ DEF_OP(AtomicXor) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
steorl(SubEmitSize, Src, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -291,10 +283,9 @@ DEF_OP(AtomicNeg) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
@@ -314,14 +305,15 @@ DEF_OP(AtomicSwap) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
mov(EmitSize, TMP2, Src);
ldswpal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -341,14 +333,14 @@ DEF_OP(AtomicFetchAdd) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -369,15 +361,15 @@ DEF_OP(AtomicFetchSub) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
neg(EmitSize, TMP2, Src);
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -398,15 +390,15 @@ DEF_OP(AtomicFetchAnd) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
mvn(EmitSize, TMP2, Src);
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -427,14 +419,14 @@ DEF_OP(AtomicFetchCLR) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -455,14 +447,14 @@ DEF_OP(AtomicFetchOr) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -483,14 +475,14 @@ DEF_OP(AtomicFetchXor) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -510,10 +502,9 @@ DEF_OP(AtomicFetchNeg) {
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
@@ -537,7 +528,8 @@ DEF_OP(TelemetrySetValue) {
if (CTX->HostFeatures.SupportsAtomics) {
stsetl(ARMEmitter::SubRegSize::i64Bit, TMP1, TMP2);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
@@ -549,4 +541,5 @@ DEF_OP(TelemetrySetValue) {
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -11,15 +11,15 @@ $end_info$
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "Interface/Core/InternalThreadState.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/MathUtils.h>
#include <Interface/HLE/Thunks/Thunks.h>
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(CallbackReturn) {
// spill back to CTX
@@ -53,39 +53,33 @@ DEF_OP(ExitFunction) {
uint64_t NewRIP;
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
#ifdef _M_ARM_64EC
if (RtlIsEcCode(NewRIP)) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, NewRIP);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
} else {
#endif
ARMEmitter::SingleUseForwardLabel l_BranchHost;
ldr(TMP1, &l_BranchHost);
blr(TMP1);
ARMEmitter::ForwardLabel l_BranchHost;
ARMEmitter::ForwardLabel l_BranchGuest;
ldr(ARMEmitter::XReg::x0, &l_BranchHost);
blr(ARMEmitter::Reg::r0);
Bind(&l_BranchHost);
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
Bind(&l_BranchGuest);
dc64(NewRIP);
Bind(&l_BranchHost);
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
dc64(NewRIP);
#ifdef _M_ARM_64EC
}
#endif
} else {
ARMEmitter::SingleUseForwardLabel FullLookup;
ARMEmitter::ForwardLabel FullLookup;
auto RipReg = GetReg(Op->NewRIP.ID());
// L1 Cache
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg, LookupCache::L1_ENTRIES_MASK);
add(TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x3, ARMEmitter::ShiftType::LSL, 4);
// Note: sub+cbnz used over cmp+br to preserve flags.
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
sub(TMP1, TMP1, RipReg.X());
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x1, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
sub(TMP1, ARMEmitter::XReg::x0, RipReg.X());
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
br(TMP2);
br(ARMEmitter::Reg::r1);
Bind(&FullLookup);
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
@@ -103,7 +97,9 @@ DEF_OP(Jump) {
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
case FEXCore::IR::COND_ANDZ:
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
case FEXCore::IR::COND_ANDNZ:
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
@@ -115,15 +111,17 @@ static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
case FEXCore::IR::COND_MI:
case FEXCore::IR::COND_PL:
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
return ARMEmitter::Condition::CC_NV;
}
}
@@ -132,26 +130,42 @@ DEF_OP(CondJump) {
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
if (Op->FromNZCV) {
b(MapBranchCC(Op->Cond), TrueTargetLabel);
} else {
[[maybe_unused]] uint64_t Const;
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
Op->Cond == FEXCore::IR::COND_ANDNZ;
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
const auto SubSize = ARMEmitter::ToVectorSizePair(Op->CompareSize == 4 ? ARMEmitter::SubRegSize::i32Bit : ARMEmitter::SubRegSize::i64Bit);
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ || Op->Cond.Val == FEXCore::IR::COND_NEQ, "CondJump: Expected simple "
"condition");
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
} else if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_NEQ) {
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
} else {
if (IsGPR(Op->Cmp1.ID())) {
if (tests) {
if (isConst) {
tst(Size, GetReg(Op->Cmp1.ID()), Const);
} else {
tst(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
}
} else {
if (isConst) {
cmp(Size, GetReg(Op->Cmp1.ID()), Const);
} else {
cmp(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
}
}
} else if (IsFPR(Op->Cmp1.ID())) {
fcmp(SubSize.Scalar, GetVReg(Op->Cmp1.ID()), GetVReg(Op->Cmp2.ID()));
} else {
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
LOGMAN_MSG_A_FMT("CondJump: Expected GPR or FPR");
}
// TODO: Wire up tbz/tbnz
b(MapBranchCC(Op->Cond), TrueTargetLabel);
}
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
@@ -187,9 +201,7 @@ DEF_OP(Syscall) {
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
continue;
}
if (Op->Header.Args[i].IsInvalid()) continue;
str(GetReg(Op->Header.Args[i].ID()).X(), ARMEmitter::Reg::rsp, i * 8);
}
@@ -201,7 +213,8 @@ DEF_OP(Syscall) {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
} else {
}
else {
blr(ARMEmitter::Reg::r3);
}
@@ -239,19 +252,20 @@ DEF_OP(InlineSyscall) {
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS - 1> RegArgs = {
{ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5}};
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS-1> RegArgs = {{
ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5
}};
bool Intersects {};
bool Intersects{};
// We always need to spill x8 since we can't know if it is live at this SSA location
uint32_t SpillMask = 1U << 8;
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
break;
}
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
auto Reg = GetReg(Op->Header.Args[i].ID());
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
if (Reg == ARMEmitter::Reg::r8 ||
Reg == ARMEmitter::Reg::r4 ||
Reg == ARMEmitter::Reg::r5) {
SpillMask |= (1U << Reg.Idx());
Intersects = true;
@@ -275,10 +289,8 @@ DEF_OP(InlineSyscall) {
const auto EmitSize = CTX->Config.Is64BitMode() ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto EmitSubSize = CTX->Config.Is64BitMode() ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
if (Intersects) {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
break;
}
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
auto Reg = GetReg(Op->Header.Args[i].ID());
// In the case of intersection with x4, x5, or x8 then these are currently SRA
@@ -286,19 +298,21 @@ DEF_OP(InlineSyscall) {
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
if (Reg == ARMEmitter::Reg::r8) {
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP]));
} else if (Reg == ARMEmitter::Reg::r4) {
}
else if (Reg == ARMEmitter::Reg::r4) {
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX]));
} else if (Reg == ARMEmitter::Reg::r5) {
}
else if (Reg == ARMEmitter::Reg::r5) {
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RCX]));
} else {
}
else {
mov(EmitSize, RegArgs[i].R(), Reg);
}
}
} else {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
break;
}
}
else {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i].ID()));
}
@@ -339,7 +353,8 @@ DEF_OP(Thunk) {
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
}
else {
blr(ARMEmitter::Reg::r2);
}
@@ -350,65 +365,70 @@ DEF_OP(Thunk) {
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
const auto* OldCode = (const uint8_t*)&Op->CodeOriginalLow;
const auto *OldCode = (const uint8_t *)&Op->CodeOriginalLow;
int len = Op->CodeLength;
int idx = 0;
LoadConstant(ARMEmitter::Size::i64Bit, GetReg(Node), 0);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Entry + Op->Offset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, 1);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Entry + Op->Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 1);
const auto Dst = GetReg(Node);
while (len >= 8) {
ldr(ARMEmitter::XReg::x2, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i64Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
while (len >= 8)
{
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 8;
idx += 8;
}
while (len >= 4) {
ldr(ARMEmitter::WReg::w2, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
while (len >= 4)
{
ldr(ARMEmitter::WReg::w2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 4;
idx += 4;
}
while (len >= 2) {
ldrh(TMP3, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint16_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
while (len >= 2)
{
ldrh(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint16_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 2;
idx += 2;
}
while (len >= 1) {
ldrb(TMP3, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint8_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
while (len >= 1)
{
ldrb(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint8_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 1;
idx += 1;
}
}
DEF_OP(ThreadRemoveCodeEntry) {
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
// Arguments are passed as follows:
// X0: Thread
// X1: RIP
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
PushDynamicRegsAndLR(TMP1);
SpillStaticRegs(TMP1);
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
}
else {
blr(ARMEmitter::Reg::r2);
}
FillStaticRegs();
@@ -420,32 +440,21 @@ DEF_OP(ThreadRemoveCodeEntry) {
DEF_OP(CPUID) {
auto Op = IROp->C<IR::IROp_CPUID>();
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function.ID()));
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf.ID()));
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
PushDynamicRegsAndLR(TMP1);
SpillStaticRegs(TMP1);
// x0 = CPUID Handler
// x1 = CPUID Function
// x2 = CPUID Leaf
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction));
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, TMP2);
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, TMP3);
}
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, GetReg(Op->Leaf.ID()));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<__uint128_t, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
} else {
blr(ARMEmitter::Reg::r3);
}
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
mov(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Reg::r1);
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
@@ -455,30 +464,26 @@ DEF_OP(CPUID) {
// Results are in x0, x1
// Results want to be in a i64v2 vector
auto Dst = GetRegPair(Node);
mov(ARMEmitter::Size::i64Bit, Dst.first, TMP1);
mov(ARMEmitter::Size::i64Bit, Dst.second, TMP2);
mov(ARMEmitter::Size::i64Bit, Dst.first, ARMEmitter::Reg::r0);
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
}
DEF_OP(XGetBV) {
auto Op = IROp->C<IR::IROp_XGetBV>();
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
PushDynamicRegsAndLR(TMP1);
SpillStaticRegs(TMP1);
// x0 = CPUID Handler
// x1 = XCR Function
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
} else {
blr(ARMEmitter::Reg::r2);
}
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
else {
blr(ARMEmitter::Reg::r2);
}
FillStaticRegs();
@@ -488,9 +493,10 @@ DEF_OP(XGetBV) {
// Results are in x0
// Results want to be in a i32v2 vector
auto Dst = GetRegPair(Node);
mov(ARMEmitter::Size::i32Bit, Dst.first, TMP1);
lsr(ARMEmitter::Size::i64Bit, Dst.second, TMP1, 32);
mov(ARMEmitter::Size::i32Bit, Dst.first, ARMEmitter::Reg::r0);
lsr(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r0, 32);
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -9,7 +9,7 @@ $end_info$
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(VInsGPR) {
const auto Op = IROp->C<IR::IROp_VInsGPR>();
const auto OpSize = IROp->Size;
@@ -20,10 +20,9 @@ DEF_OP(VInsGPR) {
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
const auto ElementsPer128Bit = 16 / ElementSize;
const auto Dst = GetVReg(Node);
@@ -95,17 +94,21 @@ DEF_OP(VCastFromGPR) {
auto Src = GetReg(Op->Src.ID());
switch (Op->Header.ElementSize) {
case 1:
uxtb(ARMEmitter::Size::i32Bit, TMP1, Src);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
break;
case 2:
uxth(ARMEmitter::Size::i32Bit, TMP1, Src);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
break;
case 4: fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src); break;
case 8: fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src); break;
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
case 1:
uxtb(ARMEmitter::Size::i32Bit, TMP1, Src);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
break;
case 2:
uxth(ARMEmitter::Size::i32Bit, TMP1, Src);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
break;
case 4:
fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src);
break;
case 8:
fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src);
break;
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
}
}
@@ -119,14 +122,14 @@ DEF_OP(VDupFromGPR) {
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto ElementSize = IROp->ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1, "Unexpected {} element size: {}",
__func__, ElementSize);
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1,
"Unexpected {} element size: {}", __func__, ElementSize);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
const auto SubEmitSize =
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (HostSupportsSVE256 && Is256Bit) {
dup(SubEmitSize, Dst.Z(), Src);
@@ -145,33 +148,34 @@ DEF_OP(Float_FromGPR_S) {
auto Src = GetReg(Op->Src.ID());
switch (Conv) {
case 0x0204: { // Half <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.H(), Src);
break;
}
case 0x0208: { // Half <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.H(), Src);
break;
}
case 0x0404: { // Float <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.S(), Src);
break;
}
case 0x0408: { // Float <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.S(), Src);
break;
}
case 0x0804: { // Double <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.D(), Src);
break;
}
case 0x0808: { // Double <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.D(), Src);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}", Conv, ElementSize, Op->SrcElementSize);
break;
case 0x0204: { // Half <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.H(), Src);
break;
}
case 0x0208: { // Half <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.H(), Src);
break;
}
case 0x0404: { // Float <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.S(), Src);
break;
}
case 0x0408: { // Float <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.S(), Src);
break;
}
case 0x0804: { // Double <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.D(), Src);
break;
}
case 0x0808: { // Double <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.D(), Src);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
Conv, ElementSize, Op->SrcElementSize);
break;
}
}
@@ -183,31 +187,31 @@ DEF_OP(Float_FToF) {
auto Src = GetVReg(Op->Scalar.ID());
switch (Conv) {
case 0x0204: { // Half <- Float
fcvt(Dst.H(), Src.S());
break;
}
case 0x0208: { // Half <- Double
fcvt(Dst.H(), Src.D());
break;
}
case 0x0402: { // Float <- Half
fcvt(Dst.S(), Src.H());
break;
}
case 0x0802: { // Double <- Half
fcvt(Dst.D(), Src.H());
break;
}
case 0x0804: { // Double <- Float
fcvt(Dst.D(), Src.S());
break;
}
case 0x0408: { // Float <- Double
fcvt(Dst.S(), Src.D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
case 0x0204: { // Half <- Float
fcvt(Dst.H(), Src.S());
break;
}
case 0x0208: { // Half <- Double
fcvt(Dst.H(), Src.D());
break;
}
case 0x0402: { // Float <- Half
fcvt(Dst.S(), Src.H());
break;
}
case 0x0802: { // Double <- Half
fcvt(Dst.D(), Src.H());
break;
}
case 0x0804: { // Double <- Float
fcvt(Dst.D(), Src.S());
break;
}
case 0x0408: { // Float <- Double
fcvt(Dst.S(), Src.D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
}
}
@@ -220,9 +224,8 @@ DEF_OP(Vector_SToF) {
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ARMEmitter::SubRegSize::i16Bit;
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
@@ -233,15 +236,19 @@ DEF_OP(Vector_SToF) {
if (OpSize == ElementSize) {
if (ElementSize == 8) {
scvtf(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
} else if (ElementSize == 4) {
}
else if (ElementSize == 4) {
scvtf(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
} else {
}
else {
scvtf(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
}
} else {
}
else {
if (OpSize == 8) {
scvtf(SubEmitSize, Dst.D(), Vector.D());
} else {
}
else {
scvtf(SubEmitSize, Dst.Q(), Vector.Q());
}
}
@@ -257,9 +264,8 @@ DEF_OP(Vector_FToZS) {
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ARMEmitter::SubRegSize::i16Bit;
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
@@ -270,15 +276,19 @@ DEF_OP(Vector_FToZS) {
if (OpSize == ElementSize) {
if (ElementSize == 8) {
fcvtzs(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
} else if (ElementSize == 4) {
}
else if (ElementSize == 4) {
fcvtzs(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
} else {
}
else {
fcvtzs(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
}
} else {
}
else {
if (OpSize == 8) {
fcvtzs(SubEmitSize, Dst.D(), Vector.D());
} else {
}
else {
fcvtzs(SubEmitSize, Dst.Q(), Vector.Q());
}
}
@@ -294,9 +304,8 @@ DEF_OP(Vector_FToS) {
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ARMEmitter::SubRegSize::i16Bit;
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
@@ -311,7 +320,8 @@ DEF_OP(Vector_FToS) {
if (OpSize == 8) {
frinti(SubEmitSize, Dst.D(), Vector.D());
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
} else {
}
else {
frinti(SubEmitSize, Dst.Q(), Vector.Q());
fcvtzs(SubEmitSize, Dst.Q(), Dst.Q());
}
@@ -328,9 +338,8 @@ DEF_OP(Vector_FToF) {
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ARMEmitter::SubRegSize::i16Bit;
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
@@ -352,41 +361,45 @@ DEF_OP(Vector_FToF) {
const auto Mask = PRED_TMP_32B.Merging();
switch (Conv) {
case 0x0402: { // Float <- Half
zip1(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Vector.Z(), Vector.Z());
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Dst.Z());
break;
}
case 0x0804: { // Double <- Float
zip1(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Vector.Z(), Vector.Z());
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Dst.Z());
break;
}
case 0x0204: { // Half <- Float
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Mask, Vector.Z());
uzp2(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
break;
}
case 0x0408: { // Float <- Double
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Vector.Z());
uzp2(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
case 0x0402: { // Float <- Half
zip1(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Vector.Z(), Vector.Z());
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Dst.Z());
break;
}
case 0x0804: { // Double <- Float
zip1(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Vector.Z(), Vector.Z());
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Dst.Z());
break;
}
case 0x0204: { // Half <- Float
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Mask, Vector.Z());
uzp2(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
break;
}
case 0x0408: { // Float <- Double
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Vector.Z());
uzp2(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
break;
}
} else {
switch (Conv) {
case 0x0402: // Float <- Half
case 0x0804: { // Double <- Float
fcvtl(SubEmitSize, Dst.D(), Vector.D());
break;
}
case 0x0204: // Half <- Float
case 0x0408: { // Float <- Double
fcvtn(SubEmitSize, Dst.D(), Vector.D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
case 0x0402: // Float <- Half
case 0x0804: { // Double <- Float
fcvtl(SubEmitSize, Dst.D(), Vector.D());
break;
}
case 0x0204: // Half <- Float
case 0x0408: { // Float <- Double
fcvtn(SubEmitSize, Dst.D(), Vector.D());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
break;
}
}
}
@@ -400,9 +413,8 @@ DEF_OP(Vector_FToI) {
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ARMEmitter::SubRegSize::i16Bit;
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
@@ -411,51 +423,82 @@ DEF_OP(Vector_FToI) {
const auto Mask = PRED_TMP_32B.Merging();
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Nearest.Val:
frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
case FEXCore::IR::Round_Towards_Zero.Val:
frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
case FEXCore::IR::Round_Host.Val:
frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
}
} else {
const auto IsScalar = ElementSize == OpSize;
if (IsScalar) {
// Since we have multiple overloads of the same name (e.g.
// frinti having AdvSIMD, AdvSIMD scalar, and an SVE version),
// we can't just use a lambda without some seriously ugly casting.
// This is fairly self-contained otherwise.
#define ROUNDING_FN(name) \
if (ElementSize == 2) { \
name(Dst.H(), Vector.H()); \
} else if (ElementSize == 4) { \
name(Dst.S(), Vector.S()); \
} else if (ElementSize == 8) { \
name(Dst.D(), Vector.D()); \
} else { \
FEX_UNREACHABLE; \
}
// Since we have multiple overloads of the same name (e.g.
// frinti having AdvSIMD, AdvSIMD scalar, and an SVE version),
// we can't just use a lambda without some seriously ugly casting.
// This is fairly self-contained otherwise.
#define ROUNDING_FN(name) \
if (ElementSize == 2) { \
name(Dst.H(), Vector.H()); \
} else if (ElementSize == 4) { \
name(Dst.S(), Vector.S()); \
} else if (ElementSize == 8) { \
name(Dst.D(), Vector.D()); \
} else { \
FEX_UNREACHABLE; \
}
switch (Op->Round) {
case IR::Round_Nearest.Val: ROUNDING_FN(frintn); break;
case IR::Round_Negative_Infinity.Val: ROUNDING_FN(frintm); break;
case IR::Round_Positive_Infinity.Val: ROUNDING_FN(frintp); break;
case IR::Round_Towards_Zero.Val: ROUNDING_FN(frintz); break;
case IR::Round_Host.Val: ROUNDING_FN(frinti); break;
case IR::Round_Nearest.Val:
ROUNDING_FN(frintn);
break;
case IR::Round_Negative_Infinity.Val:
ROUNDING_FN(frintm);
break;
case IR::Round_Positive_Infinity.Val:
ROUNDING_FN(frintp);
break;
case IR::Round_Towards_Zero.Val:
ROUNDING_FN(frintz);
break;
case IR::Round_Host.Val:
ROUNDING_FN(frinti);
break;
}
#undef ROUNDING_FN
#undef ROUNDING_FN
} else {
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Nearest.Val:
frintn(SubEmitSize, Dst.Q(), Vector.Q());
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
frintm(SubEmitSize, Dst.Q(), Vector.Q());
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
frintp(SubEmitSize, Dst.Q(), Vector.Q());
break;
case FEXCore::IR::Round_Towards_Zero.Val:
frintz(SubEmitSize, Dst.Q(), Vector.Q());
break;
case FEXCore::IR::Round_Host.Val:
frinti(SubEmitSize, Dst.Q(), Vector.Q());
break;
}
}
}
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -10,7 +10,7 @@ $end_info$
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(VAESImc) {
auto Op = IROp->C<IR::IROp_VAESImc>();
@@ -26,7 +26,8 @@ DEF_OP(VAESEnc) {
const auto State = GetVReg(Op->State.ID());
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
if (Dst == State && Dst != Key) {
// Optimal case in which Dst already contains the starting state.
@@ -34,7 +35,8 @@ DEF_OP(VAESEnc) {
aese(Dst.Q(), ZeroReg.Q());
aesmc(Dst.Q(), Dst.Q());
eor(Dst.Q(), Dst.Q(), Key.Q());
} else {
}
else {
mov(VTMP1.Q(), State.Q());
aese(VTMP1, ZeroReg.Q());
aesmc(VTMP1, VTMP1);
@@ -51,14 +53,16 @@ DEF_OP(VAESEncLast) {
const auto State = GetVReg(Op->State.ID());
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
if (Dst == State && Dst != Key) {
// Optimal case in which Dst already contains the starting state.
// This matches the common case of XMM AES.
aese(Dst.Q(), ZeroReg.Q());
eor(Dst.Q(), Dst.Q(), Key.Q());
} else {
}
else {
mov(VTMP1.Q(), State.Q());
aese(VTMP1, ZeroReg.Q());
eor(Dst.Q(), VTMP1.Q(), Key.Q());
@@ -74,7 +78,8 @@ DEF_OP(VAESDec) {
const auto State = GetVReg(Op->State.ID());
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
if (Dst == State && Dst != Key) {
// Optimal case in which Dst already contains the starting state.
@@ -82,7 +87,8 @@ DEF_OP(VAESDec) {
aesd(Dst.Q(), ZeroReg.Q());
aesimc(Dst.Q(), Dst.Q());
eor(Dst.Q(), Dst.Q(), Key.Q());
} else {
}
else {
mov(VTMP1.Q(), State.Q());
aesd(VTMP1, ZeroReg.Q());
aesimc(VTMP1, VTMP1);
@@ -99,14 +105,16 @@ DEF_OP(VAESDecLast) {
const auto State = GetVReg(Op->State.ID());
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
if (Dst == State && Dst != Key) {
// Optimal case in which Dst already contains the starting state.
// This matches the common case of XMM AES.
aesd(Dst.Q(), ZeroReg.Q());
eor(Dst.Q(), Dst.Q(), Key.Q());
} else {
}
else {
mov(VTMP1.Q(), State.Q());
aesd(VTMP1, ZeroReg.Q());
eor(Dst.Q(), VTMP1.Q(), Key.Q());
@@ -141,7 +149,8 @@ DEF_OP(VAESKeyGenAssist) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->RCON) << 32);
dup(ARMEmitter::SubRegSize::i64Bit, VTMP2.Q(), TMP1);
eor(Dst.Q(), Dst.Q(), VTMP2.Q());
} else {
}
else {
tbl(Dst.Q(), Dst.Q(), Swizzle.Q());
}
}
@@ -154,36 +163,19 @@ DEF_OP(CRC32) {
const auto Src2 = GetReg(Op->Src2.ID());
switch (Op->SrcSize) {
case 1: crc32cb(Dst.W(), Src1.W(), Src2.W()); break;
case 2: crc32ch(Dst.W(), Src1.W(), Src2.W()); break;
case 4: crc32cw(Dst.W(), Src1.W(), Src2.W()); break;
case 8: crc32cx(Dst.X(), Src1.X(), Src2.X()); break;
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
}
}
DEF_OP(VSha1H) {
auto Op = IROp->C<IR::IROp_VSha1H>();
const auto Dst = GetVReg(Node);
const auto Src = GetVReg(Op->Src.ID());
sha1h(Dst.S(), Src.S());
}
DEF_OP(VSha256U0) {
auto Op = IROp->C<IR::IROp_VSha256U0>();
const auto Dst = GetVReg(Node);
const auto Src1 = GetVReg(Op->Src1.ID());
const auto Src2 = GetVReg(Op->Src2.ID());
if (Dst == Src1) {
sha256su0(Dst, Src2);
} else {
mov(VTMP1.Q(), Src1.Q());
sha256su0(VTMP1, Src2);
mov(Dst.Q(), VTMP1.Q());
case 1:
crc32cb(Dst.W(), Src1.W(), Src2.W());
break;
case 2:
crc32ch(Dst.W(), Src1.W(), Src2.W());
break;
case 4:
crc32cw(Dst.W(), Src1.W(), Src2.W());
break;
case 8:
crc32cx(Dst.X(), Src1.X(), Src2.X());
break;
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
}
}
@@ -191,14 +183,17 @@ DEF_OP(PCLMUL) {
const auto Op = IROp->C<IR::IROp_PCLMUL>();
const auto OpSize = IROp->Size;
const auto Dst = GetVReg(Node);
const auto Dst = GetVReg(Node);
const auto Src1 = GetVReg(Op->Src1.ID());
const auto Src2 = GetVReg(Op->Src2.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
switch (Op->Selector) {
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
case 0b00000000:
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D());
break;
case 0b00000001:
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
@@ -207,10 +202,14 @@ DEF_OP(PCLMUL) {
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
break;
case 0b00010001: pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q()); break;
default: LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector); break;
case 0b00010001:
pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q());
break;
default:
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
break;
}
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -8,11 +8,12 @@ $end_info$
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(GetHostFlag) {
auto Op = IROp->C<IR::IROp_GetHostFlag>();
ubfx(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Value.ID()), Op->Flag, 1);
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
File diff suppressed because it is too large. Load diff
+101 -127
View File
@@ -9,17 +9,16 @@ $end_info$
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <aarch64/assembler-aarch64.h>
#include <aarch64/disasm-aarch64.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
@@ -30,60 +29,48 @@ $end_info$
#include <variant>
namespace FEXCore::Core {
struct InternalThreadState;
struct InternalThreadState;
}
namespace FEXCore::CPU {
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
public:
explicit Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
explicit Arm64JITCore(FEXCore::Context::ContextImpl *ctx,
FEXCore::Core::InternalThreadState *Thread);
~Arm64JITCore() override;
[[nodiscard]]
fextl::string GetName() override {
return "JIT";
}
[[nodiscard]] fextl::string GetName() override { return "JIT"; }
[[nodiscard]]
CPUBackend::CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
FEXCore::IR::RegisterAllocationData* RAData) override;
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]]
void* MapRegion(void* HostPtr, uint64_t, uint64_t) override {
return HostPtr;
}
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]]
bool NeedsOpDispatch() override {
return true;
}
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
void ClearCache() override;
void ClearRelocations() override {
Relocations.clear();
}
void ClearRelocations() override { Relocations.clear(); }
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
const bool HostSupportsSVE128 {};
const bool HostSupportsSVE256 {};
const bool HostSupportsRPRES {};
const bool HostSupportsAFP {};
const bool HostSupportsSVE128{};
const bool HostSupportsSVE256{};
const bool HostSupportsRPRES{};
const bool HostSupportsAFP{};
ARMEmitter::BiDirectionalLabel* PendingTargetLabel;
FEXCore::Context::ContextImpl* CTX;
const FEXCore::IR::IRListView* IR;
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
FEXCore::Context::ContextImpl *CTX;
FEXCore::IR::IRListView const *IR;
uint64_t Entry;
CPUBackend::CompiledCode CodeData {};
CPUBackend::CompiledCode CodeData{};
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
[[nodiscard]]
FEXCore::ARMEmitter::Register GetReg(IR::NodeID Node) const {
[[nodiscard]] FEXCore::ARMEmitter::Register GetReg(IR::NodeID Node) const {
const auto Reg = GetPhys(Node);
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
@@ -97,8 +84,7 @@ private:
FEX_UNREACHABLE;
}
[[nodiscard]]
FEXCore::ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
[[nodiscard]] FEXCore::ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
const auto Reg = GetPhys(Node);
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
@@ -112,8 +98,7 @@ private:
FEX_UNREACHABLE;
}
[[nodiscard]]
std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
[[nodiscard]] std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
const auto Reg = GetPhys(Node);
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
@@ -121,11 +106,9 @@ private:
return GeneralPairRegisters[Reg.Reg];
}
[[nodiscard]]
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
[[nodiscard]]
IR::PhysicalRegister GetPhys(IR::NodeID Node) const {
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const {
auto PhyReg = RAData->GetNodeRegister(Node);
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
@@ -133,51 +116,38 @@ private:
return PhyReg;
}
[[nodiscard]]
FEXCore::ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
uint64_t Const;
if (IsInlineConstant(Src, &Const)) {
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
return ARMEmitter::Reg::zr;
} else {
return GetReg(Src.ID());
}
}
// Converts IR-base shift type to ARMEmitter shift type.
// Will be a no-op, only a type conversion since the two definitions match.
[[nodiscard]]
ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
[[nodiscard]] ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
ARMEmitter::ShiftType::ROR;
ARMEmitter::ShiftType::ROR;
}
[[nodiscard]]
bool IsFPR(IR::NodeID Node) const;
[[nodiscard]]
bool IsGPR(IR::NodeID Node) const;
[[nodiscard]]
bool IsGPRPair(IR::NodeID Node) const;
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
[[nodiscard]]
FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(
uint8_t AccessSize, FEXCore::ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
[[nodiscard]] FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize,
FEXCore::ARMEmitter::Register Base,
IR::OrderedNodeWrapper Offset,
IR::MemOffsetType OffsetType,
uint8_t OffsetScale);
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
// the limits of the scalar plus immediate variant of SVE load/stores.
//
// TMP1 is safe to use again once this memory operand is used with its
// equivalent loads or stores that this was called for.
[[nodiscard]]
FEXCore::ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize, FEXCore::ARMEmitter::Register Base,
IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
[[nodiscard]] FEXCore::ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize,
FEXCore::ARMEmitter::Register Base,
IR::OrderedNodeWrapper Offset,
IR::MemOffsetType OffsetType,
uint8_t OffsetScale);
[[nodiscard]]
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
[[nodiscard]]
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
struct LiveRange {
uint32_t Begin;
@@ -186,85 +156,89 @@ private:
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
IR::RegisterAllocationPass* RAPass;
IR::RegisterAllocationData* RAData;
FEXCore::Core::DebugData* DebugData;
IR::RegisterAllocationPass *RAPass;
IR::RegisterAllocationData *RAData;
FEXCore::Core::DebugData *DebugData;
void ResetStack();
/**
* @name Relocations
* @{ */
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief A literal pair relocation object for named symbol literals
*/
struct NamedSymbolLiteralPair {
ARMEmitter::ForwardLabel Loc;
uint64_t Lit;
Relocation MoveABI {};
};
/**
* @brief A literal pair relocation object for named symbol literals
*/
struct NamedSymbolLiteralPair {
ARMEmitter::ForwardLabel Loc;
uint64_t Lit;
Relocation MoveABI{};
};
/**
* @brief Inserts a thunk relocation
*
* @param Reg - The GPR to move the thunk handler in to
* @param Sum - The hash of the thunk
*/
void InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum);
/**
* @brief Inserts a thunk relocation
*
* @param Reg - The GPR to move the thunk handler in to
* @param Sum - The hash of the thunk
*/
void InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum &Sum);
/**
* @brief Inserts a guest GPR move relocation
*
* @param Reg - The GPR to move the guest RIP in to
* @param Constant - The guest RIP that will be relocated
*/
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
/**
* @brief Inserts a guest GPR move relocation
*
* @param Reg - The GPR to move the guest RIP in to
* @param Constant - The guest RIP that will be relocated
*/
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
/**
* @brief Inserts a named symbol as a literal in memory
*
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
*
* @param Op The named symbol to place
*
* @return A temporary `NamedSymbolLiteralPair`
*/
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief Inserts a named symbol as a literal in memory
*
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
*
* @param Op The named symbol to place
*
* @return A temporary `NamedSymbolLiteralPair`
*/
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief Place the named symbol literal relocation in memory
*
* @param Lit - Which literal to place
*/
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit);
/**
* @brief Place the named symbol literal relocation in memory
*
* @param Lit - Which literal to place
*/
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
fextl::vector<FEXCore::CPU::Relocation> Relocations;
fextl::vector<FEXCore::CPU::Relocation> Relocations;
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
/** @} */
uint32_t SpillSlots {};
using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::NodeID Node);
uint32_t SpillSlots{};
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
// Runtime selection;
// Load and store register style.
OpType RT_LoadRegister;
OpType RT_StoreRegister;
// Load and store TSO memory style
OpType RT_LoadMemTSO;
OpType RT_StoreMemTSO;
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
// Dynamic Dispatcher supporting operations
DEF_OP(LoadRegisterSRA);
DEF_OP(StoreRegisterSRA);
DEF_OP(ParanoidLoadMemTSO);
DEF_OP(ParanoidStoreMemTSO);
File diff suppressed because it is too large. Load diff
@@ -17,7 +17,7 @@ $end_info$
#include <FEXCore/Core/SignalDelegator.h>
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(GuestOpcode) {
auto Op = IROp->C<IR::IROp_GuestOpcode>();
@@ -28,10 +28,16 @@ DEF_OP(GuestOpcode) {
DEF_OP(Fence) {
auto Op = IROp->C<IR::IROp_Fence>();
switch (Op->Fence) {
case IR::Fence_Load.Val: dmb(FEXCore::ARMEmitter::BarrierScope::LD); break;
case IR::Fence_LoadStore.Val: dmb(FEXCore::ARMEmitter::BarrierScope::SY); break;
case IR::Fence_Store.Val: dmb(FEXCore::ARMEmitter::BarrierScope::ST); break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
case IR::Fence_Load.Val:
dmb(FEXCore::ARMEmitter::BarrierScope::LD);
break;
case IR::Fence_LoadStore.Val:
dmb(FEXCore::ARMEmitter::BarrierScope::SY);
break;
case IR::Fence_Store.Val:
dmb(FEXCore::ARMEmitter::BarrierScope::ST);
break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
}
}
@@ -49,7 +55,7 @@ DEF_OP(Break) {
.err_code = Op->Reason.ErrorRegister,
};
uint64_t Constant {};
uint64_t Constant{};
memcpy(&Constant, &State, sizeof(State));
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant);
@@ -130,7 +136,8 @@ DEF_OP(Print) {
if (IsGPR(Op->Value.ID())) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue));
} else {
}
else {
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value.ID()), false);
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value.ID()), true);
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
@@ -139,10 +146,12 @@ DEF_OP(Print) {
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
if (IsGPR(Op->Value.ID())) {
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r3);
} else {
}
else {
GenerateIndirectRuntimeCall<void, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
} else {
}
else {
blr(ARMEmitter::Reg::r3);
}
@@ -222,7 +231,8 @@ DEF_OP(RDRAND) {
if (Op->GetReseeded) {
mrs(Dst.first, ARMEmitter::SystemRegister::RNDRRS);
} else {
}
else {
mrs(Dst.first, ARMEmitter::SystemRegister::RNDR);
}
@@ -235,4 +245,5 @@ DEF_OP(Yield) {
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -8,7 +8,7 @@ $end_info$
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(ExtractElementPair) {
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
LOGMAN_THROW_AA_FMT(Op->Header.Size == 4 || Op->Header.Size == 8, "Invalid size");
@@ -43,4 +43,5 @@ DEF_OP(CreateElementPair) {
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
File diff suppressed because it is too large. Load diff
+7 -3
View File
@@ -1,7 +1,7 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Interface/Core/CPUBackend.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/fextl/memory.h>
namespace FEXCore::Context {
@@ -15,8 +15,12 @@ struct InternalThreadState;
namespace FEXCore::CPU {
class CPUBackend;
[[nodiscard]]
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx,
FEXCore::Core::InternalThreadState *Thread);
CPUBackendFeatures GetX86JITBackendFeatures();
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx,
FEXCore::Core::InternalThreadState *Thread);
CPUBackendFeatures GetArm64JITBackendFeatures();
} // namespace FEXCore::CPU
@@ -13,8 +13,8 @@ $end_info$
#include "Interface/Core/LookupCache.h"
namespace FEXCore {
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
: BlockLinks_mbr {fextl::pmr::get_default_resource()}
LookupCache::LookupCache(FEXCore::Context::ContextImpl *CTX)
: BlockLinks_mbr { fextl::pmr::get_default_resource() }
, ctx {CTX} {
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
@@ -78,4 +78,5 @@ void LookupCache::ClearCache() {
BlockList.clear();
}
} // namespace FEXCore
}
+35 -61
View File
@@ -13,9 +13,6 @@
#include <stddef.h>
#include <utility>
#include <mutex>
#ifdef _M_ARM_64EC
#include <winnt.h>
#endif
namespace FEXCore {
@@ -26,12 +23,12 @@ public:
uintptr_t GuestCode;
};
LookupCache(FEXCore::Context::ContextImpl* CTX);
LookupCache(FEXCore::Context::ContextImpl *CTX);
~LookupCache();
uintptr_t FindBlock(uint64_t Address) {
// Try L1, no lock needed
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
if (L1Entry.GuestCode == Address) {
return L1Entry.HostCode;
}
@@ -40,7 +37,7 @@ public:
std::lock_guard<std::recursive_mutex> lk(WriteLock);
// Try L2
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
const auto PageIndex = (Address & (VirtualMemSize -1)) >> 12;
const auto PageOffset = Address & (0x0FFF);
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
@@ -51,7 +48,8 @@ public:
// Find there pointer for the address in the blocks
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
if (BlockPointers[PageOffset].GuestCode == Address) {
if (BlockPointers[PageOffset].GuestCode == Address)
{
L1Entry.GuestCode = Address;
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
return L1Entry.HostCode;
@@ -70,24 +68,6 @@ public:
return 0;
}
#ifdef _M_ARM_64EC
bool CheckPageEC(uint64_t Address) {
if (!RtlIsEcCode(Address)) {
return false;
}
std::lock_guard<std::recursive_mutex> lk(WriteLock);
// Mark L2 entry for this page as EC by setting the LSB, this can then be
// checked by the dispatcher to see if it needs to perform a call/return to
// EC code.
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
Pointers[PageIndex] |= 1;
return true;
}
#endif
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
// Appends Block {Address} to CodePages [Start, Start + Length)
@@ -97,8 +77,8 @@ public:
bool rv = false;
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
auto& CodePage = CodePages[CurrentPage];
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length -1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
auto &CodePage = CodePages[CurrentPage];
rv |= CodePage.size() == 0;
CodePage.push_back(Address);
}
@@ -107,7 +87,7 @@ public:
}
// Adds to Guest -> Host code mapping
void AddBlockMapping(uint64_t Address, void* HostCode) {
void AddBlockMapping(uint64_t Address, void *HostCode) {
std::lock_guard<std::recursive_mutex> lk(WriteLock);
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
@@ -115,27 +95,27 @@ public:
// There is no need to update L1 or L2, they will get updated on first lookup
// However, adding to L1 here increases performance
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
L1Entry.GuestCode = Address;
L1Entry.HostCode = (uintptr_t)HostCode;
}
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
void Erase(uint64_t Address) {
std::lock_guard<std::recursive_mutex> lk(WriteLock);
// Sever any links to this block
auto lower = BlockLinks->lower_bound({Address, nullptr});
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
auto lower = BlockLinks->lower_bound({Address, 0});
auto upper = BlockLinks->upper_bound({Address, UINTPTR_MAX});
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
it->second(Frame, it->first.HostLink);
it->second();
}
// Remove from BlockList
BlockList.erase(Address);
// Do L1
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
if (L1Entry.GuestCode == Address) {
L1Entry.GuestCode = 0;
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
@@ -144,11 +124,11 @@ public:
}
// Do full map
Address = Address & (VirtualMemSize - 1);
Address = Address & (VirtualMemSize -1);
uint64_t PageOffset = Address & (0x0FFF);
Address >>= 12;
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uintptr_t *Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uint64_t LocalPagePointer = Pointers[Address];
if (!LocalPagePointer) {
// Page for this code didn't even exist, nothing to do
@@ -161,7 +141,8 @@ public:
BlockPointers[PageOffset].HostCode = 0;
}
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
std::lock_guard<std::recursive_mutex> lk(WriteLock);
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
@@ -170,20 +151,14 @@ public:
void ClearCache();
void ClearL2Cache();
uintptr_t GetL1Pointer() const {
return L1Pointer;
}
uintptr_t GetPagePointer() const {
return PagePointer;
}
uintptr_t GetVirtualMemorySize() const {
return VirtualMemSize;
}
uintptr_t GetL1Pointer() const { return L1Pointer; }
uintptr_t GetPagePointer() const { return PagePointer; }
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
// This needs to be taken before reads or writes to L2, L3, CodePages,
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
// may only happen during cross thread invalidation (::Erase).
// All other operations must be done from the owning thread.
@@ -195,17 +170,17 @@ public:
private:
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
// Do L1
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
L1Entry.GuestCode = Address;
L1Entry.HostCode = HostCode;
// Do ful map
auto FullAddress = Address;
Address = Address & (VirtualMemSize - 1);
Address = Address & (VirtualMemSize -1);
uint64_t PageOffset = Address & (0x0FFF);
Address >>= 12;
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uintptr_t *Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uint64_t LocalPagePointer = Pointers[Address];
if (!LocalPagePointer) {
// We don't have a page pointer for this address
@@ -249,16 +224,15 @@ private:
struct BlockLinkTag {
uint64_t GuestDestination;
FEXCore::Context::ExitFunctionLinkData* HostLink;
uintptr_t HostLink;
bool operator<(const BlockLinkTag& other) const {
if (GuestDestination < other.GuestDestination) {
bool operator <(const BlockLinkTag& other) const {
if (GuestDestination < other.GuestDestination)
return true;
} else if (GuestDestination == other.GuestDestination) {
else if (GuestDestination == other.GuestDestination)
return HostLink < other.HostLink;
} else {
else
return false;
}
}
};
@@ -269,9 +243,9 @@ private:
//
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
using BlockLinksMapType = std::pmr::map<BlockLinkTag, std::function<void()>>;
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
BlockLinksMapType* BlockLinks;
BlockLinksMapType *BlockLinks;
fextl::robin_map<uint64_t, uint64_t> BlockList;
@@ -283,7 +257,7 @@ private:
size_t AllocateOffset {};
FEXCore::Context::ContextImpl* ctx;
uint64_t VirtualMemSize {};
FEXCore::Context::ContextImpl *ctx;
uint64_t VirtualMemSize{};
};
} // namespace FEXCore
}
@@ -6,81 +6,84 @@
#include <cstdint>
namespace FEXCore::CodeSerialize {
// If any of the config options mismatch on load then the cache won't be used
// Any of these will result in codegen changes
struct FEX_PACKED CodeObjectSerializationConfig {
// Cookie in the header of the file, isn't part of the config hash
uint64_t Cookie {};
// If any of the config options mismatch on load then the cache won't be used
// Any of these will result in codegen changes
struct
FEX_PACKED
CodeObjectSerializationConfig {
// Cookie in the header of the file, isn't part of the config hash
uint64_t Cookie{};
// Instructions per block configuration
int32_t MaxInstPerBlock {};
// Instructions per block configuration
int32_t MaxInstPerBlock{};
// Follows CPUID 4000_0001_EAX[3:0]
unsigned Arch : 4;
// Follows CPUID 4000_0001_EAX[3:0]
unsigned Arch : 4;
// Multiblock enabled
unsigned MultiBlock : 1;
// Multiblock enabled
unsigned MultiBlock : 1;
// Hardware TSO enabled
unsigned HardwareTSOEnabled : 1;
// Hardware TSO enabled
unsigned HardwareTSOEnabled : 1;
// TSO enabled
unsigned TSOEnabled : 1;
// TSO enabled
unsigned TSOEnabled : 1;
// ABI local flag unsafe optimization
unsigned ABILocalFlags : 1;
// ABI local flag unsafe optimization
unsigned ABILocalFlags : 1;
// Paranoid TSO mode enabled
unsigned ParanoidTSO : 1;
// Static register allocation enabled
unsigned SRA : 1;
// Guest code execution mode (We don't support live mode switch)
unsigned Is64BitMode : 1;
// Paranoid TSO mode enabled
unsigned ParanoidTSO : 1;
// SMC checks style
unsigned SMCChecks : 2;
// Guest code execution mode (We don't support live mode switch)
unsigned Is64BitMode : 1;
// x87 reduced precision
unsigned x87ReducedPrecision : 1;
// SMC checks style
unsigned SMCChecks : 2;
// Padding to remove uninitialized data warning from asan
// Shows remaining amount of bits available for config
unsigned _Pad : 19;
// x87 reduced precision
unsigned x87ReducedPrecision : 1;
bool operator==(const CodeObjectSerializationConfig& other) const {
return Cookie == other.Cookie && MaxInstPerBlock == other.MaxInstPerBlock && Arch == other.Arch && MultiBlock == other.MultiBlock &&
HardwareTSOEnabled == other.HardwareTSOEnabled && TSOEnabled == other.TSOEnabled && ABILocalFlags == other.ABILocalFlags &&
ParanoidTSO == other.ParanoidTSO && Is64BitMode == other.Is64BitMode && SMCChecks == other.SMCChecks &&
x87ReducedPrecision == other.x87ReducedPrecision;
}
static uint64_t GetHash(const CodeObjectSerializationConfig& other) {
// For < 64-bits of data just pack directly
// Skip the cookie
uint64_t Hash {};
Hash <<= 32;
Hash |= other.MaxInstPerBlock;
Hash <<= 1;
Hash |= other.Arch;
Hash <<= 1;
Hash |= other.MultiBlock;
Hash <<= 1;
Hash |= other.HardwareTSOEnabled;
Hash <<= 1;
Hash |= other.TSOEnabled;
Hash <<= 1;
Hash |= other.ABILocalFlags;
Hash <<= 1;
Hash |= other.ParanoidTSO;
Hash <<= 1;
Hash |= other.Is64BitMode;
Hash <<= 2;
Hash |= other.SMCChecks;
Hash <<= 1;
Hash |= other.x87ReducedPrecision;
return Hash;
}
};
// Padding to remove uninitialized data warning from asan
// Shows remaining amount of bits available for config
unsigned _Pad : 18;
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is "
"generated!");
} // namespace FEXCore::CodeSerialize
bool operator==(CodeObjectSerializationConfig const &other) const {
return Cookie == other.Cookie &&
MaxInstPerBlock == other.MaxInstPerBlock &&
Arch == other.Arch &&
MultiBlock == other.MultiBlock &&
HardwareTSOEnabled == other.HardwareTSOEnabled &&
TSOEnabled == other.TSOEnabled &&
ABILocalFlags == other.ABILocalFlags &&
SRA == other.SRA &&
ParanoidTSO == other.ParanoidTSO &&
Is64BitMode == other.Is64BitMode &&
SMCChecks == other.SMCChecks &&
x87ReducedPrecision == other.x87ReducedPrecision;
}
static uint64_t GetHash(CodeObjectSerializationConfig const &other) {
// For < 64-bits of data just pack directly
// Skip the cookie
uint64_t Hash{};
Hash <<= 32; Hash |= other.MaxInstPerBlock;
Hash <<= 1; Hash |= other.Arch;
Hash <<= 1; Hash |= other.MultiBlock;
Hash <<= 1; Hash |= other.HardwareTSOEnabled;
Hash <<= 1; Hash |= other.TSOEnabled;
Hash <<= 1; Hash |= other.ABILocalFlags;
Hash <<= 1; Hash |= other.SRA;
Hash <<= 1; Hash |= other.ParanoidTSO;
Hash <<= 1; Hash |= other.Is64BitMode;
Hash <<= 2; Hash |= other.SMCChecks;
Hash <<= 1; Hash |= other.x87ReducedPrecision;
return Hash;
}
};
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is generated!");
}
@@ -11,112 +11,120 @@
#include <xxhash.h>
namespace FEXCore::CodeSerialize {
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename) {
#ifndef _WIN32
// This function adds a named region *JOB* to our named region handler
// This needs to be as fast as possible to keep out of the way of the JIT
// This function adds a named region *JOB* to our named region handler
// This needs to be as fast as possible to keep out of the way of the JIT
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
if (!BaseFilename.empty()) {
// Create a new entry that once set up will be put in to our section object map
auto Entry = fextl::make_unique<CodeRegionEntry>(Base, Size, Offset, filename, NamedRegionHandler->DefaultCodeHeader(Base, Offset));
if (!BaseFilename.empty()) {
// Create a new entry that once set up will be put in to our section object map
auto Entry = fextl::make_unique<CodeRegionEntry>(
Base,
Size,
Offset,
filename,
NamedRegionHandler->DefaultCodeHeader(Base, Offset)
);
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
Entry->NamedJobRefCountMutex.lock();
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
Entry->NamedJobRefCountMutex.lock();
CodeRegionMapType::iterator EntryIterator;
CodeRegionMapType::iterator EntryIterator;
{
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
auto it = EntryMap.emplace(Base, std::move(Entry));
if (!it.second) {
// This happens when an application overwrites a previous region without unmapping what was there
// Lock this entry's Named job reference counter.
// Once this passes then we know that this section has been loaded.
it.first->second->NamedJobRefCountMutex.lock();
// Finalize anything the region needs to do first.
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
// munmap the file that was mapped
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
// Remove this entry from the unrelocated map as well
{
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
}
// Now overwrite the entry in the map
it = EntryMap.insert_or_assign(Base, std::move(Entry));
EntryIterator = it.first;
}
else {
// No overwrite, just insert
EntryIterator = it.first;
}
}
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
// do the loading for us.
//
// Create the async work queue job now so it can load
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
#endif
}
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
#ifndef _WIN32
// Removing a named region through the job system
// We need to find the entry that we are deleting first
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
{
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
auto it = EntryMap.find(Base);
if (it != EntryMap.end()) {
// Lock the job ref counter since we are erasing it
// Once this passes it will have been loaded
it->second->NamedJobRefCountMutex.lock();
auto it = EntryMap.emplace(Base, std::move(Entry));
if (!it.second) {
// This happens when an application overwrites a previous region without unmapping what was there
// Take the pointer from the map
EntryPointer = std::move(it->second);
// Lock this entry's Named job reference counter.
// Once this passes then we know that this section has been loaded.
it.first->second->NamedJobRefCountMutex.lock();
// We can now unmap the file data
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
// Finalize anything the region needs to do first.
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
// munmap the file that was mapped
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
// Remove this from the entry map
EntryMap.erase(it);
// Remove this entry from the unrelocated map as well
{
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
}
// Now overwrite the entry in the map
it = EntryMap.insert_or_assign(Base, std::move(Entry));
EntryIterator = it.first;
} else {
// No overwrite, just insert
EntryIterator = it.first;
}
}
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
// do the loading for us.
//
// Create the async work queue job now so it can load
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
#endif
}
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
#ifndef _WIN32
// Removing a named region through the job system
// We need to find the entry that we are deleting first
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
{
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
auto it = EntryMap.find(Base);
if (it != EntryMap.end()) {
// Lock the job ref counter since we are erasing it
// Once this passes it will have been loaded
it->second->NamedJobRefCountMutex.lock();
// Take the pointer from the map
EntryPointer = std::move(it->second);
// We can now unmap the file data
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
// Remove this from the entry map
EntryMap.erase(it);
// Remove this entry from the unrelocated map as well
{
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
else {
// Tried to remove something that wasn't in our code object tracking
return;
}
} else {
// Tried to remove something that wasn't in our code object tracking
return;
// Create the async work queue job now so it can finalize what it needs to do
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
// Create the async work queue job now so it can finalize what it needs to do
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
#endif
}
}
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
// XXX: Actually add serialization job
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
// XXX: Actually add serialization job
}
}
} // namespace FEXCore::CodeSerialize
@@ -7,67 +7,67 @@
#include <FEXCore/fextl/string.h>
namespace FEXCore::CodeSerialize {
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx) {
DefaultSerializationConfig.Cookie = CODE_COOKIE;
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl *ctx) {
DefaultSerializationConfig.Cookie = CODE_COOKIE;
// Initialize the Arch from CPUID
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
DefaultSerializationConfig.Arch = Arch;
// Initialize the Arch from CPUID
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
DefaultSerializationConfig.Arch = Arch;
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
}
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
DefaultSerializationConfig.SRA = ctx->Config.StaticRegisterAllocation;
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
}
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename,
const fextl::string& filename, bool Executable) {
// XXX: Add named region objects
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string &base_filename, const fextl::string &filename, bool Executable) {
// XXX: Add named region objects
// XXX: Until entry loading is complete just claim it is loaded
Entry->second->NamedJobRefCountMutex.unlock();
}
// XXX: Until entry loading is complete just claim it is loaded
Entry->second->NamedJobRefCountMutex.unlock();
}
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
// XXX: Remove named region objects
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
// XXX: Remove named region objects
// XXX: Until entry loading is complete just claim it is loaded
Entry->NamedJobRefCountMutex.unlock();
}
// XXX: Until entry loading is complete just claim it is loaded
Entry->NamedJobRefCountMutex.unlock();
}
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
// Walk through all of our jobs sequentially until the work queue is empty
while (NamedWorkQueueJobs.load()) {
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
// Walk through all of our jobs sequentially until the work queue is empty
while (NamedWorkQueueJobs.load()) {
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
{
// Lock the work queue mutex for a short moment and grab an item from the list
std::unique_lock lk {NamedWorkQueueMutex};
size_t WorkItems = WorkQueue.size();
if (WorkItems != 0) {
WorkItem = std::move(WorkQueue.front());
WorkQueue.pop();
{
// Lock the work queue mutex for a short moment and grab an item from the list
std::unique_lock lk {NamedWorkQueueMutex};
size_t WorkItems = WorkQueue.size();
if (WorkItems != 0) {
WorkItem = std::move(WorkQueue.front());
WorkQueue.pop();
}
// Atomically update the number of jobs
--NamedWorkQueueJobs;
}
// Atomically update the number of jobs
--NamedWorkQueueJobs;
}
if (WorkItem) {
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion *>(WorkItem.get());
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
}
if (WorkItem) {
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion*>(WorkItem.get());
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
}
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion*>(WorkItem.get());
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion *>(WorkItem.get());
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
}
}
}
}
}
} // namespace FEXCore::CodeSerialize
@@ -6,80 +6,80 @@
#include <FEXCore/Utils/Threads.h>
namespace {
static void* ThreadHandler(void* Arg) {
FEXCore::CodeSerialize::CodeObjectSerializeService* This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
This->ExecutionThread();
return nullptr;
static void* ThreadHandler(void *Arg) {
FEXCore::CodeSerialize::CodeObjectSerializeService *This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
This->ExecutionThread();
return nullptr;
}
}
} // namespace
namespace FEXCore::CodeSerialize {
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx)
: CTX {ctx}
, AsyncHandler {&NamedRegionHandler, this}
, NamedRegionHandler {ctx} {
Initialize();
}
void CodeObjectSerializeService::Shutdown() {
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
return;
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::ContextImpl *ctx)
: CTX {ctx}
, AsyncHandler { &NamedRegionHandler , this }
, NamedRegionHandler { ctx } {
Initialize();
}
WorkerThreadShuttingDown = true;
void CodeObjectSerializeService::Shutdown() {
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
return;
}
// Kick the working thread
WorkAvailable.NotifyAll();
WorkerThreadShuttingDown = true;
if (WorkerThread->joinable()) {
// Wait for worker thread to close down
WorkerThread->join(nullptr);
}
}
// Kick the working thread
WorkAvailable.NotifyAll();
void CodeObjectSerializeService::Initialize() {
// Add a canary so we don't crash on empty map iterator handling
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
FEXCore::Threads::SetSignalMask(OldMask);
}
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it) {
if (Base == ~0ULL) {
// Don't do closure on canary
return;
}
// XXX: Do code region closure
}
const CodeObjectFileSection* CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
// XXX: Actually fetch code objects from cache
return nullptr;
}
void CodeObjectSerializeService::ExecutionThread() {
// Set our thread name so we can see its relation
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
while (WorkerThreadShuttingDown.load() != true) {
// Wait for work
WorkAvailable.Wait();
// Handle named region async jobs first. Highest priority
NamedRegionHandler.HandleNamedRegionObjectJobs();
// XXX: Handle code serialization jobs second.
if (WorkerThread->joinable()) {
// Wait for worker thread to close down
WorkerThread->join(nullptr);
}
}
// Do final code region closures on thread shutdown
for (auto& it : AddressToEntryMap) {
DoCodeRegionClosure(it.first, it.second.get());
void CodeObjectSerializeService::Initialize() {
// Add a canary so we don't crash on empty map iterator handling
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
FEXCore::Threads::SetSignalMask(OldMask);
}
// Safely clear our maps now
AddressToEntryMap.clear();
UnrelocatedAddressToEntryMap.clear();
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it) {
if (Base == ~0ULL) {
// Don't do closure on canary
return;
}
// XXX: Do code region closure
}
CodeObjectFileSection const *CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
// XXX: Actually fetch code objects from cache
return nullptr;
}
void CodeObjectSerializeService::ExecutionThread() {
// Set our thread name so we can see its relation
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
while (WorkerThreadShuttingDown.load() != true) {
// Wait for work
WorkAvailable.Wait();
// Handle named region async jobs first. Highest priority
NamedRegionHandler.HandleNamedRegionObjectJobs();
// XXX: Handle code serialization jobs second.
}
// Do final code region closures on thread shutdown
for (auto &it : AddressToEntryMap) {
DoCodeRegionClosure(it.first, it.second.get());
}
// Safely clear our maps now
AddressToEntryMap.clear();
UnrelocatedAddressToEntryMap.clear();
}
}
} // namespace FEXCore::CodeSerialize
@@ -17,441 +17,445 @@
#include <shared_mutex>
namespace FEXCore::CodeSerialize {
// XXX: Does this need to be signal safe?
using CodeSerializationMutex = std::shared_mutex;
struct CodeSerializationData {};
// XXX: Does this need to be signal safe?
using CodeSerializationMutex = std::shared_mutex;
struct CodeSerializationData {
};
struct CodeObjectFileSection {
bool Serialized;
bool Invalid;
const CodeSerializationData* Data;
const char* HostCode;
uint64_t NumRelocations;
const char* Relocations;
};
/**
* @brief This is the file header that lives at the start of an object cache file
*
* This header is updated from multiple processes!
* Care must be taken to use OS locks when updating the file backing including this header
*/
struct CodeObjectSerializationHeader {
// The configuration that this file has
CodeObjectSerializationConfig Config;
// The original RIP that this object section was mapped at
uint64_t OriginalBase {};
// The original offset in to the file that this object section was loaded from
uint64_t OriginalOffset {};
// Total amount of code that should be in this file
uint64_t TotalCodeSize {};
// Used to reserve the TSL map
uint64_t NumCodeEntries {};
// The number of relocations that point to this section
uint64_t NumRelocationsTo {};
// Total relocations in this file
uint64_t TotalRelocationsCount {};
};
struct CodeRegionEntry {
/**
* @name Threaded initialization objects for the initial object creation
* @{ */
// Base address in memory where the code region is at
uint64_t Base {};
// Size of this code entry
uint64_t Size {};
// The offset inside the file that is mapped to Base
uint64_t Offset {};
// Filename of the object
fextl::string Filename {};
CodeObjectSerializationHeader EntryHeader {};
/** @} */
// The filename of the object cache for this entry
fextl::string ObjectEntrySourceFilename {};
// In the case of file corruption that we can detect, we can disable serialization early for an entry
// We should be resiliant to corruption but things happen
bool StillSerializing {true};
// Long lived FD for serialization if we have multiple jobs to serialize
// Bursts of code entries are common and this reduces file lock overhead
//
// Especially useful over network mounts where file locks are very slow
int CurrentSerializedFD {-1};
struct CodeObjectFileSection {
bool Serialized;
bool Invalid;
const CodeSerializationData *Data;
const char *HostCode;
uint64_t NumRelocations;
const char *Relocations;
};
/**
* @name Objects required to sync objects between threads
* @{ */
// Refcount for the number of outstanding code entries waiting to be written for this object section
CodeSerializationMutex ObjectJobRefCountMutex;
// Refcount for outstanding named object region entry loading itself
// Will block JIT code cache look up when this has a unique_lock held
CodeSerializationMutex NamedJobRefCountMutex;
/** @} */
/**
* @name Object Entry data management
* @{ */
/**
* @name This is the raw file data that we loaded from the code region entry file
* @{ */
char* CodeData {};
size_t FileSize {};
fextl::vector<CodeObjectFileSection> FileCodeSections;
/** @} */
// This per section map takes the most time to load and needs to be quick
// This is the map of all code segments for this entry
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap {};
/** @} */
// Default initialization
CodeRegionEntry() = default;
// Initializer specifically for threaded loading
CodeRegionEntry(uint64_t Base, uint64_t Size, uint64_t Offset, const fextl::string& Filename, const CodeObjectSerializationHeader& DefaultHeader)
: Base {Base}
, Size {Size}
, Offset {Offset}
, Filename {Filename}
, EntryHeader {DefaultHeader} {}
};
// Map type must use an interator that isn't invalidation on erase/insert
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
class NamedRegionObjectHandler;
class CodeObjectSerializeService;
class AsyncJobHandler final {
public:
/**
* @brief Structure containing all the data required to async serialize code objects
* @brief This is the file header that lives at the start of an object cache file
*
* This header is updated from multiple processes!
* Care must be taken to use OS locks when updating the file backing including this header
*/
struct SerializationJobData {
uint64_t GuestRIP; ///< The RIP for the guest
// XXX: Support multiblock
uint64_t GuestCodeLength; ///< The Guest's code length
uint64_t GuestCodeHash; ///< Hash of the guest code
struct CodeObjectSerializationHeader {
// The configuration that this file has
CodeObjectSerializationConfig Config;
// The original RIP that this object section was mapped at
uint64_t OriginalBase{};
// The original offset in to the file that this object section was loaded from
uint64_t OriginalOffset{};
// Total amount of code that should be in this file
uint64_t TotalCodeSize{};
// Used to reserve the TSL map
uint64_t NumCodeEntries{};
// The number of relocations that point to this section
uint64_t NumRelocationsTo{};
// Total relocations in this file
uint64_t TotalRelocationsCount{};
};
void* HostCodeBegin; ///< Host JIT code starting memory address
size_t HostCodeLength; ///< Host JIT code length
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
struct CodeRegionEntry {
/**
* @name Threaded initialization objects for the initial object creation
* @{ */
// Base address in memory where the code region is at
uint64_t Base{};
// This is the thread specific ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
// This way it will wait until the async job handler is complete with it.
CodeSerializationMutex* ThreadJobRefCount;
// Size of this code entry
uint64_t Size{};
// These are the reolocations for this serialization job
// Relatively small number of entries most of the time
fextl::vector<FEXCore::CPU::Relocation> Relocations;
// The offset inside the file that is mapped to Base
uint64_t Offset{};
// Filename of the object
fextl::string Filename{};
CodeObjectSerializationHeader EntryHeader{};
/** @} */
// The filename of the object cache for this entry
fextl::string ObjectEntrySourceFilename{};
// In the case of file corruption that we can detect, we can disable serialization early for an entry
// We should be resiliant to corruption but things happen
bool StillSerializing {true};
// Long lived FD for serialization if we have multiple jobs to serialize
// Bursts of code entries are common and this reduces file lock overhead
//
// Especially useful over network mounts where file locks are very slow
int CurrentSerializedFD {-1};
/**
* @name Objects filled in from the Code Object Serialization service when a job is added
* @name Objects required to sync objects between threads
* @{ */
// This is the code region's ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
CodeSerializationMutex* ObjectJobRefCountMutexPtr;
// Refcount for the number of outstanding code entries waiting to be written for this object section
CodeSerializationMutex ObjectJobRefCountMutex;
// This is the code region iterator to reduce the number of map lookups
// This will remain valid while jobs are outstanding for this region
CodeRegionMapType::iterator CodeRegionIterator;
// Refcount for outstanding named object region entry loading itself
// Will block JIT code cache look up when this has a unique_lock held
CodeSerializationMutex NamedJobRefCountMutex;
/** @} */
/**
* @name Object Entry data management
* @{ */
/**
* @name This is the raw file data that we loaded from the code region entry file
* @{ */
char *CodeData{};
size_t FileSize{};
fextl::vector<CodeObjectFileSection> FileCodeSections;
/** @} */
// This per section map takes the most time to load and needs to be quick
// This is the map of all code segments for this entry
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap{};
/** @} */
// Default initialization
CodeRegionEntry() = default;
// Initializer specifically for threaded loading
CodeRegionEntry(uint64_t Base,
uint64_t Size,
uint64_t Offset,
fextl::string const &Filename,
CodeObjectSerializationHeader const &DefaultHeader)
: Base {Base}
, Size {Size}
, Offset {Offset}
, Filename {Filename}
, EntryHeader {DefaultHeader} {
}
};
AsyncJobHandler(NamedRegionObjectHandler* NamedRegionHandler, CodeObjectSerializeService* CodeObjectCacheService)
: NamedRegionHandler {NamedRegionHandler}
, CodeObjectCacheService {CodeObjectCacheService} {}
// Map type must use an interator that isn't invalidation on erase/insert
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
protected:
friend class CodeObjectSerializeService;
friend class NamedRegionObjectHandler;
/**
* @name Async job submission functions
* @{ */
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename);
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
/** @} */
class NamedRegionObjectHandler;
class CodeObjectSerializeService;
/**
* @name Async named region handling
* @{ */
/**
* @brief The async named region jobs to handle.
*
* Only two, Code serialization goes in to a different queue.
*/
enum class NamedRegionJobType {
JOB_ADD_NAMED_REGION,
JOB_REMOVE_NAMED_REGION,
class AsyncJobHandler final {
public:
/**
* @brief Structure containing all the data required to async serialize code objects
*/
struct SerializationJobData {
uint64_t GuestRIP; ///< The RIP for the guest
// XXX: Support multiblock
uint64_t GuestCodeLength; ///< The Guest's code length
uint64_t GuestCodeHash; ///< Hash of the guest code
void *HostCodeBegin; ///< Host JIT code starting memory address
size_t HostCodeLength; ///< Host JIT code length
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
// This is the thread specific ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
// This way it will wait until the async job handler is complete with it.
CodeSerializationMutex *ThreadJobRefCount;
// These are the reolocations for this serialization job
// Relatively small number of entries most of the time
fextl::vector<FEXCore::CPU::Relocation> Relocations;
/**
* @name Objects filled in from the Code Object Serialization service when a job is added
* @{ */
// This is the code region's ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
CodeSerializationMutex *ObjectJobRefCountMutexPtr;
// This is the code region iterator to reduce the number of map lookups
// This will remain valid while jobs are outstanding for this region
CodeRegionMapType::iterator CodeRegionIterator;
/** @} */
};
AsyncJobHandler(NamedRegionObjectHandler *NamedRegionHandler, CodeObjectSerializeService *CodeObjectCacheService)
: NamedRegionHandler {NamedRegionHandler}
, CodeObjectCacheService {CodeObjectCacheService} {}
protected:
friend class CodeObjectSerializeService;
friend class NamedRegionObjectHandler;
/**
* @name Async job submission functions
* @{ */
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename);
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
/** @} */
/**
* @name Async named region handling
* @{ */
/**
* @brief The async named region jobs to handle.
*
* Only two, Code serialization goes in to a different queue.
*/
enum class NamedRegionJobType {
JOB_ADD_NAMED_REGION,
JOB_REMOVE_NAMED_REGION,
};
class NamedRegionWorkItem {
public:
NamedRegionJobType GetType() const { return Type; }
protected:
friend class WorkItemAddNamedRegion;
NamedRegionWorkItem(NamedRegionJobType type)
: Type {type} {}
private:
NamedRegionJobType Type;
};
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
public:
WorkItemAddNamedRegion(const fextl::string &base, const fextl::string &filename, bool executable, CodeRegionMapType::iterator entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
, BaseFilename {base}
, Filename {filename}
, Executable {executable}
, Entry {entry}
{}
const fextl::string BaseFilename;
const fextl::string Filename;
bool Executable;
CodeRegionMapType::iterator Entry;
};
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
public:
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
, Base {base}
, Size {size}
, Entry {std::move(entry)} {}
uint64_t Base;
uint64_t Size;
fextl::unique_ptr<CodeRegionEntry> Entry;
};
/** @} */
private:
NamedRegionObjectHandler *NamedRegionHandler;
CodeObjectSerializeService *CodeObjectCacheService;
};
class NamedRegionWorkItem {
public:
NamedRegionJobType GetType() const {
return Type;
}
class NamedRegionObjectHandler final {
public:
NamedRegionObjectHandler(FEXCore::Context::ContextImpl *ctx);
protected:
friend class WorkItemAddNamedRegion;
NamedRegionWorkItem(NamedRegionJobType type)
: Type {type} {}
void HandleNamedRegionObjectJobs();
private:
NamedRegionJobType Type;
CodeObjectSerializationConfig const &GetDefaultSerializationConfig() const {
return DefaultSerializationConfig;
}
protected:
friend class AsyncJobHandler;
// Return a default code header based off the default serialization config
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
return CodeObjectSerializationHeader {
.Config = DefaultSerializationConfig,
.OriginalBase = Base,
.OriginalOffset = Offset,
.NumCodeEntries = 0,
.NumRelocationsTo = 0,
.TotalRelocationsCount = 0,
};
}
/**
* @brief Adds an asynchronous add named region work item to the object queue
*
* This adds the job that will do the loading of file resources and data tracking.
*/
void AsyncAddNamedRegionWorkItem(const fextl::string &base, const fextl::string &filename, bool executable, CodeRegionMapType::iterator entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion> (
base,
filename,
executable,
entry
));
++NamedWorkQueueJobs;
}
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion> (
Base,
Size,
std::move(Entry)
));
++NamedWorkQueueJobs;
}
private:
// Code version. If the code emission changes then this needs to increment
constexpr static uint32_t CODE_VERSION = 0x0;
// Default cookie header for the file header
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
// Code serialization config for our current process configuration
CodeObjectSerializationConfig DefaultSerializationConfig;
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
std::atomic<uint64_t> NamedWorkQueueJobs{};
// Mutex for ading new jobs to the work queue
std::mutex NamedWorkQueueMutex{};
// The job queue itself
// Jobs get consumed as a FIFO
// Jobs always get appended to the end
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue{};
/**
* @name Named Region object handling
* @{ */
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string &base_filename, const fextl::string &filename, bool Executable);
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
/** @} */
};
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
public:
WorkItemAddNamedRegion(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
, BaseFilename {base}
, Filename {filename}
, Executable {executable}
, Entry {entry} {}
const fextl::string BaseFilename;
const fextl::string Filename;
bool Executable;
CodeRegionMapType::iterator Entry;
/**
* @brief Context specific code object serialization class
*
* Contains everything required for FEXCore to serialize code objects
*/
class CodeObjectSerializeService final {
public:
CodeObjectSerializeService(FEXCore::Context::ContextImpl *ctx);
/**
* @brief Initialize the internal interface
*
* Is a public interface to allow the service to reinitialize after forking
*/
void Initialize();
/**
* @brief Safely shut down the Code Object serialization service.
*
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
*/
void Shutdown();
/**
* @name Async interface
* @{ */
/**
* @brief Loads a named region in to the code serialization service. As async as possible.
*
* @param Base - Virtual address that this named region is loaded
* @param Size - The size of the region
* @param Offset - The offset from the file
* @param filename - The filename itself
*/
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename) {
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
}
/**
* @brief Unloads a named region from the code serialization service. As async as possible.
*
* @param Base - Virtual address of the named region
* @param Size - The size of the region
*/
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
}
/**
* @brief Adds a code object serialization job. As async as possible.
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
*
* @param Data - A fully filled out struct containing all the code serialization
*/
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
}
/** @} */
/**
* @name Synchronous interface
* @{ */
/**
* @brief Synchronously waits for this thread's job queue to become empty.
*
* This is necessary for when a thread is shutting down
*
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
*/
static void WaitForEmptyJobQueue(CodeSerializationMutex *ThreadJobRefCount) {
// Once the shared mutex is empty this unique lock will be gained
std::unique_lock lk {*ThreadJobRefCount};
}
/**
* @brief Fetches object code from the Code Object Cache for JIT.
*
* @param GuestRIP - Which GuestRIP to search the cache for
*
* @return Data required for the JIT to relocate the Object code.
*/
CodeObjectFileSection const *FetchCodeObjectFromCache(uint64_t GuestRIP);
/** @} */
// Public for threading
void ExecutionThread();
protected:
friend class AsyncJobHandler;
/**
* @brief Safely closes out code object regions from the map
*
* @param it - iterator to do a closure on
*/
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it);
CodeSerializationMutex &GetEntryMapMutex() { return EntryMapMutex; }
CodeSerializationMutex &GetUnrelocatedEntryMapMutex() { return EntryMapMutex; }
CodeRegionMapType &GetEntryMap() { return AddressToEntryMap; }
CodeRegionPtrMapType &GetUnrelocatedEntryMap() { return UnrelocatedAddressToEntryMap; }
/**
* @brief Notify the async thread that it has work to do
*/
void NotifyWork() { WorkAvailable.NotifyOne(); }
private:
FEXCore::Context::ContextImpl *CTX;
Event WorkAvailable{};
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
std::atomic_bool WorkerThreadShuttingDown {false};
AsyncJobHandler AsyncHandler;
NamedRegionObjectHandler NamedRegionHandler;
// Mutex to hold when modifying the entry maps
CodeSerializationMutex EntryMapMutex;
CodeSerializationMutex UnrelocatedEntryMapMutex;
// Entry maps
CodeRegionMapType AddressToEntryMap;
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
};
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
public:
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
, Base {base}
, Size {size}
, Entry {std::move(entry)} {}
uint64_t Base;
uint64_t Size;
fextl::unique_ptr<CodeRegionEntry> Entry;
};
/** @} */
private:
NamedRegionObjectHandler* NamedRegionHandler;
CodeObjectSerializeService* CodeObjectCacheService;
};
class NamedRegionObjectHandler final {
public:
NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx);
void HandleNamedRegionObjectJobs();
const CodeObjectSerializationConfig& GetDefaultSerializationConfig() const {
return DefaultSerializationConfig;
}
protected:
friend class AsyncJobHandler;
// Return a default code header based off the default serialization config
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
return CodeObjectSerializationHeader {
.Config = DefaultSerializationConfig,
.OriginalBase = Base,
.OriginalOffset = Offset,
.NumCodeEntries = 0,
.NumRelocationsTo = 0,
.TotalRelocationsCount = 0,
};
}
/**
* @brief Adds an asynchronous add named region work item to the object queue
*
* This adds the job that will do the loading of file resources and data tracking.
*/
void AsyncAddNamedRegionWorkItem(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion>(base, filename, executable, entry));
++NamedWorkQueueJobs;
}
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion>(Base, Size, std::move(Entry)));
++NamedWorkQueueJobs;
}
private:
// Code version. If the code emission changes then this needs to increment
constexpr static uint32_t CODE_VERSION = 0x0;
// Default cookie header for the file header
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
// Code serialization config for our current process configuration
CodeObjectSerializationConfig DefaultSerializationConfig;
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
std::atomic<uint64_t> NamedWorkQueueJobs {};
// Mutex for ading new jobs to the work queue
std::mutex NamedWorkQueueMutex {};
// The job queue itself
// Jobs get consumed as a FIFO
// Jobs always get appended to the end
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue {};
/**
* @name Named Region object handling
* @{ */
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename, const fextl::string& filename, bool Executable);
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
/** @} */
};
/**
* @brief Context specific code object serialization class
*
* Contains everything required for FEXCore to serialize code objects
*/
class CodeObjectSerializeService final {
public:
CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx);
/**
* @brief Initialize the internal interface
*
* Is a public interface to allow the service to reinitialize after forking
*/
void Initialize();
/**
* @brief Safely shut down the Code Object serialization service.
*
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
*/
void Shutdown();
/**
* @name Async interface
* @{ */
/**
* @brief Loads a named region in to the code serialization service. As async as possible.
*
* @param Base - Virtual address that this named region is loaded
* @param Size - The size of the region
* @param Offset - The offset from the file
* @param filename - The filename itself
*/
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
}
/**
* @brief Unloads a named region from the code serialization service. As async as possible.
*
* @param Base - Virtual address of the named region
* @param Size - The size of the region
*/
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
}
/**
* @brief Adds a code object serialization job. As async as possible.
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
*
* @param Data - A fully filled out struct containing all the code serialization
*/
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
}
/** @} */
/**
* @name Synchronous interface
* @{ */
/**
* @brief Synchronously waits for this thread's job queue to become empty.
*
* This is necessary for when a thread is shutting down
*
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
*/
static void WaitForEmptyJobQueue(CodeSerializationMutex* ThreadJobRefCount) {
// Once the shared mutex is empty this unique lock will be gained
std::unique_lock lk {*ThreadJobRefCount};
}
/**
* @brief Fetches object code from the Code Object Cache for JIT.
*
* @param GuestRIP - Which GuestRIP to search the cache for
*
* @return Data required for the JIT to relocate the Object code.
*/
const CodeObjectFileSection* FetchCodeObjectFromCache(uint64_t GuestRIP);
/** @} */
// Public for threading
void ExecutionThread();
protected:
friend class AsyncJobHandler;
/**
* @brief Safely closes out code object regions from the map
*
* @param it - iterator to do a closure on
*/
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it);
CodeSerializationMutex& GetEntryMapMutex() {
return EntryMapMutex;
}
CodeSerializationMutex& GetUnrelocatedEntryMapMutex() {
return EntryMapMutex;
}
CodeRegionMapType& GetEntryMap() {
return AddressToEntryMap;
}
CodeRegionPtrMapType& GetUnrelocatedEntryMap() {
return UnrelocatedAddressToEntryMap;
}
/**
* @brief Notify the async thread that it has work to do
*/
void NotifyWork() {
WorkAvailable.NotifyOne();
}
private:
FEXCore::Context::ContextImpl* CTX;
Event WorkAvailable {};
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
std::atomic_bool WorkerThreadShuttingDown {false};
AsyncJobHandler AsyncHandler;
NamedRegionObjectHandler NamedRegionHandler;
// Mutex to hold when modifying the entry maps
CodeSerializationMutex EntryMapMutex;
CodeSerializationMutex UnrelocatedEntryMapMutex;
// Entry maps
CodeRegionMapType AddressToEntryMap;
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
};
} // namespace FEXCore::CodeSerialize
}
@@ -3,77 +3,77 @@
#include <FEXCore/IR/IR.h>
namespace FEXCore::CPU {
enum class RelocationTypes : uint8_t {
// 8 byte literal in memory for symbol
// Aligned to struct RelocNamedSymbolLiteral
RELOC_NAMED_SYMBOL_LITERAL,
enum class RelocationTypes : uint8_t {
// 8 byte literal in memory for symbol
// Aligned to struct RelocNamedSymbolLiteral
RELOC_NAMED_SYMBOL_LITERAL,
// Fixed size named thunk move
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocNamedThunkMove
RELOC_NAMED_THUNK_MOVE,
// Fixed size named thunk move
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocNamedThunkMove
RELOC_NAMED_THUNK_MOVE,
// Fixed size guest RIP move
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocGuestRIPMove
RELOC_GUEST_RIP_MOVE,
};
struct RelocationTypeHeader final {
RelocationTypes Type;
};
struct RelocNamedSymbolLiteral final {
enum class NamedSymbol : uint8_t {
///< Thread specific relocations
// JIT Literal pointers
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
// Fixed size guest RIP move
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocGuestRIPMove
RELOC_GUEST_RIP_MOVE,
};
RelocationTypeHeader Header {};
struct RelocationTypeHeader final {
RelocationTypes Type;
};
NamedSymbol Symbol;
struct RelocNamedSymbolLiteral final {
enum class NamedSymbol : uint8_t {
///< Thread specific relocations
// JIT Literal pointers
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
};
// Offset in to the code section to begin the relocation
uint64_t Offset {};
};
RelocationTypeHeader Header{};
struct RelocNamedThunkMove final {
RelocationTypeHeader Header {};
NamedSymbol Symbol;
// GPR index the constant is being moved to
uint8_t RegisterIndex;
// Offset in to the code section to begin the relocation
uint64_t Offset{};
};
// The thunk SHA256 hash
IR::SHA256Sum Symbol;
struct RelocNamedThunkMove final {
RelocationTypeHeader Header{};
// Offset in to the code section to begin the relocation
uint64_t Offset {};
};
// GPR index the constant is being moved to
uint8_t RegisterIndex;
struct RelocGuestRIPMove final {
RelocationTypeHeader Header {};
// The thunk SHA256 hash
IR::SHA256Sum Symbol;
// GPR index the constant is being moved to
uint8_t RegisterIndex;
// Offset in to the code section to begin the relocation
uint64_t Offset{};
};
// Offset in to the code section to begin the relocation
uint64_t Offset {};
struct RelocGuestRIPMove final {
RelocationTypeHeader Header{};
// The unrelocated RIP that is being moved
uint64_t GuestRIP;
};
// GPR index the constant is being moved to
uint8_t RegisterIndex;
union Relocation {
RelocationTypeHeader Header {};
// Offset in to the code section to begin the relocation
uint64_t Offset{};
RelocNamedSymbolLiteral NamedSymbolLiteral;
// This makes our union of relocations at least 48 bytes
// It might be more efficient to not use a union
RelocNamedThunkMove NamedThunkMove;
// The unrelocated RIP that is being moved
uint64_t GuestRIP;
};
RelocGuestRIPMove GuestRIPMove;
};
} // namespace FEXCore::CPU
union Relocation {
RelocationTypeHeader Header{};
RelocNamedSymbolLiteral NamedSymbolLiteral;
// This makes our union of relocations at least 48 bytes
// It might be more efficient to not use a union
RelocNamedThunkMove NamedThunkMove;
RelocGuestRIPMove GuestRIPMove;
};
}
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -8,6 +8,7 @@ $end_info$
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Core/OpcodeDispatcher.h"
@@ -22,113 +23,92 @@ class OrderedNode;
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode* RotatedNode {};
if (CTX->HostFeatures.SupportsSHA) {
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
} else {
// SHA1 extension missing, manually rotate.
// Emulate rotate.
auto ShiftLeft = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Dest, 30);
RotatedNode = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeft, Dest, 2);
}
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
auto Tmp = _Ror(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), _Constant(32, 2));
auto Top = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Src, 3), Tmp);
auto Result = _VInsGPR(16, 4, 3, Src, Top);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode* NewVec = _VExtr(16, 8, Dest, Src, 1);
OrderedNode *NewVec = _VExtr(16, 8, Dest, Src, 1);
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
OrderedNode* Result = _VXor(16, 1, Dest, NewVec);
OrderedNode *Result = _VXor(16, 1, Dest, NewVec);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
// Do all the work without it.
// ROR by 31 is equivalent to a ROL by 1
auto ThirtyOne = _Constant(32, 31);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(OpSize::i32Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
auto W13 = _VExtractToGPR(16, 4, Src, 2);
auto W14 = _VExtractToGPR(16, 4, Src, 1);
auto W15 = _VExtractToGPR(16, 4, Src, 0);
auto W16 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), W13), ThirtyOne);
auto W17 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 2), W14), ThirtyOne);
auto W18 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 1), W15), ThirtyOne);
auto W19 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 0), W16), ThirtyOne);
// Shift the incoming source left by a 32-bit element, inserting Zeros.
// This could be slightly improved to use a VInsGPR with the zero register.
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
auto D3 = _VInsGPR(16, 4, 3, Dest, W16);
auto D2 = _VInsGPR(16, 4, 2, D3, W17);
auto D1 = _VInsGPR(16, 4, 1, D2, W18);
auto D0 = _VInsGPR(16, 4, 0, D1, W19);
// Emulate rotate.
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
// Element0 didn't get XOR'd with anything, so do it now.
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
// Emulate rotate.
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
StoreResult(FPRClass, Op, Result, -1);
StoreResult(FPRClass, Op, D0, -1);
}
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here to indicate function and constants");
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(),
"Src1 needs to be literal here to indicate function and constants");
using FnType = OrderedNode* (*)(OpDispatchBuilder&, OrderedNode*, OrderedNode*, OrderedNode*);
const auto f0 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
const auto f0 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._Andn(OpSize::i32Bit, D, B));
};
const auto f1 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
const auto f1 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
};
const auto f2 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
return Self.BitwiseAtLeastTwo(B, C, D);
const auto f2 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._And(OpSize::i32Bit, B, D)), Self._And(OpSize::i32Bit, C, D));
};
const auto f3 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
const auto f3 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
};
constexpr std::array<uint32_t, 4> k_array {
constexpr std::array<uint32_t, 4> k_array{
0x5A827999U,
0x6ED9EBA1U,
0x8F1BBCDCU,
0xCA62C1D6U,
};
constexpr std::array<FnType, 4> fn_array {
f0,
f1,
f2,
f3,
constexpr std::array<FnType, 4> fn_array{
f0, f1, f2, f3,
};
const uint64_t Imm8 = Op->Src[1].Data.Literal.Value & 0b11;
const FnType Fn = fn_array[Imm8];
auto K = _Constant(32, k_array[Imm8]);
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto W0E = _VExtractToGPR(16, 4, Src, 3);
auto W1 = _VExtractToGPR(16, 4, Src, 2);
auto W2 = _VExtractToGPR(16, 4, Src, 1);
auto W3 = _VExtractToGPR(16, 4, Src, 0);
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
@@ -138,8 +118,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
auto C = _VExtractToGPR(16, 4, Dest, 1);
auto D = _VExtractToGPR(16, 4, Dest, 0);
auto A1 =
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W0E), K);
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W0E), K);
auto B1 = A;
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
auto D1 = C;
@@ -147,14 +126,9 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
return {A1, B1, C1, D1, E1};
};
const auto Round1To3 = [&](OrderedNode* A, OrderedNode* B, OrderedNode* C, OrderedNode* D, OrderedNode* E, OrderedNode* Src,
unsigned W_idx) -> RoundResult {
// Kill W and E at the beginning
auto W = _VExtractToGPR(16, 4, Src, W_idx);
auto Q = _Add(OpSize::i32Bit, W, E);
auto ANext =
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), Q), K);
const auto Round1To3 = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C,
OrderedNode *D, OrderedNode *E, OrderedNode *W) -> RoundResult {
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W), E), K);
auto BNext = A;
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
auto DNext = C;
@@ -164,11 +138,11 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
};
auto [A1, B1, C1, D1, E1] = Round0();
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, W1);
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, W2);
auto Final = Round1To3(A3, B3, C3, D3, E3, W3);
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
auto Dest2 = _VInsGPR(16, 4, 2, Dest3, std::get<1>(Final));
auto Dest1 = _VInsGPR(16, 4, 1, Dest2, std::get<2>(Final));
auto Dest0 = _VInsGPR(16, 4, 0, Dest1, std::get<3>(Final));
@@ -177,47 +151,39 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
}
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
};
OrderedNode* Result {};
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
if (CTX->HostFeatures.SupportsSHA) {
Result = _VSha256U0(Dest, Src);
} else {
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))),
_Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
};
auto W4 = _VExtractToGPR(16, 4, Src, 0);
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
auto W4 = _VExtractToGPR(16, 4, Src, 0);
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
auto D0 = _VInsGPR(16, 4, 0, D1, Sig0);
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
Result = _VInsGPR(16, 4, 0, D1, Sig0);
}
StoreResult(FPRClass, Op, Result, -1);
StoreResult(FPRClass, Op, D0, -1);
}
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
const auto Sigma1 = [this](OrderedNode* W) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))),
_Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))), _Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
};
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto W14 = _VExtractToGPR(16, 4, Src, 2);
auto W15 = _VExtractToGPR(16, 4, Src, 3);
@@ -234,87 +200,76 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
StoreResult(FPRClass, Op, D0, -1);
}
OrderedNode* OpDispatchBuilder::BitwiseAtLeastTwo(OrderedNode* A, OrderedNode* B, OrderedNode* C) {
// Returns whether at least 2/3 of A/B/C is true.
// Expressed as (A & (B | C)) | (B & C)
//
// Equivalent to expression in SHA calculations: (A & B) ^ (A & C) ^ (B & C)
auto And = _And(OpSize::i32Bit, B, C);
auto Or = _Or(OpSize::i32Bit, B, C);
return _Or(OpSize::i32Bit, _And(OpSize::i32Bit, A, Or), And);
}
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
const auto Ch = [this](OrderedNode* E, OrderedNode* F, OrderedNode* G) -> OrderedNode* {
const auto Ch = [this](OrderedNode *E, OrderedNode *F, OrderedNode *G) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
};
const auto Sigma0 = [this](OrderedNode* A) -> OrderedNode* {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), A, ShiftType::ROR, 13), A,
ShiftType::ROR, 22);
const auto Major = [this](OrderedNode *A, OrderedNode *B, OrderedNode *C) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, A, B), _And(OpSize::i32Bit, A, C)), _And(OpSize::i32Bit, B, C));
};
const auto Sigma1 = [this](OrderedNode* E) -> OrderedNode* {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), E, ShiftType::ROR, 11), E,
ShiftType::ROR, 25);
const auto Sigma0 = [this](OrderedNode *A) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), _Ror(OpSize::i32Bit, A, _Constant(32, 13))), _Ror(OpSize::i32Bit, A, _Constant(32, 22)));
};
const auto Sigma1 = [this](OrderedNode *E) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), _Ror(OpSize::i32Bit, E, _Constant(32, 11))), _Ror(OpSize::i32Bit, E, _Constant(32, 25)));
};
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// Hardcoded to XMM0
auto XMM0 = LoadXMMRegister(0);
auto E0 = _VExtractToGPR(16, 4, Src, 1);
auto F0 = _VExtractToGPR(16, 4, Src, 0);
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
OrderedNode* Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
Q0 = _Add(OpSize::i32Bit, Q0, H0);
auto A0 = _VExtractToGPR(16, 4, Src, 3);
auto B0 = _VExtractToGPR(16, 4, Src, 2);
auto C0 = _VExtractToGPR(16, 4, Dest, 3);
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
auto D0 = _VExtractToGPR(16, 4, Dest, 2);
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
OrderedNode* Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
auto E0 = _VExtractToGPR(16, 4, Src, 1);
auto F0 = _VExtractToGPR(16, 4, Src, 0);
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
auto WK1 = _VExtractToGPR(16, 4, XMM0, 1);
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
G0 = _VExtractToGPR(16, 4, Dest, 1);
Q1 = _Add(OpSize::i32Bit, Q1, G0);
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*,
OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
const auto Round = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C, OrderedNode *D,
OrderedNode *E, OrderedNode *F, OrderedNode *G, OrderedNode *H,
OrderedNode* WK) -> RoundResult {
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Ch(E, F, G), Sigma1(E)), WK), H), Major(A, B, C)), Sigma0(A));
auto BNext = A;
auto CNext = B;
auto DNext = C;
auto ENext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Ch(E, F, G), Sigma1(E)), WK), H), D);
auto FNext = E;
auto GNext = F;
auto HNext = G;
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
return {ANext, BNext, CNext, DNext, ENext, FNext, GNext, HNext};
};
// Rematerialize C0. As with G0.
C0 = _VExtractToGPR(16, 4, Dest, 3);
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
auto Res3 = _VInsGPR(16, 4, 3, Dest, A2);
auto Res2 = _VInsGPR(16, 4, 2, Res3, A1);
auto Res1 = _VInsGPR(16, 4, 1, Res2, E2);
auto Res0 = _VInsGPR(16, 4, 0, Res1, E1);
auto [A1, B1, C1, D1, E1, F1, G1, H1] = Round(A0, B0, C0, D0, E0, F0, G0, H0, WK0);
auto Final = Round(A1, B1, C1, D1, E1, F1, G1, H1, WK1);
auto Res3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
auto Res2 = _VInsGPR(16, 4, 2, Res3, std::get<1>(Final));
auto Res1 = _VInsGPR(16, 4, 1, Res2, std::get<4>(Final));
auto Res0 = _VInsGPR(16, 4, 0, Res1, std::get<5>(Final));
StoreResult(FPRClass, Op, Res0, -1);
}
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode* Result = _VAESImc(Src);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Result = _VAESImc(Src);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode* Result = _VAESEnc(16, Dest, Src, ZeroRegister);
OrderedNode *Result = _VAESEnc(16, Dest, Src, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -325,19 +280,19 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESENC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode* Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
OrderedNode *Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode* Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
OrderedNode *Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -348,19 +303,19 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESENCLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode* Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
OrderedNode *Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode* Result = _VAESDec(16, Dest, Src, ZeroRegister);
OrderedNode *Result = _VAESDec(16, Dest, Src, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -371,19 +326,19 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESDEC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode* Result = _VAESDec(DstSize, State, Key, ZeroRegister);
OrderedNode *Result = _VAESDec(DstSize, State, Key, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode* Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
OrderedNode *Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -394,16 +349,16 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESDECLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode* Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
OrderedNode *Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
const uint64_t RCON = Op->Src[1].Data.Literal.Value;
@@ -413,15 +368,15 @@ OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
}
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
OrderedNode* Result = AESKeyGenAssistImpl(Op);
OrderedNode *Result = AESKeyGenAssistImpl(Op);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Selector needs to be literal here");
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[1].Data.Literal.Value);
auto Res = _PCLMUL(16, Dest, Src, Selector);
@@ -433,12 +388,12 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
const auto DstSize = GetDstSize(Op);
OrderedNode* Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode* Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
OrderedNode* Res = _PCLMUL(DstSize, Src1, Src2, Selector);
OrderedNode *Res = _PCLMUL(DstSize, Src1, Src2, Selector);
StoreResult(FPRClass, Op, Res, -1);
}
} // namespace FEXCore::IR
}
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -13,6 +13,7 @@ $end_info$
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/IR/IREmitter.h>
#include <stddef.h>
#include <stdint.h>
@@ -22,24 +23,24 @@ class OrderedNode;
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
// Functions in X87.cpp (no change required)
// GetX87Top
// SetX87ValidTag
// GetX87ValidTag
// GetX87Tag (will need changing once special tag handling is implemented)
// SetX87FTW
// GetX87FTW (will need changing once special tag handling is implemented)
// SetX87Top
// X87ModifySTP
// EMMS
// FFREE
// FNSTENV
// FSTCW
// LDSW
// FNSTSW
// FXCH
// FCMOV
// FST(register to register)
//Functions in X87.cpp (no change required)
//GetX87Top
//SetX87ValidTag
//GetX87ValidTag
//GetX87Tag (will need changing once special tag handling is implemented)
//SetX87FTW
//GetX87FTW (will need changing once special tag handling is implemented)
//SetX87Top
//X87ModifySTP
//EMMS
//FFREE
//FNSTENV
//FSTCW
//LDSW
//FNSTSW
//FXCH
//FCMOV
//FST(register to register)
// State loading duplicated from X87.cpp, setting host rounding mode
// See issue
@@ -64,33 +65,34 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
}
void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
const auto Size = GetSrcSize(Op);
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Src[0]);
auto Size = GetSrcSize(Op);
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
Mem = AppendSegmentOffset(Mem, Op->Flags);
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
OrderedNode* roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
//ignore the rounding precision, we're always 64-bit in F64.
//extract rounding mode
OrderedNode *roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
_SetRoundingMode(roundingMode);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
ReconstructX87StateFromFSW(NewFSW);
{
// FTW
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
}
}
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
OrderedNode* NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
OrderedNode* roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
//ignore the rounding precision, we're always 64-bit in F64.
//extract rounding mode
OrderedNode *roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
_SetRoundingMode(roundingMode);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
}
@@ -105,13 +107,13 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
size_t read_width = (width == 80) ? 16 : width / 8;
OrderedNode* data {};
OrderedNode* converted {};
OrderedNode *data{};
OrderedNode *converted{};
if (!Op->Src[0].IsNone()) {
// Read from memory
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags);
// Convert to 64bit float
// Convert to 64bit float
if constexpr (width == 32) {
converted = _Float_FToF(8, 4, data);
} else if constexpr (width == 80) {
@@ -119,7 +121,8 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
} else {
converted = data;
}
} else {
}
else {
// Implicit arg (does this need to change with width?)
auto offset = _Constant(Op->OP & 7);
data = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, orig_top, offset), mask);
@@ -134,9 +137,12 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
_StoreContextIndexed(converted, top, 8, MMBaseOffset(), 16, FPRClass);
}
template void OpDispatchBuilder::FLDF64<32>(OpcodeArgs);
template void OpDispatchBuilder::FLDF64<64>(OpcodeArgs);
template void OpDispatchBuilder::FLDF64<80>(OpcodeArgs);
template
void OpDispatchBuilder::FLDF64<32>(OpcodeArgs);
template
void OpDispatchBuilder::FLDF64<64>(OpcodeArgs);
template
void OpDispatchBuilder::FLDF64<80>(OpcodeArgs);
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
// Update TOP
@@ -147,8 +153,8 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
SetX87Top(top);
// Read from memory
OrderedNode* data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
OrderedNode* converted = _F80BCDLoad(data);
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
OrderedNode *converted = _F80BCDLoad(data);
converted = _F80CVT(8, converted);
_StoreContextIndexed(converted, top, 8, MMBaseOffset(), 16, FPRClass);
}
@@ -157,7 +163,7 @@ void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
auto orig_top = GetX87Top();
auto data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode* converted = _F80CVTTo(data, 8);
OrderedNode *converted = _F80CVTTo(data, 8);
converted = _F80BCDStore(converted);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1);
@@ -180,13 +186,20 @@ void OpDispatchBuilder::FLDF64_Const(OpcodeArgs) {
_StoreContextIndexed(data, top, 8, MMBaseOffset(), 16, FPRClass);
}
template void OpDispatchBuilder::FLDF64_Const<0x3FF0000000000000>(OpcodeArgs); // 1.0
template void OpDispatchBuilder::FLDF64_Const<0x400A934F0979A372>(OpcodeArgs); // log2l(10)
template void OpDispatchBuilder::FLDF64_Const<0x3FF71547652B82FE>(OpcodeArgs); // log2l(e)
template void OpDispatchBuilder::FLDF64_Const<0x400921FB54442D18>(OpcodeArgs); // pi
template void OpDispatchBuilder::FLDF64_Const<0x3FD34413509F79FF>(OpcodeArgs); // log10l(2)
template void OpDispatchBuilder::FLDF64_Const<0x3FE62E42FEFA39EF>(OpcodeArgs); // log(2)
template void OpDispatchBuilder::FLDF64_Const<0>(OpcodeArgs); // 0.0
template
void OpDispatchBuilder::FLDF64_Const<0x3FF0000000000000>(OpcodeArgs); // 1.0
template
void OpDispatchBuilder::FLDF64_Const<0x400A934F0979A372>(OpcodeArgs); // log2l(10)
template
void OpDispatchBuilder::FLDF64_Const<0x3FF71547652B82FE>(OpcodeArgs); // log2l(e)
template
void OpDispatchBuilder::FLDF64_Const<0x400921FB54442D18>(OpcodeArgs); // pi
template
void OpDispatchBuilder::FLDF64_Const<0x3FD34413509F79FF>(OpcodeArgs); // log10l(2)
template
void OpDispatchBuilder::FLDF64_Const<0x3FE62E42FEFA39EF>(OpcodeArgs); // log(2)
template
void OpDispatchBuilder::FLDF64_Const<0>(OpcodeArgs); // 0.0
void OpDispatchBuilder::FILDF64(OpcodeArgs) {
// Update TOP
@@ -198,7 +211,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
size_t read_width = GetSrcSize(Op);
// Read from memory
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags);
if (read_width == 2) {
if(read_width == 2) {
data = _Sbfe(OpSize::i64Bit, read_width * 8, 0, data);
}
auto converted = _Float_FromGPR_S(8, read_width == 4 ? 4 : 8, data);
@@ -211,14 +224,14 @@ void OpDispatchBuilder::FSTF64(OpcodeArgs) {
auto orig_top = GetX87Top();
auto data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
if constexpr (width == 64) {
// Store 64-bit float directly
//Store 64-bit float directly
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, data, 8, 1);
} else if constexpr (width == 32) {
// Convert to 32-bit float and store
//Convert to 32-bit float and store
auto result = _Float_FToF(4, 8, data);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, result, 4, 1);
} else if constexpr (width == 80) {
// Convert to 80-bit float
//Convert to 80-bit float
auto result = _F80CVTTo(data, 8);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, result, 10, 1);
}
@@ -232,16 +245,19 @@ void OpDispatchBuilder::FSTF64(OpcodeArgs) {
}
}
template void OpDispatchBuilder::FSTF64<32>(OpcodeArgs);
template void OpDispatchBuilder::FSTF64<64>(OpcodeArgs);
template void OpDispatchBuilder::FSTF64<80>(OpcodeArgs);
template
void OpDispatchBuilder::FSTF64<32>(OpcodeArgs);
template
void OpDispatchBuilder::FSTF64<64>(OpcodeArgs);
template
void OpDispatchBuilder::FSTF64<80>(OpcodeArgs);
template<bool Truncate>
void OpDispatchBuilder::FISTF64(OpcodeArgs) {
auto Size = GetSrcSize(Op);
auto orig_top = GetX87Top();
OrderedNode* data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode *data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
if constexpr (Truncate) {
data = _Float_ToGPR_ZS(Size == 4 ? 4 : 8, 8, data);
} else {
@@ -258,16 +274,18 @@ void OpDispatchBuilder::FISTF64(OpcodeArgs) {
}
}
template void OpDispatchBuilder::FISTF64<false>(OpcodeArgs);
template void OpDispatchBuilder::FISTF64<true>(OpcodeArgs);
template
void OpDispatchBuilder::FISTF64<false>(OpcodeArgs);
template
void OpDispatchBuilder::FISTF64<true>(OpcodeArgs);
template<size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
template <size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
void OpDispatchBuilder::FADDF64(OpcodeArgs) {
auto top = GetX87Top();
OrderedNode* StackLocation = top;
OrderedNode *StackLocation = top;
OrderedNode* arg {};
OrderedNode* b {};
OrderedNode *arg{};
OrderedNode *b{};
auto mask = _Constant(7);
@@ -275,7 +293,7 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs) {
// Memory arg
if constexpr (Integer) {
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (width == 16) {
if(width == 16) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
@@ -309,20 +327,26 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs) {
_StoreContextIndexed(result, StackLocation, 8, MMBaseOffset(), 16, FPRClass);
}
template void OpDispatchBuilder::FADDF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FADDF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template
void OpDispatchBuilder::FADDF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FADDF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template void OpDispatchBuilder::FADDF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FADDF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FADDF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FADDF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template<size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
void OpDispatchBuilder::FMULF64(OpcodeArgs) {
auto top = GetX87Top();
OrderedNode* StackLocation = top;
OrderedNode* arg {};
OrderedNode* b {};
OrderedNode *StackLocation = top;
OrderedNode *arg{};
OrderedNode *b{};
auto mask = _Constant(7);
@@ -330,7 +354,7 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs) {
// Memory arg
if constexpr (Integer) {
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (width == 16) {
if(width == 16) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
@@ -367,28 +391,34 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs) {
_StoreContextIndexed(result, StackLocation, 8, MMBaseOffset(), 16, FPRClass);
}
template void OpDispatchBuilder::FMULF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FMULF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template
void OpDispatchBuilder::FMULF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FMULF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template void OpDispatchBuilder::FMULF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FMULF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FMULF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FMULF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
auto top = GetX87Top();
OrderedNode* StackLocation = top;
OrderedNode* arg {};
OrderedNode* b {};
OrderedNode *StackLocation = top;
OrderedNode *arg{};
OrderedNode *b{};
auto mask = _Constant(7);
if (!Op->Src[0].IsNone()) {
if (!Op->Src[0].IsNone()) {
// Memory arg
if constexpr (Integer) {
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (width == 16) {
if(width == 16) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
@@ -411,10 +441,11 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode* result {};
OrderedNode *result{};
if constexpr (reverse) {
result = _VFDiv(8, 8, b, a);
} else {
}
else {
result = _VFDiv(8, 8, a, b);
}
@@ -430,38 +461,50 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
_StoreContextIndexed(result, StackLocation, 8, MMBaseOffset(), 16, FPRClass);
}
template void OpDispatchBuilder::FDIVF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FDIVF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FDIVF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
auto top = GetX87Top();
OrderedNode* StackLocation = top;
OrderedNode* arg {};
OrderedNode* b {};
OrderedNode *StackLocation = top;
OrderedNode *arg{};
OrderedNode *b{};
auto mask = _Constant(7);
if (!Op->Src[0].IsNone()) {
if (!Op->Src[0].IsNone()) {
// Memory arg
if constexpr (Integer) {
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (width == 16) {
if(width == 16) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
@@ -484,10 +527,11 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode* result {};
OrderedNode *result{};
if constexpr (reverse) {
result = _VFSub(8, 8, b, a);
} else {
}
else {
result = _VFSub(8, 8, a, b);
}
@@ -504,23 +548,35 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
_StoreContextIndexed(result, StackLocation, 8, MMBaseOffset(), 16, FPRClass);
}
template void OpDispatchBuilder::FSUBF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template void OpDispatchBuilder::FSUBF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
template
void OpDispatchBuilder::FSUBF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
void OpDispatchBuilder::FCHSF64(OpcodeArgs) {
auto top = GetX87Top();
@@ -543,18 +599,26 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
auto low = _Constant(0);
OrderedNode* data = _VCastFromGPR(8, 8, low);
OrderedNode *data = _VCastFromGPR(8, 8, low);
// We are going to clobber NZCV, make sure it's in a GPR first.
GetNZCV();
OrderedNode *Res = _FCmp(8, a, data,
(1 << FCMP_FLAG_EQ) |
(1 << FCMP_FLAG_LT) |
(1 << FCMP_FLAG_UNORDERED));
// Now we do our comparison.
_FCmp(8, a, data);
PossiblySetNZCVBits = ~0;
ConvertNZCVToX87();
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
}
// TODO: This should obey rounding mode
//TODO: This should obey rounding mode
void OpDispatchBuilder::FRNDINTF64(OpcodeArgs) {
auto top = GetX87Top();
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
@@ -591,14 +655,14 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
auto top = GetX87Top();
auto mask = _Constant(7);
OrderedNode* arg {};
OrderedNode* b {};
OrderedNode *arg{};
OrderedNode *b{};
if (!Op->Src[0].IsNone()) {
// Memory arg
if constexpr (Integer) {
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
if (width == 16) {
if(width == 16) {
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
}
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
@@ -617,21 +681,36 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
if constexpr (whichflags == FCOMIFlags::FLAGS_X87) {
// We are going to clobber NZCV, make sure it's in a GPR first.
GetNZCV();
OrderedNode *Res = _FCmp(8, a, b,
(1 << FCMP_FLAG_EQ) |
(1 << FCMP_FLAG_LT) |
(1 << FCMP_FLAG_UNORDERED));
_FCmp(8, a, b);
PossiblySetNZCVBits = ~0;
ConvertNZCVToX87();
} else {
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
if constexpr (whichflags == FCOMIFlags::FLAGS_X87) {
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
}
else {
// Invalidate deferred flags early
// OF, SF, AF, PF all undefined
InvalidateDeferredFlags();
_FCmp(8, a, b);
PossiblySetNZCVBits = ~0;
ConvertNZCVToSSE();
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
// PF is stored inverted, so invert from the host flag.
// TODO: This could perhaps be optimized?
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
}
if constexpr (poptwice) {
@@ -642,7 +721,8 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
// Set the new top now
top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
SetX87Top(top);
} else if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
}
else if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
// if we are popping then we must first mark this location as empty
SetX87ValidTag(top, false);
// Set the new top now
@@ -651,17 +731,24 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
}
}
template void OpDispatchBuilder::FCOMIF64<32, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
template
void OpDispatchBuilder::FCOMIF64<32, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
template void OpDispatchBuilder::FCOMIF64<64, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
template
void OpDispatchBuilder::FCOMIF64<64, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
template void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
template void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>(OpcodeArgs);
template void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, true>(OpcodeArgs);
template
void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
template
void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>(OpcodeArgs);
template
void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, true>(OpcodeArgs);
template void OpDispatchBuilder::FCOMIF64<16, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
template
void OpDispatchBuilder::FCOMIF64<16, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
template void OpDispatchBuilder::FCOMIF64<32, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
template
void OpDispatchBuilder::FCOMIF64<32, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
void OpDispatchBuilder::FSQRTF64(OpcodeArgs) {
@@ -680,9 +767,12 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
auto top = GetX87Top();
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
DeriveOp(result, IROp, _F64SIN(a));
auto result = _F64SIN(a);
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F64SIN || IROp == IR::OP_F64COS) {
if constexpr (IROp == IR::OP_F64SIN ||
IROp == IR::OP_F64COS) {
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
}
@@ -691,9 +781,12 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
}
template void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64F2XM1>(OpcodeArgs);
template void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64SIN>(OpcodeArgs);
template void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64COS>(OpcodeArgs);
template
void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64F2XM1>(OpcodeArgs);
template
void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64SIN>(OpcodeArgs);
template
void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64COS>(OpcodeArgs);
template<FEXCore::IR::IROps IROp>
@@ -701,15 +794,18 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
auto top = GetX87Top();
auto mask = _Constant(7);
OrderedNode* st1 = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
OrderedNode *st1 = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
st1 = _LoadContextIndexed(st1, 8, MMBaseOffset(), 16, FPRClass);
DeriveOp(result, IROp, _F64ATAN(a, st1));
auto result = _F64ATAN(a, st1);
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F64FPREM || IROp == IR::OP_F64FPREM1) {
// TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
if constexpr (IROp == IR::OP_F64FPREM ||
IROp == IR::OP_F64FPREM1) {
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
}
@@ -717,9 +813,12 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
}
template void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM1>(OpcodeArgs);
template void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM>(OpcodeArgs);
template void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64SCALE>(OpcodeArgs);
template
void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM1>(OpcodeArgs);
template
void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM>(OpcodeArgs);
template
void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64SCALE>(OpcodeArgs);
void OpDispatchBuilder::X87SinCosF64(OpcodeArgs) {
auto orig_top = GetX87Top();
@@ -749,8 +848,8 @@ void OpDispatchBuilder::X87FYL2XF64(OpcodeArgs) {
auto top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, orig_top, _Constant(1)), _Constant(7));
SetX87Top(top);
OrderedNode* st0 = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode* st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode *st0 = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode *st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
if (Plus1) {
auto one = _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000));
@@ -791,7 +890,7 @@ void OpDispatchBuilder::X87ATANF64(OpcodeArgs) {
SetX87Top(top);
auto a = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode* st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode *st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
auto result = _F64ATAN(st1, a);
@@ -799,7 +898,7 @@ void OpDispatchBuilder::X87ATANF64(OpcodeArgs) {
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
}
// This function converts to F80 on save for compatibility
//This function converts to F80 on save for compatibility
void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
// 14 bytes for 16bit
@@ -821,16 +920,18 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
// 4 bytes : data pointer offset
// 4 bytes : data pointer selector
const auto Size = GetDstSize(Op);
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Dest);
OrderedNode* Top = GetX87Top();
auto Size = GetDstSize(Op);
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
Mem = AppendSegmentOffset(Mem, Op->Flags);
OrderedNode *Top = GetX87Top();
{
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMem(GPRClass, Size, Mem, FCW, Size);
}
{
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
_StoreMem(GPRClass, Size, MemLocation, ReconstructFSW(), Size);
}
@@ -838,35 +939,35 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
{
// FTW
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
_StoreMem(GPRClass, Size, MemLocation, GetX87FTW(), Size);
}
{
// Instruction Offset
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 3));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 3));
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
}
{
// Instruction CS selector (+ Opcode)
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 4));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 4));
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
}
{
// Data pointer offset
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 5));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 5));
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
}
{
// Data pointer selector
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 6));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 6));
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
}
OrderedNode* ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
OrderedNode *ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
auto OneConst = _Constant(1);
auto SevenConst = _Constant(7);
@@ -894,16 +995,17 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
FNINIT(Op);
}
// This function converts from F80 on load for compatibility
//This function converts from F80 on load for compatibility
void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
const auto Size = GetSrcSize(Op);
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Src[0]);
auto Size = GetSrcSize(Op);
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
Mem = AppendSegmentOffset(Mem, Op->Flags);
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
OrderedNode* roundingMode = NewFCW;
//ignore the rounding precision, we're always 64-bit in F64.
//extract rounding mode
OrderedNode *roundingMode = NewFCW;
auto roundShift = _Constant(10);
auto roundMask = _Constant(3);
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
@@ -912,17 +1014,17 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
auto Top = ReconstructX87StateFromFSW(NewFSW);
{
// FTW
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
}
OrderedNode* ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
OrderedNode *ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
auto OneConst = _Constant(1);
auto SevenConst = _Constant(7);
@@ -930,14 +1032,14 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
auto low = _Constant(~0ULL);
auto high = _Constant(0xFFFF);
OrderedNode* Mask = _VCastFromGPR(16, 8, low);
OrderedNode *Mask = _VCastFromGPR(16, 8, low);
Mask = _VInsGPR(16, 8, 1, Mask, high);
for (int i = 0; i < 7; ++i) {
OrderedNode* Reg = _LoadMem(FPRClass, 16, ST0Location, 1);
OrderedNode *Reg = _LoadMem(FPRClass, 16, ST0Location, 1);
// Mask off the top bits
Reg = _VAnd(16, 16, Reg, Mask);
// Convert to double precision
//Convert to double precision
Reg = _F80CVT(8, Reg);
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
@@ -950,20 +1052,20 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
// Lower 64bits [63:0]
// upper 16 bits [79:64]
OrderedNode* Reg = _LoadMem(FPRClass, 8, ST0Location, 1);
OrderedNode *Reg = _LoadMem(FPRClass, 8, ST0Location, 1);
ST0Location = _Add(OpSize::i64Bit, ST0Location, _Constant(8));
OrderedNode* RegHigh = _LoadMem(FPRClass, 2, ST0Location, 1);
OrderedNode *RegHigh = _LoadMem(FPRClass, 2, ST0Location, 1);
Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh);
Reg = _F80CVT(8, Reg); // Convert to double precision
Reg = _F80CVT(8, Reg); //Convert to double precision
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
}
// FXAM needs change
//FXAM needs change
void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
auto top = GetX87Top();
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
OrderedNode* Result = _VExtractToGPR(8, 8, a, 0);
OrderedNode *Result = _VExtractToGPR(8, 8, a, 0);
// Extract the sign bit
Result = _Bfe(OpSize::i64Bit, 1, 63, Result);
@@ -976,7 +1078,9 @@ void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
auto OneConst = _Constant(1);
// In the case of top being invalid then C3:C2:C0 is 0b101
auto C3 = _Select(FEXCore::IR::COND_EQ, TopValid, OneConst, ZeroConst, OneConst);
auto C3 = _Select(FEXCore::IR::COND_EQ,
TopValid, OneConst,
ZeroConst, OneConst);
auto C2 = TopValid;
auto C0 = C3; // Mirror C3 until something other than zero is supported
@@ -986,4 +1090,4 @@ void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
}
} // namespace FEXCore::IR
}
@@ -0,0 +1,41 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <unistd.h>
#include <signal.h>
namespace FEXCore {
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
SetHostSignalHandler(Signal, Func, Required);
FrontendRegisterHostSignalHandler(Signal, Func, Required);
}
void SignalDelegator::HandleSignal(int Signal, void *Info, void *UContext) {
// Let the host take first stab at handling the signal
auto Thread = GetTLSThread();
HostSignalHandler &Handler = HostHandlers[Signal];
if (!Thread) {
LogMan::Msg::AFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
}
else {
for (auto &Handler : Handler.Handlers) {
if (Handler(Thread, Signal, Info, UContext)) {
// If the host handler handled the fault then we can continue now
return;
}
}
if (Handler.FrontendHandler &&
Handler.FrontendHandler(Thread, Signal, Info, UContext)) {
return;
}
// Now let the frontend handle the signal
// It's clearly a guest signal and this ends up being an OS specific issue
HandleGuestSignal(Thread, Signal, Info, UContext);
}
}
}
@@ -0,0 +1,84 @@
// SPDX-License-Identifier: MIT
#ifndef NDEBUG
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/Utils/LogManager.h>
#include <tuple>
namespace FEXCore::X86Tables::X86InstDebugInfo {
void InstallDebugInfo() {
const std::tuple<uint8_t, uint8_t, Flags> BaseOpTable[] = {
{0x50, 8, {FLAGS_MEM_ACCESS}},
{0x58, 8, {FLAGS_MEM_ACCESS}},
{0x68, 1, {FLAGS_MEM_ACCESS}},
{0x6A, 1, {FLAGS_MEM_ACCESS}},
{0xAA, 4, {FLAGS_MEM_ACCESS}},
{0xC8, 1, {FLAGS_MEM_ACCESS}},
{0xCC, 2, {FLAGS_DEBUG}},
{0xD7, 1, {FLAGS_MEM_ACCESS}},
{0xF1, 1, {FLAGS_DEBUG}},
{0xF4, 1, {FLAGS_DEBUG}},
};
const std::tuple<uint8_t, uint8_t, Flags> TwoByteOpTable[] = {
{0x0B, 1, {FLAGS_DEBUG}},
{0x19, 7, {FLAGS_DEBUG}},
{0x28, 2, {FLAGS_MEM_ALIGN_16}},
{0x31, 1, {FLAGS_DEBUG}},
{0xA2, 1, {FLAGS_DEBUG}},
{0xA3, 1, {FLAGS_MEM_ACCESS}},
{0xAB, 1, {FLAGS_MEM_ACCESS}},
{0xB3, 1, {FLAGS_MEM_ACCESS}},
{0xBB, 1, {FLAGS_MEM_ACCESS}},
{0xFF, 1, {FLAGS_DEBUG}},
};
const std::tuple<uint8_t, uint8_t, Flags> PrimaryGroupOpTable[] = {
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 2, {FLAGS_DIVIDE}},
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 6), 2, {FLAGS_DIVIDE}},
#undef OPD
};
const std::tuple<uint16_t, uint8_t, Flags> SecondaryExtensionOpTable[] = {
#define PF_NONE 0
#define PF_F3 1
#define PF_66 2
#define PF_F2 3
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
{OPD(TYPE_GROUP_15, PF_NONE, 2), 1, {FLAGS_DEBUG}},
{OPD(TYPE_GROUP_15, PF_NONE, 3), 1, {FLAGS_DEBUG}},
#undef PF_F3
#undef PF_66
#undef PF_F2
#undef OPD
};
auto GenerateDebugTable = [](auto& FinalTable, auto& LocalTable) {
for (auto Op : LocalTable) {
auto OpNum = std::get<0>(Op);
auto DebugInfo = std::get<2>(Op);
for (uint8_t i = 0; i < std::get<1>(Op); ++i) {
memcpy(&FinalTable[OpNum+i].DebugInfo, &DebugInfo, sizeof(X86InstDebugInfo::Flags));
}
}
};
GenerateDebugTable(BaseOps, BaseOpTable);
GenerateDebugTable(SecondBaseOps, TwoByteOpTable);
GenerateDebugTable(PrimaryInstGroupOps, PrimaryGroupOpTable);
GenerateDebugTable(SecondInstGroupOps, SecondaryExtensionOpTable);
}
}
#endif
@@ -39,15 +39,15 @@ X86GeneratedCode::X86GeneratedCode() {
// Falling back to this generated code segment still allows a backtrace to work, just might not show
// the symbol as VDSO since there is no ELF to parse.
constexpr std::array<uint8_t, 9> sigreturn_32_code = {
0x58, // pop eax
0x58, // pop eax
0xb8, 0x77, 0x00, 0x00, 0x00, // mov eax, 0x77
0xcd, 0x80, // int 0x80
0x90, // nop
0xcd, 0x80, // int 0x80
0x90, // nop
};
constexpr std::array<uint8_t, 7> rt_sigreturn_32_code = {
0xb8, 0xad, 0x00, 0x00, 0x00, // mov eax, 0xad
0xcd, 0x80, // int 0x80
0xcd, 0x80, // int 0x80
};
CallbackReturn = reinterpret_cast<uint64_t>(CodePtr);
@@ -84,9 +84,10 @@ void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
// We need to have the sigret handler in the lower 32bits of memory space
// Scan top down and try to allocate a location
for (size_t Location = 0xFFFF'E000; Location != 0x0; Location -= 0x1000) {
void* Ptr = ::mmap(reinterpret_cast<void*>(Location), Size, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
void *Ptr = ::mmap(reinterpret_cast<void*>(Location), Size, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
if (Ptr != MAP_FAILED && reinterpret_cast<uintptr_t>(Ptr) >= LOCATION_MAX) {
if (Ptr != MAP_FAILED &&
reinterpret_cast<uintptr_t>(Ptr) >= LOCATION_MAX) {
// Failed to map in the lower 32bits
// Try again
// Can happen in the case that host kernel ignores MAP_FIXED_NOREPLACE
@@ -107,4 +108,5 @@ void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
#endif
}
} // namespace FEXCore
}
+5 -5
View File
@@ -16,12 +16,12 @@ public:
X86GeneratedCode();
~X86GeneratedCode();
uint64_t CallbackReturn {};
uint64_t sigreturn_32 {};
uint64_t rt_sigreturn_32 {};
uint64_t CallbackReturn{};
uint64_t sigreturn_32{};
uint64_t rt_sigreturn_32{};
private:
void* CodePtr {};
void *CodePtr{};
void* AllocateGuestCodeSpace(size_t Size);
};
} // namespace FEXCore
}
Loaded 100 of 765 files, more files were not shown because too many files have changed in this diff. Show more