Compare commits

..
2 Commits
Author SHA1 Message Date
Ryan Houdek faed139c34 Docs: Update for release FEX-2312.1 2023-12-11 13:48:42 -08:00
Ryan Houdek 80927bf0e1 FEXLoader: Temporarily disable CLONE_CLEAR_SIGHAND
This flag breaks FEX heavily for now.
glibc 2.38 started using this flag as an optimization for posix_spawn.
It will fall back to a "non-optimized" implementation if the clone
syscall returns EINVAL. For now do this while we investigate a more
proper implementation.

Should be backported to 2312.1.
2023-12-11 13:47:40 -08:00
953 changed files with 141650 additions and 385052 deletions

No files matched your search

-109
View File
@@ -1,109 +0,0 @@
Language: Cpp
BasedOnStyle: WebKit
AccessModifierOffset: -2
AlignAfterOpenBracket: Align
AlignArrayOfStructures: None
AlignConsecutiveAssignments: None
AlignConsecutiveBitFields: Consecutive
AlignConsecutiveDeclarations: None
AlignConsecutiveMacros: None
AlignEscapedNewlines: Left
AlignOperands: Align
AlignTrailingComments: true
AllowAllParametersOfDeclarationOnNextLine: false
AllowShortCaseLabelsOnASingleLine: true
AllowShortEnumsOnASingleLine: true
AllowShortFunctionsOnASingleLine: Empty
AllowShortIfStatementsOnASingleLine: WithoutElse
AllowShortLambdasOnASingleLine: Inline
AlwaysBreakAfterDefinitionReturnType: None
AlwaysBreakAfterReturnType: None
AlwaysBreakBeforeMultilineStrings: false
AlwaysBreakTemplateDeclarations: true
AttributeMacros:
- JEMALLOC_NOTHROW
- FEX_ALIGNED
- FEX_ANNOTATE
- FEX_DEFAULT_VISIBILITY
- FEX_NAKED
- FEX_PACKED
- FEXCORE_PRESERVE_ALL_ATTR
- GLIBC_ALIAS_FUNCTION
BinPackArguments: true
BinPackParameters: true
BitFieldColonSpacing: Both
BreakAfterAttributes: Always # clang 16 required
BreakBeforeBraces: Attach
BreakBeforeBinaryOperators: None
BreakBeforeInlineASMColon: OnlyMultiline # clang 16 required
BreakBeforeTernaryOperators: false
BreakConstructorInitializers: BeforeComma
BreakInheritanceList: BeforeColon
ColumnLimit: 140
CompactNamespaces: false
ConstructorInitializerIndentWidth: 2
ContinuationIndentWidth: 2
Cpp11BracedListStyle: true
DerivePointerAlignment: false
EmptyLineAfterAccessModifier: Leave
EmptyLineBeforeAccessModifier: Leave
ExperimentalAutoDetectBinPacking: false
FixNamespaceComments: true
IncludeBlocks: Preserve
IndentAccessModifiers: false
IndentCaseBlocks: false
IndentCaseLabels: false
IndentExternBlock: AfterExternBlock
IndentGotoLabels: false
IndentPPDirectives: None
IndentRequires: false
IndentWidth: 2
InsertBraces: true
KeepEmptyLinesAtTheStartOfBlocks: true
LambdaBodyIndentation: OuterScope
LineEnding: LF # clang 16 required
MaxEmptyLinesToKeep: 2
NamespaceIndentation: Inner
QualifierAlignment: Left
PackConstructorInitializers: Never
PenaltyBreakAssignment: 2
PenaltyBreakBeforeFirstCallParameter: 2
PenaltyBreakOpenParenthesis: 2
PenaltyBreakString: 10
PenaltyBreakTemplateDeclaration: 8
PenaltyExcessCharacter: 2
PenaltyReturnTypeOnItsOwnLine: 16
PointerAlignment: Left
RemoveBracesLLVM: false
ReferenceAlignment: Left
ReflowComments: true
RequiresClausePosition: WithPreceding
SeparateDefinitionBlocks: Leave
SortIncludes: Never
SpaceAfterCStyleCast: false
SpaceAfterLogicalNot: false
SpaceAfterTemplateKeyword: false
SpaceAroundPointerQualifiers: Default
SpaceBeforeAssignmentOperators: true
SpaceBeforeCaseColon: false
SpaceBeforeCpp11BracedList: true
SpaceBeforeInheritanceColon: true
SpaceBeforeParens: Custom
SpaceBeforeParensOptions:
AfterControlStatements: true
AfterFunctionDeclarationName: false
AfterFunctionDefinitionName: false
AfterOverloadedOperator: false
AfterRequiresInClause: true
BeforeNonEmptyParentheses: false
SpaceBeforeRangeBasedForLoopColon: true
SpaceBeforeSquareBrackets: false
SpaceInEmptyBlock: false
SpaceInEmptyParentheses: false
SpacesBeforeTrailingComments: 1
SpacesInAngles: Leave
SpacesInCStyleCastParentheses: false
SpacesInConditionalStatement: false
SpacesInParentheses: false
Standard: c++20
UseTab: Never
-12
View File
@@ -1,12 +0,0 @@
# This file is used to ignore files and directories from clang-format
# Ignore all files in the External directory
External/*
# SoftFloat-3e code doesn't belong to us
FEXCore/Source/Common/SoftFloat-3e/*
Source/Common/cpp-optparse/*
# Files with human-indented tables for readability - don't mess with these
FEXCore/Source/Interface/Core/X86Tables/*
-15
View File
@@ -1,15 +0,0 @@
# Since version 2.23 (released in August 2019), git-blame has a feature
# to ignore or bypass certain commits.
#
# This file contains a list of commits that are not likely what you
# are looking for in a blame, such as mass reformatting or renaming.
# You can set this file as a default ignore file for blame by running
# the following command.
#
# $ git config blame.ignoreRevsFile .git-blame-ignore-revs
# Whole tree reformat PR#3571
2b4ec88daebd35fefb5bf5c73d7fc2b4155771ed
# Second reformat to find fixed point PR#3577
905aa935f5ce344a48ef4d5edab3c31efa8d793e
@@ -37,6 +37,7 @@ If applicable, add screenshots and video to help explain your problem.
**Additional context**
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
- Is this a Vulkan game: [Yes/No/Unknown]
- If Yes, What is your Vulkan driver:
+15 -1
View File
@@ -13,6 +13,8 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
FEX_ENABLEAVX: 1
jobs:
build_plus_test:
@@ -76,6 +78,18 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -236,7 +250,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
+15 -1
View File
@@ -20,6 +20,8 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_FORCE32BITALLOCATOR: 1
FEX_ENABLEAVX: 1
jobs:
glibc_fault_test:
@@ -83,6 +85,18 @@ jobs:
shell: bash
run: cmake --build . --config $BUILD_TYPE --target install
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: gcc target tests 64
working-directory: ${{runner.workspace}}/build
shell: bash
@@ -170,7 +184,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Remove old SHM regions
if: ${{ always() }}
+2 -1
View File
@@ -13,6 +13,7 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_ENABLEAVX: 1
jobs:
hostrunner_tests:
@@ -89,7 +90,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+2 -1
View File
@@ -13,6 +13,7 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_ENABLEAVX: 1
jobs:
instcountci_tests:
@@ -120,7 +121,7 @@ jobs:
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+1
View File
@@ -10,6 +10,7 @@ on:
env:
BUILD_TYPE: Debug
FEX_ENABLEAVX: 1
jobs:
mingw_build:
-76
View File
@@ -1,76 +0,0 @@
# Inspired by LLVM's pr-code-format.yml at
# https://github.com/llvm/llvm-project/blob/main/.github/workflows/pr-code-format.yml
name: "Check code formatting"
on:
pull_request:
branches:
- main
jobs:
code_formatter:
runs-on: [self-hosted, X64]
if: github.repository == 'FEX-Emu/FEX'
steps:
- name: Fetch FEX sources
uses: actions/checkout@v4
with:
ref: ${{ github.event.pull_request.head.sha }}
- name: Checkout through merge base
uses: rmacklin/fetch-through-merge-base@v0
timeout-minutes: 3
with:
base_ref: ${{ github.event.pull_request.base.ref }}
head_ref: ${{ github.event.pull_request.head.sha }}
deepen_length: 500
- name: Get changed files
id: changed-files
uses: tj-actions/changed-files@v39
with:
separator: ","
skip_initial_fetch: true
- name: "Listed files"
env:
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
run: |
echo "Formatting files:"
echo "$CHANGED_FILES"
- name: Check for correct clang-format version
run: clang-format --version | grep -qF '16.0.6'
- name: Check git-clang-format-16 exists
run: which git-clang-format-16
- name: Setup Python env
uses: actions/setup-python@v4
with:
python-version: '3.11'
cache: 'pip'
cache-dependency-path: './External/code-format-helper/requirements_formatting.txt'
- name: Install python dependencies
run: pip install -r ./External/code-format-helper/requirements_formatting.txt
- name: Run code formatter
env:
CLANG_FORMAT_PATH: 'git-clang-format-16'
GITHUB_PR_NUMBER: ${{ github.event.pull_request.number }}
START_REV: ${{ github.event.pull_request.base.sha }}
END_REV: ${{ github.event.pull_request.head.sha }}
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
# TODO(pmatos): Once we adopt v18, we should be able
# to take advantage of the new --diff_from_common_commit option
# explicitly in code-format-helper.py and not have to diff starting at
# the merge base.
run: |
python ./External/code-format-helper/code-format-helper.py \
--repo "FEX-emu/FEX" \
--issue-number $GITHUB_PR_NUMBER \
--start-rev $(git merge-base $START_REV $END_REV) \
--end-rev $END_REV \
--changed-files "$CHANGED_FILES"
+14 -1
View File
@@ -13,6 +13,7 @@ env:
BUILD_TYPE: Release
CC: clang
CXX: clang++
FEX_ENABLEAVX: 1
jobs:
vixl_simulator:
@@ -99,13 +100,25 @@ jobs:
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM128bit.log || true
- name: IR Tests
working-directory: ${{runner.workspace}}/build
shell: bash
# Execute the unit tests
run: cmake --build . --config $BUILD_TYPE --target ir_tests
- name: IR Test Results move
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
- name: Truncate test results
if: ${{ always() }}
shell: bash
working-directory: ${{runner.workspace}}/build
# Cap out the log files at 20M in case something crash spins and dumps fault text
# ASM tests get quite close to 10MB
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
- name: Set runner name
if: ${{ always() }}
+6 -6
View File
@@ -15,19 +15,19 @@
path = External/tiny-json
url = https://github.com/Sonicadvance1/tiny-json.git
[submodule "External/xbyak"]
shallow = true
shallow = true
path = External/xbyak
url = https://github.com/herumi/xbyak.git
url = https://github.com/FEX-Emu/xbyak.git
[submodule "External/fex-posixtest-bins"]
shallow = true
shallow = true
path = External/fex-posixtest-bins
url = https://github.com/FEX-Emu/fex-posixtest-bins.git
[submodule "External/fex-gvisor-tests-bins"]
shallow = true
shallow = true
path = External/fex-gvisor-tests-bins
url = https://github.com/FEX-Emu/fex-gvisor-tests-bins.git
[submodule "External/fex-gcc-target-tests-bins"]
shallow = true
shallow = true
path = External/fex-gcc-target-tests-bins
url = https://github.com/FEX-Emu/fex-gcc-target-tests-bins.git
[submodule "External/jemalloc"]
@@ -41,7 +41,7 @@
url = https://github.com/FEX-Emu/drm-headers.git
[submodule "External/xxhash"]
path = External/xxhash
url = https://github.com/Cyan4973/xxHash.git
url = https://github.com/FEX-Emu/xxHash.git
[submodule "External/Catch2"]
path = External/Catch2
url = https://github.com/catchorg/Catch2.git
+69 -36
View File
@@ -9,6 +9,7 @@ option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
option(BUILD_THUNKS "Build thunks" FALSE)
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
option(ENABLE_IWYU "Enables include what you use program" FALSE)
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
@@ -25,6 +26,7 @@ option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
@@ -43,14 +45,6 @@ if (NOT CONTAINS_MINGW EQUAL -1)
set (ENABLE_JEMALLOC FALSE)
endif()
if (NOT MINGW_BUILD)
message (STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
set (CLANG_MINIMUM_VERSION 12.0)
if (CMAKE_CXX_COMPILER_VERSION VERSION_LESS ${CLANG_MINIMUM_VERSION})
message (FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
endif()
endif()
if (ENABLE_FEXCORE_PROFILER)
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
@@ -118,12 +112,6 @@ else()
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
if (NOT ENABLE_X86_HOST_DEBUG)
message(FATAL_ERROR
" FEX-Emu doesn't support compiling for x86-64 hosts!"
" This is /only/ a supported configuration for FEX CI and nothing else!")
endif()
set(_M_X86_64 1)
add_definitions(-D_M_X86_64=1)
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
@@ -134,14 +122,6 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
add_definitions(-D_M_ARM_64=1)
endif()
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
set(_M_ARM_64EC 1)
add_definitions(-D_M_ARM_64EC=1)
# Required as FEX is not allowed to lock the CRT heap lock during compilation or callbacks
set(ENABLE_JEMALLOC TRUE)
endif()
if (ENABLE_CCACHE)
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
@@ -178,6 +158,18 @@ if (NOT ENABLE_OFFLINE_TELEMETRY)
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
endif()
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
add_definitions(-DTERMUX_BUILD=1)
set(TERMUX_BUILD 1)
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
set(ENABLE_JEMALLOC FALSE)
# Termux builds can't rely on X11 packages
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
set(BUILD_FEXCONFIG FALSE)
endif()
if (ENABLE_ASAN)
add_definitions(-DENABLE_ASAN=1)
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
@@ -240,15 +232,15 @@ endif()
find_package(PkgConfig REQUIRED)
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
set(XXHASH_BUNDLED_MODE TRUE)
set(XXHASH_BUILD_XXHSUM FALSE)
set(BUILD_SHARED_LIBS OFF)
add_subdirectory(External/xxhash/cmake_unofficial/)
add_subdirectory(External/xxhash/)
include_directories(External/xxhash/)
add_definitions(-Wno-trigraphs)
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
if (BUILD_TESTS)
option(CATCH_BUILD_STATIC_LIBRARY "" ON)
set(CATCH_BUILD_STATIC_LIBRARY ON)
add_subdirectory(External/Catch2/)
# Pull in catch_discover_tests definition
@@ -358,6 +350,57 @@ if (ENABLE_IWYU)
endif()
endif()
if (ENABLE_CLANG_FORMAT)
find_program(CLANG_TIDY_EXE "clang-tidy")
if (NOT CLANG_TIDY_EXE)
message(FATAL_ERROR "Couldn't find clang-tidy")
endif()
set(CLANG_TIDY_FLAGS
"-checks=*"
"-fuchsia*"
"-bugprone-macro-parentheses"
"-clang-analyzer-core.*"
"-cppcoreguidelines-pro-type-*"
"-cppcoreguidelines-pro-bounds-array-to-pointer-decay"
"-cppcoreguidelines-pro-bounds-pointer-arithmetic"
"-cppcoreguidelines-avoid-c-arrays"
"-cppcoreguidelines-avoid-magic-numbers"
"-cppcoreguidelines-pro-bounds-constant-array-index"
"-cppcoreguidelines-no-malloc"
"-cppcoreguidelines-special-member-functions"
"-cppcoreguidelines-owning-memory"
"-cppcoreguidelines-macro-usage"
"-cppcoreguidelines-avoid-goto"
"-google-readability-function-size"
"-google-readability-namespace-comments"
"-google-readability-braces-around-statements"
"-google-build-using-namespace"
"-hicpp-*"
"-llvm-namespace-comment"
"-llvm-include-order" # Messes up with case sensitivity
"-llvmlibc-*"
"-misc-unused-parameters"
"-modernize-loop-convert"
"-modernize-use-auto"
"-modernize-avoid-c-arrays"
"-modernize-use-nodiscard"
"readability-*"
"-readability-function-size"
"-readability-implicit-bool-conversion"
"-readability-braces-around-statements"
"-readability-else-after-return"
"-readability-magic-numbers"
"-readability-named-parameter"
"-readability-uppercase-literal-suffix"
"-cert-err34-c"
"-cert-err58-cpp"
"-bugprone-exception-escape"
)
string(REPLACE ";" "," CLANG_TIDY_FLAGS "${CLANG_TIDY_FLAGS}")
set(CMAKE_CXX_CLANG_TIDY ${CLANG_TIDY_EXE} "${CLANG_TIDY_FLAGS}")
endif()
add_compile_options(-Wall)
configure_file(
@@ -368,19 +411,9 @@ if (BUILD_TESTS)
include(CTest)
enable_testing()
message(STATUS "Unit tests are enabled")
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
if (TEST_JOB_COUNT)
message(STATUS "Running tests with ${TEST_JOB_COUNT} jobs")
endif()
if (CMAKE_VERSION VERSION_LESS "3.29")
execute_process(COMMAND "nproc" OUTPUT_STRIP_TRAILING_WHITESPACE OUTPUT_VARIABLE TEST_JOB_COUNT)
endif()
set(TEST_JOB_FLAG "-j${TEST_JOB_COUNT}")
endif()
add_subdirectory(FEXHeaderUtils/)
add_subdirectory(CodeEmitter/)
add_subdirectory(FEXCore/)
# Binfmt_misc files must be installed prior to Source/ installs
-2
View File
@@ -1,2 +0,0 @@
add_library(CodeEmitter INTERFACE)
target_include_directories(CodeEmitter INTERFACE .)
File diff suppressed because it is too large. Load diff
-106
View File
@@ -1,106 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <cstddef>
#include <cstdint>
#include <cstring>
namespace ARMEmitter {
class Buffer {
public:
Buffer() {
SetBuffer(nullptr, 0);
}
Buffer(uint8_t* Base, uint64_t BaseSize) {
SetBuffer(Base, BaseSize);
}
void SetBuffer(uint8_t* Base, uint64_t BaseSize) {
BufferBase = Base;
CurrentOffset = BufferBase;
Size = BaseSize;
}
void dc8(uint8_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc16(uint16_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc32(uint32_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc64(uint64_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void EmitString(const char* String) {
const auto StringLength = strlen(String);
memcpy(CurrentOffset, String, StringLength);
CurrentOffset += StringLength;
}
void Align() {
// Align the buffer to instruction size
auto CurrentAlignment = reinterpret_cast<uint64_t>(CurrentOffset) & 0b11;
if (!CurrentAlignment) {
return;
}
CurrentOffset += 4 - CurrentAlignment;
}
template<typename T>
T GetCursorAddress() const {
return reinterpret_cast<T>(CurrentOffset);
}
static void ClearICache(void* Begin, std::size_t Length) {
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
}
size_t GetCursorOffset() const {
return static_cast<size_t>(CurrentOffset - BufferBase);
}
uint8_t* GetBufferBase() const {
return BufferBase;
}
void CursorIncrement(size_t Size) {
CurrentOffset += Size;
}
void SetCursorOffset(size_t Offset) {
CurrentOffset = BufferBase + Offset;
}
uint64_t GetBufferSize() const {
return Size;
}
template<typename T>
size_t GetCursorOffsetFromAddress(const T* Address) const {
return static_cast<size_t>(reinterpret_cast<const uint8_t*>(Address) - BufferBase);
}
protected:
void ResetBuffer() {
CurrentOffset = BufferBase;
}
uint8_t* BufferBase;
uint8_t* CurrentOffset;
uint64_t Size;
};
} // namespace ARMEmitter
-831
View File
@@ -1,831 +0,0 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <CodeEmitter/Buffer.h>
#include <CodeEmitter/Registers.h>
#include <array>
#include <cstdint>
#include <utility>
#include <type_traits>
/*
* Welcome to FEX-Emu's custom AArch64 emitter.
* This was written specifically to avoid the performance cost of the vixl emitter.
*
* There are some specific design constraints in this design to target a couple features:
* - High performance
* - Low CPU cache performance hit
* - Significantly reduced code footprint
* - Low number of branches
*
* These requirements are mostly achieved by removing a bunch of developer conveniences
* that vixl provides. The developer needs to take a lot of care to not shoot themselves in the foot.
*
* Misc design decisions:
* - Registers are encoded as basic uint32_t enums.
* - Converting between different registers is zero-cost.
* - Passing around as arguments are as cheap as registers
* - Contrast to vixl where every register requires living on the stack.
* - Registers can get encoded in to instructions with a simple `BFM` instruction.
*
* - Instructions are very simply emitted, allowing direct inlining most of the time.
* - These are simple enough that multiple back-to-back instructions get optimized to 128-bit load-store operations.
* - Contrast to vixl where pretty much no instruction emitter gets inlined.
*
* - Instruction emitters are /mostly/ unsized. Most instructions take a size argument first, which gets encoded
* directly in to the instruction.
* - Contrast to vixl where the register arguments are how the instructions determine operating size.
* - Size argument allows FEX to use `CSEL` to select a size at runtime, instead of branching.
* - Some instructions are explicitly sized based on register type. Read comments in the respective `inl` files to
* see why.
* Some scalar/vector operations are an example of this.
*
* - Almost zero helper functions.
* - Primary exception to this rule is load-store operations. These will use a helper to make
* it easier to select the correct load-store instruction. Mostly because these are a nightmare selecting
* the right instruction.
*/
namespace ARMEmitter {
/*
* This `Size` enum is used for most ALU operations.
* These follow the AArch64 encoding style in most cases.
*/
enum class Size : uint32_t {
i32Bit = 0,
i64Bit,
};
// This allows us to get the `Size` enum in bits.
[[nodiscard]]
constexpr size_t RegSizeInBits(Size size) {
return size_t {32} << FEXCore::ToUnderlying(size);
}
/* This `SubRegSize` enum is used for most ASIMD operations.
* These follow the AArch64 encoding style in most cases.
*/
enum class SubRegSize : uint32_t {
i8Bit = 0b00,
i16Bit = 0b01,
i32Bit = 0b10,
i64Bit = 0b11,
i128Bit = 0b100,
};
// This allows us to get the `SubRegSize` in bits.
[[nodiscard]]
constexpr size_t SubRegSizeInBits(SubRegSize size) {
return size_t {8} << FEXCore::ToUnderlying(size);
}
/* This `ScalarRegSize` enum is used for most scalar float
* operations.
*
* This is specifically duplicated from `SubRegSize` to have strongly
* typed functions.
*
* `ScalarRegSize` specifically doesn't have `i128Bit` because scalar operations
* can't operate at 128-bit.
*/
enum class ScalarRegSize : uint32_t {
i8Bit = 0b00,
i16Bit = 0b01,
i32Bit = 0b10,
i64Bit = 0b11,
};
// This allows us to get the `ScalarRegSize` in bits.
[[nodiscard]]
constexpr size_t ScalarRegSizeInBits(ScalarRegSize size) {
return size_t {8} << FEXCore::ToUnderlying(size);
}
/* This `VectorRegSizePair` union allows us to have an overlapping type
* to select a scalar operation or a vector depending on which operation
* we pass in.
* Useful in FEX's vector operations that behave as scalar or vector
* depending on various factors. But since the operation will have the sa,e
* element size, we want to choose the operation more easily
*/
union VectorRegSizePair {
ScalarRegSize Scalar;
SubRegSize Vector;
};
// This allows us to create a `VectorRegSizePair` union.
[[nodiscard]]
constexpr VectorRegSizePair ToVectorSizePair(SubRegSize size) {
return VectorRegSizePair {.Vector = size};
}
[[nodiscard]]
constexpr VectorRegSizePair ToVectorSizePair(ScalarRegSize size) {
return VectorRegSizePair {.Scalar = size};
}
// This `ShiftType` enum is used for ALU shift-register encoded instructions.
enum class ShiftType : uint32_t {
LSL = 0,
LSR,
ASR,
ROR,
};
// This `ExtendedType` enum is used for ALU extended-register encoded instructions.
enum class ExtendedType : uint32_t {
UXTB = 0b000,
UXTH = 0b001,
UXTW = 0b010,
UXTX = 0b011,
SXTB = 0b100,
SXTH = 0b101,
SXTW = 0b110,
SXTX = 0b111,
LSL_32 = UXTW,
LSL_64 = UXTX,
};
// This `Condition` enum is used for various conditional instructions.
enum class Condition : uint32_t {
// Meaning: Int - Float
CC_EQ = 0, // Equal - Equal
CC_NE, // Not Eq - Not Eq or unordered
CC_CS, // Carry set - Greater than, equal, or unordered
CC_CC, // Carry clear - Less than
CC_MI, // Minus/Negative - Less than
CC_PL, // Plus, positive or zero - GT, equal, or unordered
CC_VS, // Overflow - Unordered
CC_VC, // No Overflow - Ordered
CC_HI, // Unsigned higher - GT, or unordered
CC_LS, // Unsigned lower or same - LT or EQ
CC_GE, // Signed GT or EQ - GT or EQ
CC_LT, // Signed LT - LT or Unordered
CC_GT, // Signed GT - GT
CC_LE, // Signed LT or EQ - LT, EQ, or Unordered
CC_AL, // Always - Always
CC_NV, // Always - Always
// Aliases
CC_HS = CC_CS,
CC_LO = CC_CC,
};
/*
* This `StatusFlags` enum is used for conditional compare encoded instructions.
* These directly encode to the `nzcv` flags.
*/
enum class StatusFlags : uint32_t {
None = 0,
Flag_V = 0b0001,
Flag_C = 0b0010,
Flag_Z = 0b0100,
Flag_N = 0b1000,
Flag_NZCV = Flag_N | Flag_Z | Flag_C | Flag_V,
};
/*
* This `IndexType` enum is used for load-store instructions.
* Not all load-store instructions use this, so the user needs to be careful.
*/
enum class IndexType {
POST,
OFFSET,
PRE,
UNPRIVILEGED,
};
// Used with adr and scalar + vector load/store variants to denote
// a modifier operation.
enum class SVEModType : uint8_t {
MOD_UXTW,
MOD_SXTW,
MOD_LSL,
MOD_NONE,
};
/* This `SVEMemOperand` class is used for the helper SVE load-store instructions.
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
*/
class SVEMemOperand final {
public:
enum class Type {
ScalarPlusScalar,
ScalarPlusImm,
ScalarPlusVector,
VectorPlusImm,
};
SVEMemOperand(XRegister rn, XRegister rm = XReg::zr)
: rn {rn}
, MemType {Type::ScalarPlusScalar}
, MetaType {.ScalarScalarType {
.rm = rm,
}} {}
SVEMemOperand(XRegister rn, int32_t imm = 0)
: rn {rn}
, MemType {Type::ScalarPlusImm}
, MetaType {.ScalarImmType {
.Imm = imm,
}} {}
SVEMemOperand(XRegister rn, ZRegister zm, SVEModType mod = SVEModType::MOD_NONE, uint8_t scale = 0)
: rn {rn}
, MemType {Type::ScalarPlusVector}
, MetaType {.ScalarVectorType {
.zm = zm,
.mod = mod,
.scale = scale,
}} {}
SVEMemOperand(ZRegister zn, uint32_t imm)
: rn {Register {zn.Idx()}}
, MemType {Type::VectorPlusImm}
, MetaType {.VectorImmType {
.Imm = imm,
}} {}
[[nodiscard]]
bool IsScalarPlusScalar() const {
return MemType == Type::ScalarPlusScalar;
}
[[nodiscard]]
bool IsScalarPlusImm() const {
return MemType == Type::ScalarPlusImm;
}
[[nodiscard]]
bool IsScalarPlusVector() const {
return MemType == Type::ScalarPlusVector;
}
[[nodiscard]]
bool IsVectorPlusImm() const {
return MemType == Type::VectorPlusImm;
}
union Data {
struct {
Register rm;
} ScalarScalarType;
struct {
int32_t Imm;
} ScalarImmType;
struct {
ZRegister zm;
SVEModType mod;
uint8_t scale;
} ScalarVectorType;
struct {
// rn will be a ZRegister
uint32_t Imm;
} VectorImmType;
};
Register rn;
Type MemType;
Data MetaType;
};
/* This `ExtendedMemOperand` class is used for the helper load-store instructions.
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
*/
class ExtendedMemOperand final {
public:
ExtendedMemOperand(XRegister rn, XRegister rm = XReg::zr, ExtendedType Option = ExtendedType::LSL_64, uint32_t Shift = 0)
: rn {rn}
, MetaType {.ExtendedType {
.Header = {.MemType = TYPE_EXTENDED},
.rm = rm,
.Option = Option,
.Shift = Shift,
}} {}
ExtendedMemOperand(XRegister rn, IndexType Index = IndexType::OFFSET, int32_t Imm = 0)
: rn {rn}
, MetaType {.ImmType {
.Header = {.MemType = TYPE_IMM},
.Index = Index,
.Imm = Imm,
}} {}
Register rn;
enum Type {
TYPE_EXTENDED,
TYPE_IMM,
};
struct HeaderStruct {
Type MemType;
};
union {
HeaderStruct Header;
struct {
HeaderStruct Header;
Register rm;
ExtendedType Option;
uint32_t Shift;
} ExtendedType;
struct {
HeaderStruct Header;
IndexType Index;
int32_t Imm;
} ImmType;
} MetaType;
};
template<uint32_t op0, uint32_t op1, uint32_t CRn, uint32_t CRm, uint32_t op2>
constexpr uint32_t GenSystemReg() {
return op0 << 19 | op1 << 16 | CRn << 12 | CRm << 8 | op2 << 5;
};
// This `SystemRegister` enum is used for the mrs/msr instructions.
enum class SystemRegister : uint32_t {
CTR_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b001>(),
DCZID_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b111>(),
TPIDR_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b010>(),
RNDR = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b000>(),
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>(),
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>(),
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>(),
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>(),
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>(),
};
template<uint32_t op1, uint32_t CRm, uint32_t op2>
constexpr uint32_t GenDCReg() {
return op1 << 16 | CRm << 8 | op2 << 5;
};
// This `DataCacheOperation` enum is used for the dc instruction.
enum class DataCacheOperation : uint32_t {
IVAC = GenDCReg<0b000, 0b0110, 0b001>(),
ISW = GenDCReg<0b000, 0b0110, 0b010>(),
CSW = GenDCReg<0b000, 0b1010, 0b010>(),
CISW = GenDCReg<0b000, 0b1110, 0b010>(),
ZVA = GenDCReg<0b011, 0b0100, 0b001>(),
CVAC = GenDCReg<0b011, 0b1010, 0b001>(),
CVAU = GenDCReg<0b011, 0b1011, 0b001>(),
CIVAC = GenDCReg<0b011, 0b1110, 0b001>(),
// MTE2
IGVAC = GenDCReg<0b000, 0b0110, 0b011>(),
IGSW = GenDCReg<0b000, 0b0110, 0b100>(),
IGDVAC = GenDCReg<0b000, 0b0110, 0b101>(),
IGDSW = GenDCReg<0b000, 0b0110, 0b110>(),
CGSW = GenDCReg<0b000, 0b1010, 0b100>(),
CGDSW = GenDCReg<0b000, 0b1010, 0b110>(),
CIGSW = GenDCReg<0b000, 0b1110, 0b100>(),
CIGDSW = GenDCReg<0b000, 0b1110, 0b110>(),
// MTE
GVA = GenDCReg<0b011, 0b0100, 0b011>(),
GZVA = GenDCReg<0b011, 0b0100, 0b100>(),
CGVAC = GenDCReg<0b011, 0b1010, 0b011>(),
CGDVAC = GenDCReg<0b011, 0b1010, 0b101>(),
CGVAP = GenDCReg<0b011, 0b1100, 0b011>(),
CGDVAP = GenDCReg<0b011, 0b1100, 0b101>(),
CGVADP = GenDCReg<0b011, 0b1101, 0b011>(),
CGDVADP = GenDCReg<0b011, 0b1101, 0b101>(),
CIGVAC = GenDCReg<0b011, 0b1110, 0b011>(),
CIGDVAC = GenDCReg<0b011, 0b1110, 0b101>(),
// DPB
CVAP = GenDCReg<0b011, 0b1100, 0b001>(),
// DPB2
CVADP = GenDCReg<0b011, 0b1101, 0b001>(),
};
template<uint32_t CRm, uint32_t op2>
constexpr uint32_t GenHintBarrierReg() {
return CRm << 8 | op2 << 5;
}
// This `HintRegister` enum is used for the hint instruction.
enum class HintRegister : uint32_t {
NOP = GenHintBarrierReg<0b0000, 0b000>(),
YIELD = GenHintBarrierReg<0b0000, 0b001>(),
WFE = GenHintBarrierReg<0b0000, 0b010>(),
WFI = GenHintBarrierReg<0b0000, 0b011>(),
SEV = GenHintBarrierReg<0b0000, 0b100>(),
SEVL = GenHintBarrierReg<0b0000, 0b101>(),
DGH = GenHintBarrierReg<0b0000, 0b110>(),
CSDB = GenHintBarrierReg<0b0010, 0b100>(),
};
// This `BarrierRegister` enum is used for the various barrier instructions.
enum class BarrierRegister : uint32_t {
CLREX = GenHintBarrierReg<0b0000, 0b010>(),
TCOMMIT = GenHintBarrierReg<0b0000, 0b011>(),
DSB = GenHintBarrierReg<0b0000, 0b100>(),
DMB = GenHintBarrierReg<0b0000, 0b101>(),
ISB = GenHintBarrierReg<0b0000, 0b110>(),
SB = GenHintBarrierReg<0b0000, 0b111>(),
};
// This `BarrierScope` enum is used for the dsb/dmb instructions.
enum class BarrierScope : uint32_t {
// Outer shareable
OSHLD = 0b0001,
OSHST = 0b0010,
OSH = 0b0011,
// Non shareable
NSHLD = 0b0101,
NSHST = 0b0110,
NSH = 0b0111,
// Inner shareable
ISHLD = 0b1001,
ISHST = 0b1010,
ISH = 0b1011,
// Full System visibility
LD = 0b1101,
ST = 0b1110,
SY = 0b1111,
};
// This `Prefetch` enum is used for prefetch instructions.
enum class Prefetch : uint32_t {
// Prefetch for load
PLDL1KEEP = 0b00000,
PLDL1STRM = 0b00001,
PLDL2KEEP = 0b00010,
PLDL2STRM = 0b00011,
PLDL3KEEP = 0b00100,
PLDL3STRM = 0b00101,
// Preload instructions
PLIL1KEEP = 0b01000,
PLIL1STRM = 0b01001,
PLIL2KEEP = 0b01010,
PLIL2STRM = 0b01011,
PLIL3KEEP = 0b01100,
PLIL3STRM = 0b01101,
// Preload for store
PSTL1KEEP = 0b10000,
PSTL1STRM = 0b10001,
PSTL2KEEP = 0b10010,
PSTL2STRM = 0b10011,
PSTL3KEEP = 0b10100,
PSTL3STRM = 0b10101,
};
// This `PredicatePattern` enun is used for some SVE instructions.
enum class PredicatePattern : uint32_t {
SVE_POW2 = 0b00000,
SVE_VL1 = 0b00001,
SVE_VL2 = 0b00010,
SVE_VL3 = 0b00011,
SVE_VL4 = 0b00100,
SVE_VL5 = 0b00101,
SVE_VL6 = 0b00110,
SVE_VL7 = 0b00111,
SVE_VL8 = 0b01000,
SVE_VL16 = 0b01001,
SVE_VL32 = 0b01010,
SVE_VL64 = 0b01011,
SVE_VL128 = 0b01100,
SVE_VL256 = 0b01101,
SVE_MUL4 = 0b11101,
SVE_MUL3 = 0b11110,
SVE_ALL = 0b11111,
};
// Used with SVE FP immediate arithmetic instructions
enum class SVEFAddSubImm : uint32_t {
_0_5,
_1_0,
};
enum class SVEFMulImm : uint32_t {
_0_5,
_2_0,
};
enum class SVEFMaxMinImm : uint32_t {
_0_0,
_1_0,
};
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is logically `below` an instruction that uses it.
* Which means that a branch would jump backwards.
*/
struct BackwardLabel {
uint8_t* Location {};
};
/* This `SingleUseForwardLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is logically `above` an instruction that uses it.
* Which means that a branch would jump forwards.
*
* The `ForwardLabel` struct can be bound to multiple instructions, so it needs a vector for each bind instruction type.
*/
struct SingleUseForwardLabel {
enum class InstType {
UNKNOWN,
ADR,
ADRP,
B,
BC,
TEST_BRANCH,
RELATIVE_LOAD,
LONG_ADDRESS_GEN,
};
uint8_t* Location {};
InstType Type = InstType::UNKNOWN;
};
struct ForwardLabel {
fextl::vector<SingleUseForwardLabel> Insts {};
};
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is in either direction of an instruction that uses it.
* Which means a branch could jump backwards or forwards depending on situation.
*/
struct BiDirectionalLabel {
BackwardLabel Backward;
ForwardLabel Forward;
};
static inline void AddLocationToLabel(SingleUseForwardLabel* Label, SingleUseForwardLabel&& Location) {
LOGMAN_THROW_A_FMT(Label->Type == SingleUseForwardLabel::InstType::UNKNOWN, "Trying to bind a SingleUseForwardLabel to multiple "
"locations. Use ForwardLabel instead.");
*Label = std::move(Location);
}
static inline void AddLocationToLabel(ForwardLabel* Label, SingleUseForwardLabel&& Location) {
Label->Insts.emplace_back(std::move(Location));
}
// Some FCMA ASIMD instructions support a rotation argument.
enum class Rotation : uint32_t {
ROTATE_0 = 0b00,
ROTATE_90 = 0b01,
ROTATE_180 = 0b10,
ROTATE_270 = 0b11,
};
// Concept for contraining some instructions to accept only an XRegister or WRegister.
// Particularly for operations that differ encodings depending on which one is used.
template<typename T>
concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegister>;
// Whether or not a given set of vector registers are sequential
// in increasing order as far as the register file is concerned (modulo its size)
//
// For example, a set of registers like:
//
// v1, v2, v3 and
// v31, v0, v1
//
// would both be considered sequential sequences, and some instructions in particular
// limit register lists to these kind of sequences.
//
template<typename T, typename... Args>
constexpr bool AreVectorsSequential(T first, const Args&... args) {
// Ensure we always have a pair of registers to compare against.
static_assert(sizeof...(args) >= 1, "Number of arguments must be greater than 1");
const auto fn = [](auto& lhs, const auto& rhs) {
const auto result = ((lhs.Idx() + 1) % 32) == rhs.Idx();
lhs = rhs;
return result;
};
return (fn(first, args) && ...);
}
// This is an emitter that is designed around the smallest code bloat as possible.
// Eschewing most developer convenience in order to keep code as small as possible.
// Choices:
// - Size of ops passed as an argument rather than template to let the compiler use csel instead of branching.
// - Registers are unsized so they can be passed in a GPR and not need conversion operations
class Emitter : public ARMEmitter::Buffer {
public:
Emitter() = default;
Emitter(uint8_t* Base, uint64_t BaseSize)
: Buffer(Base, BaseSize) {}
// Bind a backward label to an address.
// Address that is bound is the current emitter location.
void Bind(BackwardLabel* Label) {
LOGMAN_THROW_AA_FMT(Label->Location == nullptr, "Trying to bind a label twice");
Label->Location = GetCursorAddress<uint8_t*>();
}
void Bind(const SingleUseForwardLabel* Label) {
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
// Patch up the instructions
switch (Label->Type) {
case SingleUseForwardLabel::InstType::ADR: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::ADRP: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::B: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= Offset;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::TEST_BRANCH: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::BC:
case SingleUseForwardLabel::InstType::RELATIVE_LOAD: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN: {
uint32_t* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
auto OriginalOffset = GetCursorOffset();
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
SetCursorOffset(InstOffset);
// We encoded the destination register in to the first instruction space.
// Read it back.
ARMEmitter::Register DestReg(Instructions[0]);
if (IsADRRange(ImmInstTwo)) {
// If within ADR range from the second instruction, then we can emit NOP+ADR
nop();
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
} else if (IsADRPRange(ImmInstOne)) {
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
// First check if we are in non-offset range for second instruction.
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
// We can emit nop + adrp
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
} else {
// Not aligned, need adrp + add
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
}
} else {
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
FEX_UNREACHABLE;
}
SetCursorOffset(OriginalOffset);
break;
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
}
// Bind a forward label to a location.
// This walks all the instructions in the label's vector.
// Then backpatching all instructions that have used the label.
template<bool WarnAboutEmpty = false>
void Bind(ForwardLabel* Label) {
if constexpr (WarnAboutEmpty) {
LOGMAN_THROW_A_FMT(Label->Insts.empty() == false, "Binding forward label that didn't have any instructions using it");
}
for (auto& Inst : Label->Insts) {
Bind(&Inst);
}
}
// Bind a bidirectional location to a location.
// Binds both forwards and backwards depending on how the label was used.
void Bind(BiDirectionalLabel* Label) {
if (!Label->Backward.Location) {
Bind(&Label->Backward);
}
Bind<false>(&Label->Forward);
}
#include <CodeEmitter/VixlUtils.inl>
public:
// TODO: Implement SME when it matters.
#include <CodeEmitter/ALUOps.inl>
#include <CodeEmitter/BranchOps.inl>
#include <CodeEmitter/LoadstoreOps.inl>
#include <CodeEmitter/SystemOps.inl>
#include <CodeEmitter/ScalarOps.inl>
#include <CodeEmitter/ASIMDOps.inl>
#include <CodeEmitter/SVEOps.inl>
private:
template<typename T>
uint32_t Encode_ra(T Reg) const {
return Reg.Idx() << 10;
}
uint32_t Encode_ra(uint32_t Reg) const {
return Reg << 10;
}
template<typename T>
uint32_t Encode_rt2(T Reg) const {
return Reg.Idx() << 10;
}
template<>
uint32_t Encode_rt2(uint32_t Reg) const {
return Reg << 10;
}
template<typename T>
uint32_t Encode_rm(T Reg) const {
return Reg.Idx() << 16;
}
uint32_t Encode_rm(uint32_t Reg) const {
return Reg << 16;
}
template<typename T>
uint32_t Encode_rs(T Reg) const {
return Reg.Idx() << 16;
}
uint32_t Encode_rs(uint32_t Reg) const {
return Reg << 16;
}
template<typename T>
uint32_t Encode_rn(T Reg) const {
return Reg.Idx() << 5;
}
uint32_t Encode_rn(uint32_t Reg) const {
return Reg << 5;
}
template<typename T>
uint32_t Encode_rd(T Reg) const {
return Reg.Idx();
}
uint32_t Encode_rd(uint32_t Reg) const {
return Reg;
}
template<typename T>
uint32_t Encode_rt(T Reg) const {
return Reg.Idx();
}
template<>
uint32_t Encode_rt(Prefetch Reg) const {
return FEXCore::ToUnderlying(Reg);
}
uint32_t Encode_rt(uint32_t Reg) const {
return Reg;
}
template<typename T>
uint32_t Encode_pd(T Reg) const {
return FEXCore::ToUnderlying(Reg);
}
};
} // namespace ARMEmitter
File diff suppressed because it is too large. Load diff
-311
View File
@@ -1,311 +0,0 @@
// Collection of utilities from vixl.
// Following is the vixl license.
// Copyright 2015, VIXL authors
// All rights reserved.
//
// Redistribution and use in source and binary forms, with or without
// modification, are permitted provided that the following conditions are met:
//
// * Redistributions of source code must retain the above copyright notice,
// this list of conditions and the following disclaimer.
// * Redistributions in binary form must reproduce the above copyright notice,
// this list of conditions and the following disclaimer in the documentation
// and/or other materials provided with the distribution.
// * Neither the name of ARM Limited nor the names of its contributors may be
// used to endorse or promote products derived from this software without
// specific prior written permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS CONTRIBUTORS "AS IS" AND
// ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
// WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
// DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE
// FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
// DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
// SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
// CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
// OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
// Test if a given value can be encoded in the immediate field of a logical
// instruction.
// If it can be encoded, the function returns true, and values pointed to by n,
// imm_s and imm_r are updated with immediates encoded in the format required
// by the corresponding fields in the logical instruction.
// If it can not be encoded, the function returns false, and the values pointed
// to by n, imm_s and imm_r are undefined.
static bool IsImmLogical(uint64_t value,
unsigned width,
unsigned* n,
unsigned* imm_s,
unsigned* imm_r) {
[[maybe_unused]] constexpr auto kBRegSize = 8;
[[maybe_unused]] constexpr auto kHRegSize = 16;
[[maybe_unused]] constexpr auto kSRegSize = 32;
[[maybe_unused]] constexpr auto kDRegSize = 64;
constexpr auto kWRegSize = 32;
constexpr auto kXRegSize = 64;
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) ||
(width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
bool negate = false;
// Logical immediates are encoded using parameters n, imm_s and imm_r using
// the following table:
//
// N imms immr size S R
// 1 ssssss rrrrrr 64 UInt(ssssss) UInt(rrrrrr)
// 0 0sssss xrrrrr 32 UInt(sssss) UInt(rrrrr)
// 0 10ssss xxrrrr 16 UInt(ssss) UInt(rrrr)
// 0 110sss xxxrrr 8 UInt(sss) UInt(rrr)
// 0 1110ss xxxxrr 4 UInt(ss) UInt(rr)
// 0 11110s xxxxxr 2 UInt(s) UInt(r)
// (s bits must not be all set)
//
// A pattern is constructed of size bits, where the least significant S+1 bits
// are set. The pattern is rotated right by R, and repeated across a 32 or
// 64-bit value, depending on destination register width.
//
// Put another way: the basic format of a logical immediate is a single
// contiguous stretch of 1 bits, repeated across the whole word at intervals
// given by a power of 2. To identify them quickly, we first locate the
// lowest stretch of 1 bits, then the next 1 bit above that; that combination
// is different for every logical immediate, so it gives us all the
// information we need to identify the only logical immediate that our input
// could be, and then we simply check if that's the value we actually have.
//
// (The rotation parameter does give the possibility of the stretch of 1 bits
// going 'round the end' of the word. To deal with that, we observe that in
// any situation where that happens the bitwise NOT of the value is also a
// valid logical immediate. So we simply invert the input whenever its low bit
// is set, and then we know that the rotated case can't arise.)
if (value & 1) {
// If the low bit is 1, negate the value, and set a flag to remember that we
// did (so that we can adjust the return values appropriately).
negate = true;
value = ~value;
}
if (width <= kWRegSize) {
// To handle 8/16/32-bit logical immediates, the very easiest thing is to repeat
// the input value to fill a 64-bit word. The correct encoding of that as a
// logical immediate will also be the correct encoding of the value.
// Avoid making the assumption that the most-significant 56/48/32 bits are zero by
// shifting the value left and duplicating it.
for (unsigned bits = width; bits <= kWRegSize; bits *= 2) {
value <<= bits;
uint64_t mask = (UINT64_C(1) << bits) - 1;
value |= ((value >> bits) & mask);
}
}
// The basic analysis idea: imagine our input word looks like this.
//
// 0011111000111110001111100011111000111110001111100011111000111110
// c b a
// |<--d-->|
//
// We find the lowest set bit (as an actual power-of-2 value, not its index)
// and call it a. Then we add a to our original number, which wipes out the
// bottommost stretch of set bits and replaces it with a 1 carried into the
// next zero bit. Then we look for the new lowest set bit, which is in
// position b, and subtract it, so now our number is just like the original
// but with the lowest stretch of set bits completely gone. Now we find the
// lowest set bit again, which is position c in the diagram above. Then we'll
// measure the distance d between bit positions a and c (using CLZ), and that
// tells us that the only valid logical immediate that could possibly be equal
// to this number is the one in which a stretch of bits running from a to just
// below b is replicated every d bits.
uint64_t a = LowestSetBit(value);
uint64_t value_plus_a = value + a;
uint64_t b = LowestSetBit(value_plus_a);
uint64_t value_plus_a_minus_b = value_plus_a - b;
uint64_t c = LowestSetBit(value_plus_a_minus_b);
int d, clz_a, out_n;
uint64_t mask;
if (c != 0) {
// The general case, in which there is more than one stretch of set bits.
// Compute the repeat distance d, and set up a bitmask covering the basic
// unit of repetition (i.e. a word with the bottom d bits set). Also, in all
// of these cases the N bit of the output will be zero.
clz_a = CountLeadingZeros(a, kXRegSize);
int clz_c = CountLeadingZeros(c, kXRegSize);
d = clz_a - clz_c;
mask = ((UINT64_C(1) << d) - 1);
out_n = 0;
} else {
// Handle degenerate cases.
//
// If any of those 'find lowest set bit' operations didn't find a set bit at
// all, then the word will have been zero thereafter, so in particular the
// last lowest_set_bit operation will have returned zero. So we can test for
// all the special case conditions in one go by seeing if c is zero.
if (a == 0) {
// The input was zero (or all 1 bits, which will come to here too after we
// inverted it at the start of the function), for which we just return
// false.
return false;
} else {
// Otherwise, if c was zero but a was not, then there's just one stretch
// of set bits in our word, meaning that we have the trivial case of
// d == 64 and only one 'repetition'. Set up all the same variables as in
// the general case above, and set the N bit in the output.
clz_a = CountLeadingZeros(a, kXRegSize);
d = 64;
mask = ~UINT64_C(0);
out_n = 1;
}
}
// If the repeat period d is not a power of two, it can't be encoded.
if (!IsPowerOf2(d)) {
return false;
}
if (((b - a) & ~mask) != 0) {
// If the bit stretch (b - a) does not fit within the mask derived from the
// repeat period, then fail.
return false;
}
// The only possible option is b - a repeated every d bits. Now we're going to
// actually construct the valid logical immediate derived from that
// specification, and see if it equals our original input.
//
// To repeat a value every d bits, we multiply it by a number of the form
// (1 + 2^d + 2^(2d) + ...), i.e. 0x0001000100010001 or similar. These can
// be derived using a table lookup on CLZ(d).
static const uint64_t multipliers[] = {
0x0000000000000001UL,
0x0000000100000001UL,
0x0001000100010001UL,
0x0101010101010101UL,
0x1111111111111111UL,
0x5555555555555555UL,
};
uint64_t multiplier = multipliers[CountLeadingZeros(d, kXRegSize) - 57];
uint64_t candidate = (b - a) * multiplier;
if (value != candidate) {
// The candidate pattern doesn't match our input value, so fail.
return false;
}
// We have a match! This is a valid logical immediate, so now we have to
// construct the bits and pieces of the instruction encoding that generates
// it.
// Count the set bits in our basic stretch. The special case of clz(0) == -1
// makes the answer come out right for stretches that reach the very top of
// the word (e.g. numbers like 0xffffc00000000000).
int clz_b = (b == 0) ? -1 : CountLeadingZeros(b, kXRegSize);
int s = clz_a - clz_b;
// Decide how many bits to rotate right by, to put the low bit of that basic
// stretch in position a.
int r;
if (negate) {
// If we inverted the input right at the start of this function, here's
// where we compensate: the number of set bits becomes the number of clear
// bits, and the rotation count is based on position b rather than position
// a (since b is the location of the 'lowest' 1 bit after inversion).
s = d - s;
r = (clz_b + 1) & (d - 1);
} else {
r = (clz_a + 1) & (d - 1);
}
// Now we're done, except for having to encode the S output in such a way that
// it gives both the number of set bits and the length of the repeated
// segment. The s field is encoded like this:
//
// imms size S
// ssssss 64 UInt(ssssss)
// 0sssss 32 UInt(sssss)
// 10ssss 16 UInt(ssss)
// 110sss 8 UInt(sss)
// 1110ss 4 UInt(ss)
// 11110s 2 UInt(s)
//
// So we 'or' (2 * -d) with our computed s to form imms.
if ((n != NULL) || (imm_s != NULL) || (imm_r != NULL)) {
*n = out_n;
*imm_s = ((2 * -d) | (s - 1)) & 0x3f;
*imm_r = r;
}
return true;
}
private:
template <typename V>
static inline bool IsPowerOf2(V value) {
return (value != 0) && ((value & (value - 1)) == 0);
}
// Some compilers dislike negating unsigned integers,
// so we provide an equivalent.
template <typename T>
static inline T UnsignedNegate(T value) {
static_assert(std::is_unsigned<T>::value);
return ~value + 1;
}
static inline uint64_t LowestSetBit(uint64_t value) {
return value & UnsignedNegate(value);
}
template <typename V>
static inline int CountLeadingZeros(V value, int width = (sizeof(V) * 8)) {
#if COMPILER_HAS_BUILTIN_CLZ
if (width == 32) {
return (value == 0) ? 32 : __builtin_clz(static_cast<unsigned>(value));
} else if (width == 64) {
return (value == 0) ? 64 : __builtin_clzll(value);
}
#endif
return CountLeadingZerosFallBack(value, width);
}
static inline int CountLeadingZerosFallBack(uint64_t value, int width) {
LOGMAN_THROW_A_FMT(IsPowerOf2(width) && (width <= 64), "Invalid width");
if (value == 0) {
return width;
}
int count = 0;
value = value << (64 - width);
if ((value & UINT64_C(0xffffffff00000000)) == 0) {
count += 32;
value = value << 32;
}
if ((value & UINT64_C(0xffff000000000000)) == 0) {
count += 16;
value = value << 16;
}
if ((value & UINT64_C(0xff00000000000000)) == 0) {
count += 8;
value = value << 8;
}
if ((value & UINT64_C(0xf000000000000000)) == 0) {
count += 4;
value = value << 4;
}
if ((value & UINT64_C(0xc000000000000000)) == 0) {
count += 2;
value = value << 2;
}
if ((value & UINT64_C(0x8000000000000000)) == 0) {
count += 1;
}
count += (value == 0);
return count;
}
public:
+2 -3
View File
@@ -1,6 +1,5 @@
{
"Comment": "Bypasses libGL's glX and instead sends GLX requests directly via xcb",
"ThunksDB": {
"GL": 0
"Config": {
"AdditionalArguments": "--no-sandbox"
}
}
+9
View File
@@ -2,6 +2,9 @@
"DB": {
"GL": {
"Library" : "libGL-guest.so",
"Depends": [
"X11"
],
"Overlay": [
"@PREFIX_LIB@/libGL.so",
"@PREFIX_LIB@/libGL.so.1",
@@ -30,10 +33,16 @@
},
"Vulkan": {
"Library": "libvulkan-guest.so",
"Depends": [
"xcb"
],
"Overlay": [
"@PREFIX_LIB@/libvulkan.so",
"@PREFIX_LIB@/libvulkan.so.1",
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
],
"Comment": [
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
]
},
"xcb": {
+4 -5
View File
@@ -3,12 +3,11 @@ FROM ubuntu:20.04 as builder
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
clang-10 llvm-10 nasm ninja-build pkg-config \
libcap-dev libglfw3-dev libepoxy-dev python3-dev libsdl2-dev \
python3 linux-headers-generic \
git
clang-10 llvm-10 nasm ninja-build \
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
python3 linux-headers-generic
RUN git clone --recurse-submodules https://github.com/FEX-Emu/FEX.git
COPY . /opt/FEX
CMD [ "mkdir /opt/FEX/build" ]
+1 -1
-336
View File
@@ -1,336 +0,0 @@
#!/usr/bin/env python3
#
# ====- code-format-helper, runs code formatters from the ci or in a hook --*- python -*--==#
#
# Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
# See https://llvm.org/LICENSE.txt for license information.
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#
# ==--------------------------------------------------------------------------------------==#
import argparse
import os
import subprocess
import sys
from typing import List, Optional
"""
This script is run by GitHub actions to ensure that the code in PR's conform to
the coding style of LLVM. It can also be installed as a pre-commit git hook to
check the coding style before submitting it. The canonical source of this script
is in the LLVM source tree under llvm/utils/git.
For C/C++ code it uses clang-format.
You can learn more about the LLVM coding style on llvm.org:
https://llvm.org/docs/CodingStandards.html
You can install this script as a git hook by symlinking it to the .git/hooks
directory:
ln -s $(pwd)/llvm/utils/git/code-format-helper.py .git/hooks/pre-commit
You can control the exact path to clang-format with the following
environment variable: $CLANG_FORMAT_PATH.
"""
class FormatArgs:
start_rev: str = None
end_rev: str = None
repo: str = None
changed_files: List[str] = []
token: str = None
verbose: bool = True
issue_number: int = 0
write_comment_to_file: str = None
def __init__(self, args: argparse.Namespace = None) -> None:
if not args is None:
self.start_rev = args.start_rev
self.end_rev = args.end_rev
self.repo = args.repo
self.token = args.token
self.changed_files = args.changed_files
self.issue_number = args.issue_number
self.write_comment_to_file = args.write_comment_to_file
class FormatHelper:
COMMENT_TAG = "<!--CODE FORMAT COMMENT: {fmt}-->"
name: str
friendly_name: str
comment: dict = None
@property
def comment_tag(self) -> str:
return self.COMMENT_TAG.replace("fmt", self.name)
@property
def instructions(self) -> str:
raise NotImplementedError()
def has_tool(self) -> bool:
raise NotImplementedError()
def format_run(self, changed_files: List[str], args: FormatArgs) -> Optional[str]:
raise NotImplementedError()
def pr_comment_text_for_diff(self, diff: str) -> str:
return f"""
:warning: {self.friendly_name}, {self.name} found issues in your code. :warning:
<details>
<summary>
You can test this locally with the following command:
</summary>
``````````bash
{self.instructions}
``````````
</details>
<details>
<summary>
View the diff from {self.name} here.
</summary>
``````````diff
{diff}
``````````
</details>
"""
# TODO: any type should be replaced with the correct github type, but it requires refactoring to
# not require the github module to be installed everywhere.
def find_comment(self, pr: any) -> any:
for comment in pr.as_issue().get_comments():
if self.comment_tag in comment.body:
return comment
return None
def update_pr(self, comment_text: str, args: FormatArgs, create_new: bool) -> None:
import github
from github import IssueComment, PullRequest
repo = github.Github(args.token).get_repo(args.repo)
pr = repo.get_issue(args.issue_number).as_pull_request()
comment_text = self.comment_tag + "\n\n" + comment_text
existing_comment = self.find_comment(pr)
if args.write_comment_to_file:
if create_new or existing_comment:
self.comment = {"body": comment_text}
if existing_comment:
self.comment["id"] = existing_comment.id
return
if existing_comment:
existing_comment.edit(comment_text)
elif create_new:
pr.as_issue().create_comment(comment_text)
def run(self, changed_files: List[str], args: FormatArgs) -> bool:
changed_files = [arg for arg in changed_files if "third-party" not in arg]
diff = self.format_run(changed_files, args)
should_update_gh = args.token is not None and args.repo is not None
if diff is None:
if should_update_gh:
comment_text = (
":white_check_mark: With the latest revision "
f"this PR passed the {self.friendly_name}."
)
self.update_pr(comment_text, args, create_new=False)
return True
elif len(diff) > 0:
if should_update_gh:
comment_text = self.pr_comment_text_for_diff(diff)
self.update_pr(comment_text, args, create_new=True)
else:
print(
f"Warning: {self.friendly_name}, {self.name} detected "
"some issues with your code formatting..."
)
return False
else:
# The formatter failed but didn't output a diff (e.g. some sort of
# infrastructure failure).
comment_text = (
f":warning: The {self.friendly_name} failed without printing "
"a diff. Check the logs for stderr output. :warning:"
)
self.update_pr(comment_text, args, create_new=False)
return False
class ClangFormatHelper(FormatHelper):
name = "clang-format"
friendly_name = "C/C++ code formatter"
@property
def cformat_wrapper_path(self) -> str:
relpath = "../../Scripts/clang-format.py"
curpath = os.path.dirname(os.path.abspath(__file__))
return os.path.abspath(os.path.normpath(os.path.join(curpath, relpath)))
@property
def instructions(self) -> str:
return " ".join(self.cf_cmd)
def should_include_extensionless_file(self, path: str) -> bool:
return path.startswith("libcxx/include")
def filter_changed_files(self, changed_files: List[str]) -> List[str]:
filtered_files = []
for path in changed_files:
_, ext = os.path.splitext(path)
if ext in (".cpp", ".c", ".h", ".hpp", ".hxx", ".cxx", ".inc", ".cppm"):
filtered_files.append(path)
elif ext == "" and self.should_include_extensionless_file(path):
filtered_files.append(path)
return filtered_files
@property
def clang_fmt_path(self) -> str:
if "CLANG_FORMAT_PATH" in os.environ:
return os.environ["CLANG_FORMAT_PATH"]
return "git-clang-format"
def has_tool(self) -> bool:
cmd = [self.clang_fmt_path, "-h"]
proc = None
try:
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
except:
return False
return proc.returncode == 0
def format_run(self, changed_files: List[str], args: FormatArgs) -> Optional[str]:
cpp_files = self.filter_changed_files(changed_files)
if not cpp_files:
return None
cf_cmd = [
self.clang_fmt_path,
f"--binary={self.cformat_wrapper_path}",
"--diff",
]
if args.start_rev and args.end_rev:
cf_cmd.append(args.start_rev)
cf_cmd.append(args.end_rev)
cf_cmd.append("--")
cf_cmd += cpp_files
if args.verbose:
print(f"Running: {' '.join(cf_cmd)}")
self.cf_cmd = cf_cmd
proc = subprocess.run(cf_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
sys.stdout.write(proc.stderr.decode("utf-8"))
if proc.returncode != 0:
# formatting needed, or the command otherwise failed
if args.verbose:
print(f"error: {self.name} exited with code {proc.returncode}")
# Print the diff in the log so that it is viewable there
print(proc.stdout.decode("utf-8"))
return proc.stdout.decode("utf-8")
else:
return None
ALL_FORMATTERS = [ClangFormatHelper()]
def hook_main():
# fill out args
args = FormatArgs()
args.verbose = False
# find the changed files
cmd = ["git", "diff", "--cached", "--name-only", "--diff-filter=d"]
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
output = proc.stdout.decode("utf-8")
for line in output.splitlines():
args.changed_files.append(line)
failed_fmts = []
for fmt in ALL_FORMATTERS:
if fmt.has_tool():
if not fmt.run(args.changed_files, args):
failed_fmts.append(fmt.name)
if fmt.comment:
comments.append(fmt.comment)
else:
print(f"Couldn't find {fmt.name}, can't check " + fmt.friendly_name.lower())
if len(failed_fmts) > 0:
sys.exit(1)
sys.exit(0)
if __name__ == "__main__":
script_path = os.path.abspath(__file__)
if ".git/hooks" in script_path:
hook_main()
sys.exit(0)
parser = argparse.ArgumentParser()
parser.add_argument(
"--token", type=str, required=False, help="GitHub authentication token"
)
parser.add_argument(
"--repo",
type=str,
default=os.getenv("GITHUB_REPOSITORY", "llvm/llvm-project"),
help="The GitHub repository that we are working with in the form of <owner>/<repo> (e.g. llvm/llvm-project)",
)
parser.add_argument("--issue-number", type=int, required=True)
parser.add_argument(
"--start-rev",
type=str,
required=True,
help="Compute changes from this revision.",
)
parser.add_argument(
"--end-rev", type=str, required=True, help="Compute changes to this revision"
)
parser.add_argument(
"--changed-files",
type=str,
help="Comma separated list of files that has been changed",
)
parser.add_argument(
"--write-comment-to-file",
type=str,
help="Don't post comments on the PR, instead write the comments and metadata a file",
)
args = FormatArgs(parser.parse_args())
changed_files = []
if args.changed_files:
changed_files = args.changed_files.split(",")
failed_formatters = []
comments = []
for fmt in ALL_FORMATTERS:
if not fmt.run(changed_files, args):
failed_formatters.append(fmt.name)
if fmt.comment:
comments.append(fmt.comment)
if len(comments):
with open(args.write_comment_to_file, "w") as f:
import json
json.dump(comments, f)
if len(failed_formatters) > 0:
print(f"error: some formatters failed: {' '.join(failed_formatters)}")
sys.exit(1)
-52
View File
@@ -1,52 +0,0 @@
#
# This file is autogenerated by pip-compile with Python 3.11
# by the following command:
#
# pip-compile --output-file=llvm/utils/git/requirements_formatting.txt llvm/utils/git/requirements_formatting.txt.in
#
black==23.9.1
# via
# -r llvm/utils/git/requirements_formatting.txt.in
# darker
certifi==2023.7.22
# via requests
cffi==1.15.1
# via
# cryptography
# pynacl
charset-normalizer==3.2.0
# via requests
click==8.1.7
# via black
cryptography==41.0.3
# via pyjwt
darker==1.7.2
# via -r llvm/utils/git/requirements_formatting.txt.in
deprecated==1.2.14
# via pygithub
idna==3.4
# via requests
mypy-extensions==1.0.0
# via black
packaging==23.1
# via black
pathspec==0.11.2
# via black
platformdirs==3.10.0
# via black
pycparser==2.21
# via cffi
pygithub==1.59.1
# via -r llvm/utils/git/requirements_formatting.txt.in
pyjwt[crypto]==2.8.0
# via pygithub
pynacl==1.5.0
# via pygithub
requests==2.31.0
# via pygithub
toml==0.10.2
# via darker
urllib3==2.0.4
# via requests
wrapt==1.15.0
# via deprecated
+1 -1
+1 -1
+1 -1
+4 -3
View File
@@ -13,6 +13,8 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
set(_M_ARM_64 1)
endif()
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
include(CheckPIESupported)
@@ -28,11 +30,10 @@ set(CMAKE_REQUIRED_FLAGS "-std=c++11 -Wattributes -Werror=attributes")
check_cxx_source_compiles(
"
__attribute__((preserve_all))
int Testy(int a, int b, int c, int d, int e, int f) {
return a + b + c + d + e + f;
void Testy() {
}
int main() {
return Testy(0, 1, 2, 3, 4, 5);
return 0;
}"
HAS_CLANG_PRESERVE_ALL)
unset(CMAKE_REQUIRED_FLAGS)
-21
View File
@@ -441,24 +441,6 @@ def print_parse_envloader_options(options):
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_jsonloader_options(options):
output_argloader.write("#ifdef JSONLOADER\n")
output_argloader.write("#undef JSONLOADER\n")
output_argloader.write("if (false) {}\n")
for op_group, group_vals in options.items():
for op_key, op_vals in group_vals.items():
value_type = op_vals["Type"]
if (value_type == "strenum"):
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
output_argloader.write("Set(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
output_argloader.write("}\n")
output_argloader.write("else {{\n".format(op_key))
output_argloader.write("Set(KeyOption, ConfigString);\n")
output_argloader.write("}\n")
output_argloader.write("#endif\n")
def print_parse_enum_options(options):
output_argloader.write("#ifdef ENUMDEFINES\n")
output_argloader.write("#undef ENUMDEFINES\n")
@@ -574,9 +556,6 @@ print_parse_argloader_options(options);
# Generate environment loader code
print_parse_envloader_options(options);
# Generate json loader code
print_parse_jsonloader_options(options);
# Generate enum variable options
print_parse_enum_options(options);
+19 -54
View File
@@ -55,7 +55,6 @@ class OpDefinition:
DynamicDispatch: bool
JITDispatch: bool
JITDispatchOverride: str
TiedSource: int
Arguments: list
EmitValidation: list
Desc: list
@@ -78,7 +77,6 @@ class OpDefinition:
self.DynamicDispatch = False
self.JITDispatch = True
self.JITDispatchOverride = None
self.TiedSource = -1
self.Arguments = []
self.EmitValidation = []
self.Desc = []
@@ -250,9 +248,6 @@ def parse_ops(ops):
if "JITDispatchOverride" in op_val:
OpDef.JITDispatchOverride = op_val["JITDispatchOverride"]
if "TiedSource" in op_val:
OpDef.TiedSource = op_val["TiedSource"]
# Do some fixups of the data here
if len(OpDef.EmitValidation) != 0:
for i in range(len(OpDef.EmitValidation)):
@@ -377,30 +372,13 @@ def print_ir_sizes():
output_file.write("[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }\n\n")
output_file.write(
'[[nodiscard, gnu::const, gnu::visibility("default")]] std::string_view const& GetName(IROps Op);\n'
)
output_file.write(
'[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetArgs(IROps Op);\n'
)
output_file.write(
'[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetRAArgs(IROps Op);\n'
)
output_file.write(
'[[nodiscard, gnu::const, gnu::visibility("default")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n'
)
output_file.write(
'[[nodiscard, gnu::const, gnu::visibility("default")]] bool HasSideEffects(IROps Op);\n'
)
output_file.write(
'[[nodiscard, gnu::const, gnu::visibility("default")]] bool ImplicitFlagClobber(IROps Op);\n'
)
output_file.write(
'[[nodiscard, gnu::const, gnu::visibility("default")]] bool GetHasDest(IROps Op);\n'
)
output_file.write(
'[[nodiscard, gnu::const, gnu::visibility("default")]] int8_t TiedSource(IROps Op);\n'
)
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] std::string_view const& GetName(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetArgs(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool ImplicitFlagClobber(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
output_file.write("#undef IROP_SIZES\n")
output_file.write("#endif\n\n")
@@ -493,25 +471,15 @@ def print_ir_getraargs():
def print_ir_hassideeffects():
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
for array, prop, T in [
("SideEffects", "HasSideEffects", "bool"),
("ImplicitFlagClobbers", "ImplicitFlagClobber", "bool"),
("TiedSources", "TiedSource", "int8_t"),
]:
output_file.write(
f"constexpr std::array<{'uint8_t' if T == 'bool' else T}, OP_LAST + 1> {array} = {{\n"
)
for array, prop in [("SideEffects", "HasSideEffects"),
("ImplicitFlagClobbers", "ImplicitFlagClobber")]:
output_file.write(f"constexpr std::array<uint8_t, OP_LAST + 1> {array} = {{\n")
for op in IROps:
if T == "bool":
output_file.write(
"\t{},\n".format(("true" if getattr(op, prop) else "false"))
)
else:
output_file.write(f"\t{getattr(op, prop)},\n")
output_file.write("\t{},\n".format(("true" if getattr(op, prop) else "false")))
output_file.write("};\n\n")
output_file.write(f"{T} {prop}(IROps Op) {{\n")
output_file.write(f"bool {prop}(IROps Op) {{\n")
output_file.write(f" return {array}[Op];\n")
output_file.write("}\n")
@@ -552,20 +520,14 @@ def print_ir_arg_printer():
output_file.write("\t*out << \" \";\n")
SSAArgNum = 0
FirstArg = True
for i in range(0, len(op.Arguments)):
arg = op.Arguments[i]
LastArg = len(op.Arguments) - i - 1 == 0
# No point printing temporaries that we can't recover
if arg.Temporary:
continue
if FirstArg:
FirstArg = False
else:
output_file.write('\t*out << ", ";\n')
if arg.IsSSA:
# Temporary that we can't recover
output_file.write("\t*out << \"{}:Tmp:{}\";\n".format(arg.Type, arg.Name))
elif arg.IsSSA:
# SSA value
output_file.write("\tPrintArg(out, IR, Op->Header.Args[{}], RAData);\n".format(SSAArgNum))
SSAArgNum = SSAArgNum + 1
@@ -573,6 +535,9 @@ def print_ir_arg_printer():
# User defined op that is stored
output_file.write("\tPrintArg(out, IR, Op->{});\n".format(arg.Name))
if not LastArg:
output_file.write("\t*out << \", \";\n")
output_file.write("break;\n")
output_file.write("}\n")
+18 -6
View File
@@ -7,7 +7,6 @@ set (FEXCORE_BASE_SRCS
Utils/FileLoading.cpp
Utils/ForcedAssert.cpp
Utils/LogManager.cpp
Utils/SpinWaitLock.cpp
)
if (NOT MINGW_BUILD)
@@ -95,7 +94,6 @@ set (SRCS
Interface/Core/ObjectCache/JobHandling.cpp
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
Interface/Core/ObjectCache/ObjectCacheService.cpp
Interface/Core/OpcodeDispatcher/AVX_128.cpp
Interface/Core/OpcodeDispatcher/Crypto.cpp
Interface/Core/OpcodeDispatcher/Flags.cpp
Interface/Core/OpcodeDispatcher/Vector.cpp
@@ -135,16 +133,23 @@ set (SRCS
Interface/GDBJIT/GDBJIT.cpp
Interface/IR/AOTIR.cpp
Interface/IR/IRDumper.cpp
Interface/IR/IRParser.cpp
Interface/IR/IREmitter.cpp
Interface/IR/PassManager.cpp
Interface/IR/Passes/ConstProp.cpp
Interface/IR/Passes/DeadCodeElimination.cpp
Interface/IR/Passes/DeadContextStoreElimination.cpp
Interface/IR/Passes/IRCompaction.cpp
Interface/IR/Passes/IRDumperPass.cpp
Interface/IR/Passes/IRValidation.cpp
Interface/IR/Passes/RAValidation.cpp
Interface/IR/Passes/LongDivideRemovalPass.cpp
Interface/IR/Passes/ValueDominanceValidation.cpp
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
Interface/IR/Passes/DeadStoreElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/InlineCallOptimization.cpp
Utils/NetStream.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/Profiler.cpp
@@ -189,15 +194,12 @@ endif()
# Some defines for the softfloat library
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils CodeEmitter)
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
if (NOT MINGW_BUILD)
list (APPEND LIBS dl)
else()
list (APPEND LIBS synchronization)
if (_M_ARM_64EC)
list (APPEND LIBS kernelbase)
endif()
endif()
if (ENABLE_JEMALLOC)
@@ -368,6 +370,16 @@ function(AddLibrary Name Type)
target_link_libraries(${Name} FEXCore_Base)
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
if (MINGW_BUILD)
# Mingw build isn't building a linux shared library, so it can't have a SONAME.
set_target_properties(${Name} PROPERTIES NO_SONAME ON)
# Change the suffixes otherwise cmake continues using .a and .so
if (${Type} STREQUAL SHARED)
set_target_properties(${Name} PROPERTIES SUFFIX ".dll")
elseif(${Type} STREQUAL STATIC)
set_target_properties(${Name} PROPERTIES SUFFIX ".lib")
endif()
endif()
AddDefaultOptionsToTarget(${Name})
endfunction()
+9 -11
View File
@@ -18,14 +18,14 @@ struct BitSet final {
constexpr static size_t MinimumSize = sizeof(ElementType);
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
ElementType* Memory;
ElementType *Memory;
void Allocate(size_t Elements) {
size_t AllocateSize = ToBytes(Elements);
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
}
void Realloc(size_t Elements) {
size_t AllocateSize = ToBytes(Elements);
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
}
@@ -43,13 +43,10 @@ struct BitSet final {
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
}
void MemClear(size_t Elements) {
memset(Memory, 0, ToBytes(Elements));
memset(Memory, 0, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
}
void MemSet(size_t Elements) {
memset(Memory, 0xFF, ToBytes(Elements));
}
uint32_t ToBytes(size_t Elements) {
return AlignUp(Elements, MinimumSizeBits) / MinimumSize;
memset(Memory, 0xFF, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
}
// This very explicitly doesn't let you take an address
@@ -65,10 +62,11 @@ struct BitSetView final {
constexpr static size_t MinimumSize = sizeof(ElementType);
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
ElementType* Memory;
ElementType *Memory;
void GetView(BitSet<T>& Set, uint64_t ElementOffset) {
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0,
"Bitset view offset needs to be aligned to size of backing element");
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
}
+122 -131
View File
@@ -7,142 +7,133 @@
#include <unistd.h>
namespace FEXCore {
JITSymbols::JITSymbols() {}
JITSymbols::~JITSymbols() {
if (fd != -1) {
close(fd);
}
}
void JITSymbols::InitFile() {
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
#ifdef __ANDROID__
// Android simpleperf looks in /data/local/tmp instead of /tmp
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
#else
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
#endif
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
}
void JITSymbols::RegisterNamedRegion(const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) {
return;
JITSymbols::JITSymbols() {
}
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::RegisterJITSpace(const void* HostAddr, uint32_t CodeSize) {
if (fd == -1) {
return;
}
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
// Buffered JIT symbols.
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (fd == -1) {
return;
}
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, GuestAddr, CodeSize);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (fd == -1) {
return;
}
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult =
fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, CodeSize, Name, Offset);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) {
return;
}
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}\n", HostAddr, CodeSize, Name);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
RegisterNamedRegion(Buffer, HostAddr, CodeSize, Name);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite) {
auto Now = std::chrono::steady_clock::now();
if (!ForceWrite) {
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) && Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
// Still buffering, no need to write.
return;
JITSymbols::~JITSymbols() {
if (fd != -1) {
close(fd);
}
}
Buffer->LastWrite = Now;
auto Result = write(fd, Buffer->Buffer, Buffer->Offset);
if (Result == -1 && errno == EBADF) {
fd = -1;
void JITSymbols::InitFile() {
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
#ifdef __ANDROID__
// Android simpleperf looks in /data/local/tmp instead of /tmp
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
#else
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
#endif
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
}
Buffer->Offset = 0;
}
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
if (fd == -1) return;
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto Buffer = fextl::fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
auto Result = write(fd, Buffer.c_str(), Buffer.size());
if (Result == -1 && errno == EBADF) {
fd = -1;
}
}
// Buffered JIT symbols.
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, GuestAddr, CodeSize);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
Register(Buffer, HostAddr, CodeSize, Name, Offset);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
if (fd == -1) return;
// Calculate remaining sizes.
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
// Linux perf format is very straightforward
// `<HostPtr> <Size> <Name>\n`
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}\n", HostAddr, CodeSize, Name);
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
// Couldn't fit, need to force a write.
WriteBuffer(Buffer, true);
// Rerun
RegisterNamedRegion(Buffer, HostAddr, CodeSize, Name);
return;
}
Buffer->Offset += FMTResult.size;
WriteBuffer(Buffer);
}
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite) {
auto Now = std::chrono::steady_clock::now();
if (!ForceWrite) {
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) &&
Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
// Still buffering, no need to write.
return;
}
}
Buffer->LastWrite = Now;
auto Result = write(fd, Buffer->Buffer, Buffer->Offset);
if (Result == -1 && errno == EBADF) {
fd = -1;
}
Buffer->Offset = 0;
}
} // namespace FEXCore
+8 -8
View File
@@ -17,20 +17,20 @@ public:
~JITSymbols();
void InitFile();
void RegisterNamedRegion(const void* HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void* HostAddr, uint32_t CodeSize);
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
// Allocate JIT buffer.
static fextl::unique_ptr<Core::JITSymbolBuffer> AllocateBuffer() {
return fextl::make_unique<Core::JITSymbolBuffer>();
}
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name);
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
void RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name);
private:
int fd {-1};
void WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite = false);
int fd{-1};
void WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite = false);
};
} // namespace FEXCore
}
+123 -93
View File
@@ -1,10 +1,10 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/string.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <cmath>
#include <cstring>
@@ -45,13 +45,13 @@ struct FEX_PACKED X80SoftFloat {
uint16_t Exponent : 15;
uint16_t Sign : 1;
X80SoftFloat() {
memset(this, 0, sizeof(*this));
}
X80SoftFloat() { memset(this, 0, sizeof(*this)); }
X80SoftFloat(uint16_t _Sign, uint16_t _Exponent, uint64_t _Significand)
: Significand {_Significand}
, Exponent {_Exponent}
, Sign {_Sign} {}
, Sign {_Sign}
{
}
fextl::string str() const {
fextl::ostringstream string;
@@ -63,19 +63,21 @@ struct FEX_PACKED X80SoftFloat {
}
// Ops
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FADD(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
faddp;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -83,19 +85,21 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSUB(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fsubp;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -103,19 +107,21 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FMUL(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fmulp;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -123,19 +129,21 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FDIV(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
fdivp;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -143,10 +151,11 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
@@ -154,9 +163,10 @@ struct FEX_PACKED X80SoftFloat {
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -164,10 +174,11 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM1(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
@@ -175,9 +186,10 @@ struct FEX_PACKED X80SoftFloat {
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -185,27 +197,30 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(const X80SoftFloat& lhs, uint_fast8_t RoundMode) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
return extF80_roundToInt(lhs, RoundMode, false);
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FXTRACT_SIG(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fxtract;
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st", "st(1)");
return Result;
#else
@@ -216,19 +231,20 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FXTRACT_EXP(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
#if defined(DEBUG_X86_FLOAT)
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fxtract;
ffreep %%st(0);
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st", "st(1)");
return Result;
#else
@@ -237,17 +253,19 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static void FCMP(const X80SoftFloat& lhs, const X80SoftFloat& rhs, bool* eq, bool* lt, bool* nan) {
FEXCORE_PRESERVE_ALL_ATTR
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
*eq = extF80_eq(lhs, rhs);
*lt = extF80_lt(lhs, rhs);
*nan = IsNan(lhs) || IsNan(rhs);
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSCALE(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st1
fldt %[lhs]; # st0
@@ -255,9 +273,10 @@ struct FEX_PACKED X80SoftFloat {
fstpt %[result];
ffreep %%st(0);
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -270,19 +289,20 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat F2XM1(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
f2xm1; # st0 = 2^st(0) - 1
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -293,20 +313,22 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FYL2X(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[rhs]; # st(1)
fldt %[lhs]; # st(0)
fyl2x; # st(1) * log2l(st(0))
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -317,20 +339,22 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FATAN(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs];
fldt %[rhs];
fpatan;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs), [rhs] "m"(rhs)
: "st", "st(1)");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
, [rhs] "m" (rhs)
: "st", "st(1)");
return Result;
#else
@@ -341,20 +365,21 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FTAN(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fptan;
ffreep %%st(0);
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -364,19 +389,20 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSIN(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fsin;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -386,19 +412,20 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FCOS(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fcos;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -408,18 +435,19 @@ struct FEX_PACKED X80SoftFloat {
#endif
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSQRT(const X80SoftFloat& lhs) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
#ifdef DEBUG_X86_FLOAT
BIGFLOAT Result;
asm(R"(
asm (R"(
fninit;
fldt %[lhs]; # st0
fsqrt;
fstpt %[result];
)"
: [result] "=m"(Result)
: [lhs] "m"(lhs)
: "st");
: [result] "=m" (Result)
: [lhs] "m" (lhs)
: "st");
return Result;
#else
@@ -443,7 +471,7 @@ struct FEX_PACKED X80SoftFloat {
const float128_t Result = extF80_to_f128(*this);
return FEXCore::BitCast<BIGFLOAT>(Result);
#else
BIGFLOAT result {};
BIGFLOAT result{};
memcpy(&result, this, sizeof(result));
return result;
#endif
@@ -542,17 +570,19 @@ struct FEX_PACKED X80SoftFloat {
}
operator extFloat80_t() const {
extFloat80_t Result {};
extFloat80_t Result{};
Result.signif = Significand;
Result.signExp = Exponent | (Sign << 15);
return Result;
}
static bool IsNan(const X80SoftFloat& lhs) {
return (lhs.Exponent == 0x7FFF) && (lhs.Significand & IntegerBit) && (lhs.Significand & Bottom62Significand);
static bool IsNan(X80SoftFloat const &lhs) {
return (lhs.Exponent == 0x7FFF) &&
(lhs.Significand & IntegerBit) &&
(lhs.Significand & Bottom62Significand);
}
static bool SignBit(const X80SoftFloat& lhs) {
static bool SignBit(X80SoftFloat const &lhs) {
return lhs.Sign;
}
+34 -41
View File
@@ -7,51 +7,44 @@
#include <optional>
namespace FEXCore::StrConv {
[[maybe_unused]]
static bool Conv(std::string_view Value, bool* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, bool *Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint8_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, uint8_t *Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint16_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, uint16_t *Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint32_t* Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, uint32_t *Result) {
*Result = std::strtoul(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, int32_t* Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, int32_t *Result) {
*Result = std::strtol(Value.data(), nullptr, 0);
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, uint64_t* Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
[[maybe_unused]]
static bool Conv(std::string_view Value, T* Result) {
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
return true;
}
[[maybe_unused]] static bool Conv(std::string_view Value, uint64_t *Result) {
*Result = std::strtoull(Value.data(), nullptr, 0);
return true;
}
template <typename T,
typename = std::enable_if<std::is_enum<T>::value, T>>
[[maybe_unused]] static bool Conv(std::string_view Value, T *Result) {
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
return true;
}
[[maybe_unused]]
static bool Conv(std::string_view Value, fextl::string* Result) {
*Result = Value;
return true;
[[maybe_unused]] static bool Conv(std::string_view Value, fextl::string *Result) {
*Result = Value;
return true;
}
}
} // namespace FEXCore::StrConv
+417 -409
View File
@@ -29,7 +29,7 @@
#include <utility>
namespace FEXCore::Context {
class Context;
class Context;
}
namespace FEXCore::Config {
@@ -40,464 +40,472 @@ namespace DefaultValues {
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
#include <FEXCore/Config/ConfigValues.inl>
} // namespace DefaultValues
enum Paths {
PATH_DATA_DIR = 0,
PATH_CONFIG_DIR_LOCAL,
PATH_CONFIG_DIR_GLOBAL,
PATH_CONFIG_FILE_LOCAL,
PATH_CONFIG_FILE_GLOBAL,
PATH_CONFIG_TELEMETRY_FOLDER,
PATH_LAST,
};
static std::array<fextl::string, Paths::PATH_LAST> Paths;
void SetDataDirectory(const std::string_view Path) {
Paths[PATH_DATA_DIR] = Path;
}
void SetConfigDirectory(const std::string_view Path, bool Global) {
Paths[PATH_CONFIG_DIR_LOCAL + Global] = Path;
}
enum Paths {
PATH_DATA_DIR = 0,
PATH_CONFIG_DIR_LOCAL,
PATH_CONFIG_DIR_GLOBAL,
PATH_CONFIG_FILE_LOCAL,
PATH_CONFIG_FILE_GLOBAL,
PATH_LAST,
};
static std::array<fextl::string, Paths::PATH_LAST> Paths;
void SetConfigFileLocation(const std::string_view Path, bool Global) {
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
}
void SetDataDirectory(const std::string_view Path) {
Paths[PATH_DATA_DIR] = Path;
}
const fextl::string& GetTelemetryDirectory() {
auto& Path = Paths[PATH_CONFIG_TELEMETRY_FOLDER];
if (Path.empty()) {
FEX_CONFIG_OPT(TelemetryDirectory, TELEMETRYDIRECTORY);
if (!TelemetryDirectory().empty()) {
Path = TelemetryDirectory;
Path += "/";
} else {
Path = Config::GetDataDirectory() + "Telemetry/";
void SetConfigDirectory(const std::string_view Path, bool Global) {
Paths[PATH_CONFIG_DIR_LOCAL + Global] = Path;
}
void SetConfigFileLocation(const std::string_view Path, bool Global) {
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
}
fextl::string const& GetDataDirectory() {
return Paths[PATH_DATA_DIR];
}
fextl::string const& GetConfigDirectory(bool Global) {
return Paths[PATH_CONFIG_DIR_LOCAL + Global];
}
fextl::string const& GetConfigFileLocation(bool Global) {
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
}
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
fextl::string ConfigFile = GetConfigDirectory(Global);
if (!Global &&
!FHU::Filesystem::Exists(ConfigFile) &&
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
// Let's go local in this case
return fextl::fmt::format("./{}.json", Program);
}
}
return Path;
}
ConfigFile += "AppConfig/";
const fextl::string& GetDataDirectory() {
return Paths[PATH_DATA_DIR];
}
const fextl::string& GetConfigDirectory(bool Global) {
return Paths[PATH_CONFIG_DIR_LOCAL + Global];
}
const fextl::string& GetConfigFileLocation(bool Global) {
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
}
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
fextl::string ConfigFile = GetConfigDirectory(Global);
if (!Global && !FHU::Filesystem::Exists(ConfigFile) && !FHU::Filesystem::CreateDirectories(ConfigFile)) {
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
// Let's go local in this case
return fextl::fmt::format("./{}.json", Program);
}
ConfigFile += "AppConfig/";
// Attempt to create the local folder if it doesn't exist
if (!Global && !FHU::Filesystem::Exists(ConfigFile) && !FHU::Filesystem::CreateDirectories(ConfigFile)) {
// Let's go local in this case
return fextl::fmt::format("./{}.json", Program);
}
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
}
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, uint64_t Config) {}
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, const fextl::string& Config) {}
uint64_t GetConfig(FEXCore::Context::Context* CTX, ConfigOption Option) {
return 0;
}
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
static FEXCore::Config::Layer* Meta {};
constexpr std::array<FEXCore::Config::LayerType, 10> LoadOrder = {
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN, FEXCore::Config::LayerType::LAYER_MAIN,
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP, FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP, FEXCore::Config::LayerType::LAYER_LOCAL_APP,
FEXCore::Config::LayerType::LAYER_ARGUMENTS, FEXCore::Config::LayerType::LAYER_USER_OVERRIDE,
FEXCore::Config::LayerType::LAYER_ENVIRONMENT, FEXCore::Config::LayerType::LAYER_TOP};
Layer::Layer(const LayerType _Type)
: Type {_Type} {}
Layer::~Layer() {}
class MetaLayer final : public FEXCore::Config::Layer {
public:
MetaLayer(const LayerType _Type)
: FEXCore::Config::Layer(_Type) {}
~MetaLayer() {}
void Load();
private:
void MergeConfigMap(const LayerOptions& Options);
void MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value);
};
void MetaLayer::Load() {
OptionMap.clear();
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
auto it = ConfigLayers.find(*CurrentLayer);
if (it != ConfigLayers.end() && *CurrentLayer != Type) {
// Merge this layer's options to this layer
MergeConfigMap(it->second->GetOptionMap());
// Attempt to create the local folder if it doesn't exist
if (!Global &&
!FHU::Filesystem::Exists(ConfigFile) &&
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
// Let's go local in this case
return fextl::fmt::format("./{}.json", Program);
}
}
}
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value) {
// Environment variables need a bit of additional work
// We want to merge the arrays rather than overwrite entirely
auto MetaEnvironment = OptionMap.find(Option);
if (MetaEnvironment == OptionMap.end()) {
// Doesn't exist, just insert
OptionMap.insert_or_assign(Option, Value);
return;
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
}
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
const auto AddToMap = [&LookupMap](const FEXCore::Config::LayerValue& Value) {
for (const auto& EnvVar : Value) {
const auto ItEq = EnvVar.find_first_of('=');
if (ItEq == fextl::string::npos) {
// Broken environment variable
// Skip
continue;
}
auto Key = fextl::string(EnvVar.begin(), EnvVar.begin() + ItEq);
auto Value = fextl::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
}
// Add the key to the map, overwriting whatever previous value was there
LookupMap.insert_or_assign(std::move(Key), std::move(Value));
}
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, fextl::string const &Config) {
}
uint64_t GetConfig(FEXCore::Context::Context *CTX, ConfigOption Option) {
return 0;
}
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
static FEXCore::Config::Layer *Meta{};
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
FEXCore::Config::LayerType::LAYER_MAIN,
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP,
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP,
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
FEXCore::Config::LayerType::LAYER_ARGUMENTS,
FEXCore::Config::LayerType::LAYER_ENVIRONMENT,
FEXCore::Config::LayerType::LAYER_TOP
};
AddToMap(MetaEnvironment->second);
AddToMap(Value);
// Now with the two layers merged in the map
// Add all the values to the option
Erase(Option);
for (auto& Val : LookupMap) {
// Set will emplace multiple options in to its list
Set(Option, Val.first + "=" + Val.second);
Layer::Layer(const LayerType _Type)
: Type {_Type} {
}
}
void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
// Insert this layer's options, overlaying previous options that exist here
for (auto& it : Options) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
MergeEnvironmentVariables(it.first, it.second);
} else {
OptionMap.insert_or_assign(it.first, it.second);
Layer::~Layer() {
}
class MetaLayer final : public FEXCore::Config::Layer {
public:
MetaLayer(const LayerType _Type)
: FEXCore::Config::Layer (_Type) {
}
}
}
void Initialize() {
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
Meta = ConfigLayers.begin()->second.get();
}
void Shutdown() {
ConfigLayers.clear();
Meta = nullptr;
}
void Load() {
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
auto it = ConfigLayers.find(*CurrentLayer);
if (it != ConfigLayers.end()) {
it->second->Load();
~MetaLayer() {
}
}
}
void Load();
fextl::string ExpandPath(const fextl::string& ContainerPrefix, fextl::string PathName) {
if (PathName.empty()) {
return {};
private:
void MergeConfigMap(const LayerOptions &Options);
void MergeEnvironmentVariables(ConfigOption const &Option, LayerValue const &Value);
};
void MetaLayer::Load() {
OptionMap.clear();
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
auto it = ConfigLayers.find(*CurrentLayer);
if (it != ConfigLayers.end() && *CurrentLayer != Type) {
// Merge this layer's options to this layer
MergeConfigMap(it->second->GetOptionMap());
}
}
}
// Expand home if it exists
if (FHU::Filesystem::IsRelative(PathName)) {
fextl::string Home = getenv("HOME") ?: "";
// Home expansion only works if it is the first character
// This matches bash behaviour
if (PathName.at(0) == '~') {
PathName.replace(0, 1, Home);
return PathName;
void MetaLayer::MergeEnvironmentVariables(ConfigOption const &Option, LayerValue const &Value) {
// Environment variables need a bit of additional work
// We want to merge the arrays rather than overwrite entirely
auto MetaEnvironment = OptionMap.find(Option);
if (MetaEnvironment == OptionMap.end()) {
// Doesn't exist, just insert
OptionMap.insert_or_assign(Option, Value);
return;
}
// Expand relative path to absolute
char ExistsTempPath[PATH_MAX];
char* RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
if (RealPath) {
PathName = RealPath;
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
const auto AddToMap = [&LookupMap](FEXCore::Config::LayerValue const &Value) {
for (const auto &EnvVar : Value) {
const auto ItEq = EnvVar.find_first_of('=');
if (ItEq == fextl::string::npos) {
// Broken environment variable
// Skip
continue;
}
auto Key = fextl::string(EnvVar.begin(), EnvVar.begin() + ItEq);
auto Value = fextl::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
// Add the key to the map, overwriting whatever previous value was there
LookupMap.insert_or_assign(std::move(Key), std::move(Value));
}
};
AddToMap(MetaEnvironment->second);
AddToMap(Value);
// Now with the two layers merged in the map
// Add all the values to the option
Erase(Option);
for (auto &Val : LookupMap) {
// Set will emplace multiple options in to its list
Set(Option, Val.first + "=" + Val.second);
}
}
void MetaLayer::MergeConfigMap(const LayerOptions &Options) {
// Insert this layer's options, overlaying previous options that exist here
for (auto &it : Options) {
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV ||
it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
MergeEnvironmentVariables(it.first, it.second);
}
else {
OptionMap.insert_or_assign(it.first, it.second);
}
}
}
void Initialize() {
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
Meta = ConfigLayers.begin()->second.get();
}
void Shutdown() {
ConfigLayers.clear();
Meta = nullptr;
}
void Load() {
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
auto it = ConfigLayers.find(*CurrentLayer);
if (it != ConfigLayers.end()) {
it->second->Load();
}
}
}
fextl::string ExpandPath(fextl::string const &ContainerPrefix, fextl::string PathName) {
if (PathName.empty()) {
return {};
}
// Only return if it exists
if (FHU::Filesystem::Exists(PathName)) {
return PathName;
// Expand home if it exists
if (FHU::Filesystem::IsRelative(PathName)) {
fextl::string Home = getenv("HOME") ?: "";
// Home expansion only works if it is the first character
// This matches bash behaviour
if (PathName.at(0) == '~') {
PathName.replace(0, 1, Home);
return PathName;
}
// Expand relative path to absolute
char ExistsTempPath[PATH_MAX];
char *RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
if (RealPath) {
PathName = RealPath;
}
// Only return if it exists
if (FHU::Filesystem::Exists(PathName)) {
return PathName;
}
}
} else {
// If the containerprefix and pathname isn't empty
// Then we check if the pathname exists in our current namespace
// If the path DOESN'T exist but DOES exist with the prefix applied
// then redirect to the prefix
//
// This might not be expected behaviour for some edge cases but since
// all paths aren't mounted inside the container, then it'll be fine
//
// Main catch case for this is the default thunk install folders
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
if (!ContainerPrefix.empty() && !PathName.empty()) {
if (!FHU::Filesystem::Exists(PathName)) {
auto ContainerPath = ContainerPrefix + PathName;
if (FHU::Filesystem::Exists(ContainerPath)) {
return ContainerPath;
else {
// If the containerprefix and pathname isn't empty
// Then we check if the pathname exists in our current namespace
// If the path DOESN'T exist but DOES exist with the prefix applied
// then redirect to the prefix
//
// This might not be expected behaviour for some edge cases but since
// all paths aren't mounted inside the container, then it'll be fine
//
// Main catch case for this is the default thunk install folders
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
if (!ContainerPrefix.empty() && !PathName.empty()) {
if (!FHU::Filesystem::Exists(PathName)) {
auto ContainerPath = ContainerPrefix + PathName;
if (FHU::Filesystem::Exists(ContainerPath)) {
return ContainerPath;
}
}
}
}
return {};
}
return {};
}
constexpr char ContainerManager[] = "/run/host/container-manager";
constexpr char ContainerManager[] = "/run/host/container-manager";
fextl::string FindContainer() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::vector<char> Manager {};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
return ManagerStr;
}
}
return {};
}
fextl::string FindContainerPrefix() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::vector<char> Manager {};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
// We are running inside of pressure vessel
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
return "/run/host/";
fextl::string FindContainer() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::vector<char> Manager{};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
return ManagerStr;
}
}
return {};
}
return {};
}
void ReloadMetaLayer() {
Meta->Load();
fextl::string FindContainerPrefix() {
// We only support pressure-vessel at the moment
if (FHU::Filesystem::Exists(ContainerManager)) {
fextl::vector<char> Manager{};
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
// Trim the whitespace, may contain a newline
fextl::string ManagerStr = Manager.data();
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
// We are running inside of pressure vessel
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
return "/run/host/";
}
}
}
return {};
}
// Do configuration option fix ups after everything is reloaded
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
// Sanitize Core option
FEX_CONFIG_OPT(Core, CORE);
void ReloadMetaLayer() {
Meta->Load();
// Do configuration option fix ups after everything is reloaded
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
// Sanitize Core option
FEX_CONFIG_OPT(Core, CORE);
#if (_M_X86_64)
constexpr uint32_t MaxCoreNumber = 1;
constexpr uint32_t MaxCoreNumber = 1;
#else
constexpr uint32_t MaxCoreNumber = 0;
constexpr uint32_t MaxCoreNumber = 0;
#endif
if (Core > MaxCoreNumber) {
// Sanitize the core option by setting the core to the JIT if invalid
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(Core, CORE);
}
fextl::string ContainerPrefix {FindContainerPrefix()};
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
auto NewPath = ExpandPath(ContainerPrefix, PathName);
if (!NewPath.empty()) {
FEXCore::Config::EraseSet(Config, NewPath);
}
};
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
FEX_CONFIG_OPT(PathName, ROOTFS);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
} else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
if (FHU::Filesystem::Exists(NamedRootFS)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
if (Core > MaxCoreNumber) {
// Sanitize the core option by setting the core to the JIT if invalid
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
}
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKHOSTLIBS)) {
FEX_CONFIG_OPT(PathName, THUNKHOSTLIBS);
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKGUESTLIBS)) {
FEX_CONFIG_OPT(PathName, THUNKGUESTLIBS);
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
} else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
if (FHU::Filesystem::Exists(NamedConfig)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(Core, CORE);
}
fextl::string ContainerPrefix { FindContainerPrefix() };
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
auto NewPath = ExpandPath(ContainerPrefix, PathName);
if (!NewPath.empty()) {
FEXCore::Config::EraseSet(Config, NewPath);
}
};
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
FEX_CONFIG_OPT(PathName, ROOTFS);
auto ExpandedString = ExpandPath(ContainerPrefix,PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
}
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
if (FHU::Filesystem::Exists(NamedRootFS)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
}
}
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
if (PathName() != "stdout" && PathName() != "stderr" && PathName() != "server") {
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, PathName());
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKHOSTLIBS)) {
FEX_CONFIG_OPT(PathName, THUNKHOSTLIBS);
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKGUESTLIBS)) {
FEX_CONFIG_OPT(PathName, THUNKGUESTLIBS);
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, PathName());
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
if (!ExpandedString.empty()) {
// Adjust the path if it ended up being relative
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
}
else if (!PathName().empty()) {
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
if (FHU::Filesystem::Exists(NamedConfig)) {
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
}
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
if (PathName() != "stdout" && PathName() != "stderr" && PathName() != "server") {
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, PathName());
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_DUMPIR) &&
!FEXCore::Config::Exists(FEXCore::Config::CONFIG_PASSMANAGERDUMPIR)) {
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
FEX_CONFIG_OPT(PathName, DUMPIR);
if (PathName() != "no") {
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR, fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
// Single stepping also enforces single instruction size blocks
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_DUMPIR) && !FEXCore::Config::Exists(FEXCore::Config::CONFIG_PASSMANAGERDUMPIR)) {
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
FEX_CONFIG_OPT(PathName, DUMPIR);
if (PathName() != "no") {
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer) {
ConfigLayers.emplace(_Layer->GetLayerType(), std::move(_Layer));
}
bool Exists(ConfigOption Option) {
return Meta->OptionExists(Option);
}
std::optional<LayerValue*> All(ConfigOption Option) {
return Meta->All(Option);
}
std::optional<fextl::string*> Get(ConfigOption Option) {
return Meta->Get(Option);
}
void Set(ConfigOption Option, std::string_view Data) {
Meta->Set(Option, Data);
}
void Erase(ConfigOption Option) {
Meta->Erase(Option);
}
void EraseSet(ConfigOption Option, std::string_view Data) {
Meta->EraseSet(Option, Data);
}
template<typename T>
T Value<T>::Get(FEXCore::Config::ConfigOption Option) {
T Result;
auto Value = FEXCore::Config::Get(Option);
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
}
}
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
// Single stepping also enforces single instruction size blocks
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
}
}
void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer) {
ConfigLayers.emplace(_Layer->GetLayerType(), std::move(_Layer));
}
bool Exists(ConfigOption Option) {
return Meta->OptionExists(Option);
}
std::optional<LayerValue*> All(ConfigOption Option) {
return Meta->All(Option);
}
std::optional<fextl::string*> Get(ConfigOption Option) {
return Meta->Get(Option);
}
void Set(ConfigOption Option, std::string_view Data) {
Meta->Set(Option, Data);
}
void Erase(ConfigOption Option) {
Meta->Erase(Option);
}
void EraseSet(ConfigOption Option, std::string_view Data) {
Meta->EraseSet(Option, Data);
}
template<typename T>
T Value<T>::Get(FEXCore::Config::ConfigOption Option) {
T Result;
auto Value = FEXCore::Config::Get(Option);
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
}
return Result;
}
template<typename T>
T Value<T>::GetIfExists(FEXCore::Config::ConfigOption Option, T Default) {
T Result;
auto Value = FEXCore::Config::Get(Option);
if (Value && FEXCore::StrConv::Conv(**Value, &Result)) {
return Result;
} else {
return Default;
}
template<typename T>
T Value<T>::GetIfExists(FEXCore::Config::ConfigOption Option, T Default) {
T Result;
auto Value = FEXCore::Config::Get(Option);
if (Value && FEXCore::StrConv::Conv(**Value, &Result)) {
return Result;
}
else {
return Default;
}
}
template<>
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, fextl::string Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
}
else {
return Default;
}
}
template<>
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
}
else {
return fextl::string(Default);
}
}
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
template int16_t Value<int16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int16_t Default);
template uint16_t Value<uint16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint16_t Default);
template int32_t Value<int32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int32_t Default);
template uint32_t Value<uint32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint32_t Default);
template int64_t Value<int64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int64_t Default);
template uint64_t Value<uint64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint64_t Default);
// Constructor
template Value<fextl::string>::Value(FEXCore::Config::ConfigOption _Option, fextl::string Default);
template Value<bool>::Value(FEXCore::Config::ConfigOption _Option, bool Default);
template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t Default);
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
template<typename T>
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List) {
auto Value = FEXCore::Config::All(Option);
List->clear();
if (Value) {
*List = **Value;
}
}
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
}
template<>
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, fextl::string Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
} else {
return Default;
}
}
template<>
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
auto Value = FEXCore::Config::Get(Option);
if (Value) {
return **Value;
} else {
return fextl::string(Default);
}
}
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
template int16_t Value<int16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int16_t Default);
template uint16_t Value<uint16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint16_t Default);
template int32_t Value<int32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int32_t Default);
template uint32_t Value<uint32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint32_t Default);
template int64_t Value<int64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int64_t Default);
template uint64_t Value<uint64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint64_t Default);
// Constructor
template Value<fextl::string>::Value(FEXCore::Config::ConfigOption _Option, fextl::string Default);
template Value<bool>::Value(FEXCore::Config::ConfigOption _Option, bool Default);
template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t Default);
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
template<typename T>
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List) {
auto Value = FEXCore::Config::All(Option);
List->clear();
if (Value) {
*List = **Value;
}
}
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List);
} // namespace FEXCore::Config
+25 -66
View File
@@ -50,6 +50,8 @@
"DISABLESVE": "disablesve",
"ENABLEAVX": "enableavx",
"DISABLEAVX": "disableavx",
"ENABLEAVX2": "enableavx2",
"DISABLEAVX2": "disableavx2",
"ENABLEAFP": "enableafp",
"DISABLEAFP": "disableafp",
"ENABLELRCPC": "enablelrcpc",
@@ -75,15 +77,14 @@
"ENABLECRYPTO": "enablecrypto",
"DISABLECRYPTO": "disablecrypto",
"ENABLERPRES": "enablerpres",
"DISABLERPRES": "disablerpres",
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
"DISABLEPRESERVEALLABI": "disablepreserveallabi"
"DISABLERPRES": "disablerpres"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
"\toff: Default CPU features queried from CPU features",
"\t{enable,disable}sve: Will force enable or disable sve even if the host doesn't support it",
"\t{enable,disable}avx: Will force enable or disable avx even if the host doesn't support it",
"\t{enable,disable}avx2: Will force enable or disable avx2 even if the host doesn't support it",
"\t{enable,disable}afp: Will force enable or disable afp even if the host doesn't support it",
"\t{enable,disable}lrcpc: Will force enable or disable lrcpc even if the host doesn't support it",
"\t{enable,disable}lrcpc2: Will force enable or disable lrcpc2 even if the host doesn't support it",
@@ -96,28 +97,7 @@
"\t{enable,disable}flagm: Will force enable or disable flagm even if the host doesn't support it",
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it",
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
]
},
"CPUID": {
"Type": "strenum",
"Default": "FEXCore::Config::CPUID::OFF",
"Enums": {
"ENABLESHA": "enablesha",
"DISABLESHA": "disablesha"
},
"Desc": [
"Allows controlling of the CPU features are exposed in CPUID.",
"\toff: Default CPU features queried from CPU features",
"\t{enable,disable}sha: Will force enable or disable sha even if the host doesn't support it"
]
},
"SmallTSCScale": {
"Type": "bool",
"Default": "true",
"Desc": [
"Scales the cycle counter on systems that have low frequencies."
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it"
]
}
},
@@ -267,6 +247,23 @@
"Disables optimizations passes for debugging."
]
},
"SRA": {
"Type": "bool",
"Default": "true",
"Desc": [
"Set to false to disable Static Register Allocation"
]
},
"Force32BitAllocator": {
"Type": "bool",
"Default": "false",
"Desc": [
"Forces use of the 32-bit allocator on 32-bit applications",
"Used to work around ulimit problems of CI runner",
"Potentially useful for debugging memory problems",
"32-bit allocator is always used if your host kernel is older than 4.17"
]
},
"GlobalJITNaming": {
"Type": "bool",
"Default": "false",
@@ -365,14 +362,6 @@
"File to write FEX output to.",
"[stdout, stderr, server, <Filename>]"
]
},
"TelemetryDirectory": {
"Type": "str",
"Default": "",
"Desc": [
"Redirects the telemetry folder that FEX usually writes to.",
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
]
}
},
"Hacks": {
@@ -384,8 +373,9 @@
"Desc": [
"Checks code for modification before execution.",
"\tnone: No checks",
"\tmtrack: Page tracking based invalidation (default)",
"\tfull: Validate code before every run (slow)"
"\tmtrack: Page tracking based invalidation",
"\tfull: Validate code before every run (slow)",
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
]
},
"TSOEnabled": {
@@ -396,29 +386,6 @@
"Highly likely to break any multithreaded application if disabled."
]
},
"VectorTSOEnabled": {
"Type": "bool",
"Default": "false",
"Desc": [
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
]
},
"MemcpySetTSOEnabled": {
"Type": "bool",
"Default": "false",
"Desc": [
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
"Only affects REP MOVS and REP STOS instructions"
]
},
"HalfBarrierTSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"When TSO emulation is enabled, controls if unaligned loads and stores should be backpatched to half-barrier atomics.",
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
@@ -466,14 +433,6 @@
"Hides the hypervisor CPUID bit when set.",
"Should only be used for applications that have issues with this set."
]
},
"StartupSleep": {
"Type": "uint32",
"Default": "0",
"Desc": [
"Sleeps the process at startup for a duration of seconds.",
"Useful if an application crashes too quickly to attach a debugger."
]
}
},
"Misc": {
+52 -44
View File
@@ -12,62 +12,70 @@
#include <string.h>
#include <utility>
namespace FEXCore::HLE {
class SyscallVisitor;
}
namespace FEXCore::Context {
void InitializeStaticTables(OperatingMode Mode) {
X86Tables::InitializeInfoTables(Mode);
IR::InstallOpcodeHandlers(Mode);
}
void InitializeStaticTables(OperatingMode Mode) {
X86Tables::InitializeInfoTables(Mode);
IR::InstallOpcodeHandlers(Mode);
}
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext() {
return fextl::make_unique<FEXCore::Context::ContextImpl>();
}
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext() {
return fextl::make_unique<FEXCore::Context::ContextImpl>();
}
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
CustomExitHandler = std::move(handler);
}
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
CustomExitHandler = std::move(handler);
}
ExitHandler FEXCore::Context::ContextImpl::GetExitHandler() const {
return CustomExitHandler;
}
ExitHandler FEXCore::Context::ContextImpl::GetExitHandler() const {
return CustomExitHandler;
}
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
CompileBlock(Thread->CurrentFrame, GuestRIP);
}
void FEXCore::Context::ContextImpl::Stop() {
Stop(false);
}
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
}
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
CompileBlock(Thread->CurrentFrame, GuestRIP);
}
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
CustomCPUFactory = std::move(Factory);
}
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
}
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
return HostFeatures;
}
bool FEXCore::Context::ContextImpl::IsDone() const {
return IsPaused();
}
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator* _SignalDelegation) {
SignalDelegation = _SignalDelegation;
}
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
CustomCPUFactory = std::move(Factory);
}
void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) {
SyscallHandler = Handler;
SourcecodeResolver = Handler->GetSourcecodeResolver();
}
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
return HostFeatures;
}
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
return CPUID.RunFunction(Function, Leaf);
}
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator *_SignalDelegation) {
SignalDelegation = _SignalDelegation;
}
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
return CPUID.RunXCRFunction(Function);
}
void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) {
SyscallHandler = Handler;
SourcecodeResolver = Handler->GetSourcecodeResolver();
}
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
return CPUID.RunFunctionName(Function, Leaf, CPU);
}
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
return CPUID.RunFunction(Function, Leaf);
}
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const {
return Thread->CPUBackend->IsAddressInCodeBuffer(Address);
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
return CPUID.RunXCRFunction(Function);
}
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
return CPUID.RunFunctionName(Function, Leaf, CPU);
}
}
} // namespace FEXCore::Context
+347 -314
View File
@@ -13,7 +13,6 @@
#include <FEXCore/Core/HostFeatures.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
@@ -45,376 +44,410 @@ namespace CodeSerialize {
namespace CPU {
class Arm64JITCore;
class X86JITCore;
class Dispatcher;
} // namespace CPU
}
namespace HLE {
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
} // namespace HLE
} // namespace FEXCore
struct SyscallArguments;
class SyscallHandler;
class SourcecodeResolver;
struct SourcecodeMap;
}
}
namespace FEXCore::IR {
class RegisterAllocationData;
struct IRListCopy;
class IRListView;
class RegisterAllocationData;
class IRListView;
namespace Validation {
class IRValidation;
}
} // namespace FEXCore::IR
}
namespace FEXCore::Context {
enum CoreRunningMode {
MODE_RUN = 0,
MODE_SINGLESTEP = 1,
};
enum CoreRunningMode {
MODE_RUN = 0,
MODE_SINGLESTEP = 1,
};
struct ExitFunctionLinkData {
uint64_t HostBranch;
uint64_t GuestRIP;
};
class ContextImpl final : public FEXCore::Context::Context {
public:
// Context base class implementation.
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
constexpr uint32_t TSC_SCALE = 128;
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
void SetExitHandler(ExitHandler handler) override;
ExitHandler GetExitHandler() const override;
class ContextImpl final : public FEXCore::Context::Context {
public:
// Context base class implementation.
bool InitCore() override;
void Pause() override;
void Run() override;
void Stop() override;
void Step() override;
void SetExitHandler(ExitHandler handler) override;
ExitHandler GetExitHandler() const override;
ExitReason RunUntilExit() override;
ExitReason RunUntilExit(FEXCore::Core::InternalThreadState* Thread) override;
void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) override;
void ExecuteThread(FEXCore::Core::InternalThreadState* Thread) override;
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) override;
void CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
bool IsDone() const override;
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
HostFeatures GetHostFeatures() const override;
HostFeatures GetHostFeatures() const override;
void HandleCallback(FEXCore::Core::InternalThreadState* Thread, uint64_t RIP) override;
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) override;
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) override;
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) override;
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) override;
void ReconstructXMMRegisters(const FEXCore::Core::InternalThreadState* Thread, __uint128_t* XMM_Low, __uint128_t* YMM_High) override;
void SetXMMRegistersFromState(FEXCore::Core::InternalThreadState* Thread, const __uint128_t* XMM_Low, const __uint128_t* YMM_High) override;
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param InitialRIP The starting RIP of this thread
* @param StackPointer The starting RSP of this thread
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* Parent thread Creation:
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
* - CTX->RunUntilExit(Thread);
* OS thread Creation:
* - Thread = CreateThread(0, 0, NewState, PPID);
* - InitializeThread(Thread);
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(0, 0, NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
/**
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
*
* @param InitialRIP The starting RIP of this thread
* @param StackPointer The starting RSP of this thread
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
* @param ParentTID The PID that was the parent thread that created this
*
* @return The InternalThreadState object that tracks all of the emulated thread's state
*
* Usecases:
* Parent thread Creation:
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
* - CTX->RunUntilExit(Thread);
* OS thread Creation:
* - Thread = CreateThread(0, 0, NewState, PPID);
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
* - ThreadHandler calls `CTX->ExecutionThread(Thread)`
* OS fork (New thread created with a clone of thread state):
* - clone{2, 3}
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
* - ExecutionThread(Thread); // Starts executing without creating another host thread
* Thunk callback executing guest code from native host thread
* - Thread = CreateThread(0, 0, NewState, PPID);
* - InitializeThreadTLSData(Thread);
* - HandleCallback(Thread, RIP);
*/
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
FEXCore::Core::InternalThreadState*
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState* Thread) override;
/**
* @brief Destroys this FEX thread object and stops tracking it internally
*
* @param Thread The internal FEX thread state object
*/
void DestroyThread(FEXCore::Core::InternalThreadState* Thread, bool NeedsTLSUninstall) override;
// Public for threading
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Initializes the OS thread object and prepares to start executing on that new OS thread
*
* @param Thread The internal FEX thread state object
*
* The OS thread will wait until RunThread is executed
*/
void InitializeThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Starts the OS thread object to start executing guest code
*
* @param Thread The internal FEX thread state object
*/
void RunThread(FEXCore::Core::InternalThreadState *Thread) override;
void StopThread(FEXCore::Core::InternalThreadState *Thread) override;
/**
* @brief Destroys this FEX thread object and stops tracking it internally
*
* @param Thread The internal FEX thread state object
*/
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
#ifndef _WIN32
void LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) override;
void UnlockAfterFork(FEXCore::Core::InternalThreadState* Thread, bool Child) override;
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override;
#endif
void SetSignalDelegator(FEXCore::SignalDelegator* SignalDelegation) override;
void SetSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) override;
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) override;
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) override;
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
}
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
}
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
}
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
}
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
}
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
}
void FinalizeAOTIRCache() override {
IRCaptureCache.FinalizeAOTIRCache();
}
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void FinalizeAOTIRCache() override {
IRCaptureCache.FinalizeAOTIRCache();
}
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
IRCaptureCache.WriteFilesWithCode(Writer);
}
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
void MarkMemoryShared() override;
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
return CodeInvalidationMutex;
}
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
// returns false if a handler was already registered
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) override;
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
public:
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
#ifdef JIT_X86_64
friend class FEXCore::CPU::X86JITCore;
#endif
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
friend class FEXCore::IR::Validation::IRValidation;
// returns false if a handler was already registered
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator = nullptr, void* Data = nullptr);
struct {
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
uint64_t VirtualMemSize{1ULL << 36};
void AppendThunkDefinitions(const fextl::vector<FEXCore::IR::ThunkDefinition>& Definitions) override;
// this is for internal use
bool ValidateIRarser { false };
public:
friend class FEXCore::HLE::SyscallHandler;
#ifdef JIT_ARM64
friend class FEXCore::CPU::Arm64JITCore;
#endif
// Used if the JIT needs to have its interrupt fault code emitted.
bool NeedsPendingInterruptFaultCheck { false };
friend class FEXCore::IR::Validation::IRValidation;
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(Core, CORE);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
} Config;
struct {
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
uint64_t VirtualMemSize {1ULL << 36};
FEXCore::HostFeatures HostFeatures;
// Used if the JIT needs to have its interrupt fault code emitted.
bool NeedsPendingInterruptFaultCheck {false};
std::mutex ThreadCreationMutex;
FEXCore::Core::InternalThreadState* ParentThread{};
fextl::vector<FEXCore::Core::InternalThreadState*> Threads;
std::atomic_bool CoreShuttingDown{false};
bool NeedToCheckXID{true};
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
FEX_CONFIG_OPT(Core, CORE);
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
} Config;
std::mutex IdleWaitMutex;
std::condition_variable IdleWaitCV;
std::atomic<uint32_t> IdleWaitRefCount{};
Event PauseWait;
bool Running{};
std::atomic_bool CoreShuttingDown {false};
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
FEXCore::CPUIDEmu CPUID;
FEXCore::HLE::SyscallHandler *SyscallHandler{};
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
FEXCore::HostFeatures HostFeatures;
// CPUID depends on HostFeatures so needs to be initialized after that.
FEXCore::CPUIDEmu CPUID;
FEXCore::HLE::SyscallHandler* SyscallHandler {};
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CustomCPUFactoryType CustomCPUFactory;
FEXCore::Context::ExitHandler CustomExitHandler;
CustomCPUFactoryType CustomCPUFactory;
FEXCore::Context::ExitHandler CustomExitHandler;
#ifdef BLOCKSTATS
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
#endif
SignalDelegator* SignalDelegation {};
X86GeneratedCode X86CodeGen;
SignalDelegator *SignalDelegation{};
X86GeneratedCode X86CodeGen;
ContextImpl();
~ContextImpl();
ContextImpl();
~ContextImpl();
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestDestination,
FEXCore::Context::ExitFunctionLinkData* HostLink, const BlockDelinkerFunc& delinker);
bool IsPaused() const { return !Running; }
void WaitForThreadsToRun() override;
void Stop(bool IgnoreCurrentThread);
void WaitForIdle() override;
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, ExitFunctionLinkData* Record) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
return Fn(Frame, Record);
}
template<auto Fn>
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
// Must be called from owning thread
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
return Fn(Frame, record);
}
LOGMAN_THROW_A_FMT(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}",
Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
// Must be called from owning thread
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
auto Thread = Frame->Thread;
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
struct GenerateIRResult {
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
uint64_t TotalInstructions;
uint64_t TotalInstructionsLength;
uint64_t StartAddr;
uint64_t Length;
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
struct GenerateIRResult {
FEXCore::IR::IRListView* IRList;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
uint64_t TotalInstructions;
uint64_t TotalInstructionsLength;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
struct CompileCodeResult {
void* CompiledCode;
FEXCore::IR::IRListView* IRData;
FEXCore::Core::DebugData* DebugData;
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
bool GeneratedIR;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
// same as CompileBlock, but aborts on failure
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
// Used for thread creation from syscalls
/**
* @brief Initializes TID, PID and TLS data for a thread
*
* @param Thread The internal FEX thread state object
*/
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread);
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
FEXCore::JITSymbols Symbols;
void GetVDSOSigReturn(VDSOSigReturn *VDSOPointers) override {
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
}
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
}
}
void IncrementIdleRefCount() override {
++IdleWaitRefCount;
}
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
// If Atomic-based TSO emulation is enabled or not.
bool IsAtomicTSOEnabled() const { return AtomicTSOEmulationEnabled; }
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
SupportsHardwareTSO = HardwareTSOSupported;
UpdateAtomicTSOEmulationConfig();
}
// Returns if Software TSO emulation is required.
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
// This will still return true if on a single thread and TSO is currently disabled.
//
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
// we return consistent results.
//
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
bool SoftwareTSORequired() const {
if (SupportsHardwareTSO) return false;
return Config.TSOEnabled;
}
void EnableExitOnHLT() override { ExitOnHLT = true; }
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
ThreadsState GetThreads() override {
return ThreadsState {
.ParentThread = ParentThread,
.Threads = &Threads,
};
}
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
protected:
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
// If the hardware supports TSO then we don't need to emulate it through atomics.
AtomicTSOEmulationEnabled = false;
}
else {
// Atomic TSO emulation only enabled if the config option is enabled.
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
}
}
private:
/**
* @brief Initializes the JIT compilers for the thread
*
* @param State The internal FEX thread state object
*
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
void WaitForIdleWithTimeout();
void NotifyPause();
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
// Entry Cache
std::mutex ExitMutex;
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
bool StartPaused = false;
bool IsMemoryShared = false;
bool SupportsHardwareTSO = false;
bool AtomicTSOEmulationEnabled = true;
bool ExitOnHLT = false;
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
std::shared_mutex CustomIRMutex;
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void *, void *>> CustomIRHandlers;
FEXCore::CPU::DispatcherConfig DispatcherConfig;
};
[[nodiscard]]
GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
struct CompileCodeResult {
void* CompiledCode;
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
FEXCore::Core::DebugData* DebugData;
bool GeneratedIR;
uint64_t StartAddr;
uint64_t Length;
};
[[nodiscard]]
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
// Used for thread creation from syscalls
/**
* @brief Initializes TID, PID and TLS data for a thread
*
* @param Thread The internal FEX thread state object
*/
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState* Thread);
void CopyMemoryMapping(FEXCore::Core::InternalThreadState* ParentThread, FEXCore::Core::InternalThreadState* ChildThread);
uint8_t GetGPRSize() const {
return Config.Is64BitMode ? 8 : 4;
}
FEXCore::JITSymbols Symbols;
void GetVDSOSigReturn(VDSOSigReturn* VDSOPointers) override {
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
}
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
}
}
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
// If Atomic-based TSO emulation is enabled or not.
bool IsAtomicTSOEnabled() const {
return AtomicTSOEmulationEnabled;
}
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
SupportsHardwareTSO = HardwareTSOSupported;
UpdateAtomicTSOEmulationConfig();
}
// Returns if Software TSO emulation is required.
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
// This will still return true if on a single thread and TSO is currently disabled.
//
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
// we return consistent results.
//
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
bool SoftwareTSORequired() const {
if (SupportsHardwareTSO) {
return false;
}
return Config.TSOEnabled;
}
void EnableExitOnHLT() override {
ExitOnHLT = true;
}
bool ExitOnHLTEnabled() const {
return ExitOnHLT;
}
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
protected:
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
// If the hardware supports TSO then we don't need to emulate it through atomics.
AtomicTSOEmulationEnabled = false;
} else {
// Atomic TSO emulation only enabled if the config option is enabled.
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
}
}
private:
/**
* @brief Initializes the JIT compilers for the thread
*
* @param State The internal FEX thread state object
*
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* Ptr);
IR::AOTIRCaptureCache IRCaptureCache;
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
bool StartPaused = false;
bool IsMemoryShared = false;
bool SupportsHardwareTSO = false;
bool AtomicTSOEmulationEnabled = true;
bool ExitOnHLT = false;
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
std::shared_mutex CustomIRMutex;
std::atomic<bool> HasCustomIRHandlers {};
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void*, void*>> CustomIRHandlers;
};
} // namespace FEXCore::Context
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
}
File diff suppressed because it is too large. Load diff
@@ -2,6 +2,9 @@
#pragma once
#include "FEXCore/Utils/EnumUtils.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
#include "Interface/Core/ObjectCache/Relocations.h"
#include <aarch64/assembler-aarch64.h>
@@ -18,9 +21,6 @@
#endif
#include <FEXCore/Config/Config.h>
#include <FEXCore/fextl/vector.h>
#include <CodeEmitter/Emitter.h>
#include <CodeEmitter/Registers.h>
#include <array>
#include <cstddef>
@@ -34,74 +34,67 @@ class ContextImpl;
namespace FEXCore::CPU {
// Contains the address to the currently available CPU state
constexpr auto STATE = ARMEmitter::XReg::x28;
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
#ifndef _M_ARM_64EC
// GPR temporaries. Only x3 can be used across spill boundaries
// so if these ever need to change, be very careful about that.
constexpr auto TMP1 = ARMEmitter::XReg::x0;
constexpr auto TMP2 = ARMEmitter::XReg::x1;
constexpr auto TMP3 = ARMEmitter::XReg::x2;
constexpr auto TMP4 = ARMEmitter::XReg::x3;
constexpr bool TMP_ABIARGS = true;
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
constexpr auto REG_PF = ARMEmitter::Reg::r26;
constexpr auto REG_AF = ARMEmitter::Reg::r27;
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x0;
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x1;
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x2;
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x3;
// Vector temporaries
constexpr auto VTMP1 = ARMEmitter::VReg::v0;
constexpr auto VTMP2 = ARMEmitter::VReg::v1;
#else
constexpr auto TMP1 = ARMEmitter::XReg::x10;
constexpr auto TMP2 = ARMEmitter::XReg::x11;
constexpr auto TMP3 = ARMEmitter::XReg::x12;
constexpr auto TMP4 = ARMEmitter::XReg::x13;
constexpr bool TMP_ABIARGS = false;
// We pin r11/r12 as PF/AF respectively for arm64ec, as r26/r27 are used for SRA.
constexpr auto REG_PF = ARMEmitter::Reg::r9;
constexpr auto REG_AF = ARMEmitter::Reg::r24;
// Vector temporaries
constexpr auto VTMP1 = ARMEmitter::VReg::v16;
constexpr auto VTMP2 = ARMEmitter::VReg::v17;
// Entry/Exit ABI
constexpr auto EC_CALL_CHECKER_PC_REG = ARMEmitter::XReg::x9;
constexpr auto EC_ENTRY_CPUAREA_REG = ARMEmitter::XReg::x17;
#endif
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v0;
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
// Predicate register temporaries (used when AVX support is enabled)
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
constexpr ARMEmitter::PRegister PRED_TMP_16B = ARMEmitter::PReg::p6;
constexpr ARMEmitter::PRegister PRED_TMP_32B = ARMEmitter::PReg::p7;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
// This class contains common emitter utility functions that can
// be used by both Arm64 JIT and ARM64 Dispatcher
class Arm64Emitter : public ARMEmitter::Emitter {
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
protected:
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr = nullptr, size_t size = 0);
FEXCore::Context::ContextImpl* EmitterCTX;
FEXCore::Context::ContextImpl *EmitterCTX;
vixl::aarch64::CPU CPU;
std::span<const ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
std::span<const ARMEmitter::Register> StaticRegisters {};
std::span<const ARMEmitter::Register> GeneralRegisters {};
std::span<const ARMEmitter::VRegister> StaticFPRegisters {};
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
uint32_t PairRegisters = 0;
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase{};
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters{};
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters{};
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters{};
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters{};
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters{};
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
/**
* @name Register Allocation
* @{ */
constexpr static uint32_t RegisterClasses = 6;
constexpr static uint64_t GPRBase = (0ULL << 32);
constexpr static uint64_t FPRBase = (1ULL << 32);
constexpr static uint64_t GPRPairBase = (2ULL << 32);
/** @} */
constexpr static uint8_t RA_32 = 0;
constexpr static uint8_t RA_64 = 1;
constexpr static uint8_t RA_FPR = 2;
void LoadConstant(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
// TMP4 is left alone.
void SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
void SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
// Register 0-18 + 29 + 30 are caller saved
@@ -112,13 +105,13 @@ protected:
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
// Generic push and pop vector registers.
void PushVectorRegisters(ARMEmitter::Register TmpReg, bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs);
void PushGeneralRegisters(ARMEmitter::Register TmpReg, std::span<const ARMEmitter::Register> Regs);
void PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs);
void PushGeneralRegisters(FEXCore::ARMEmitter::Register TmpReg, std::span<const FEXCore::ARMEmitter::Register> Regs);
void PopVectorRegisters(bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs);
void PopGeneralRegisters(std::span<const ARMEmitter::Register> Regs);
void PopVectorRegisters(bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs);
void PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Register> Regs);
void PushDynamicRegsAndLR(ARMEmitter::Register TmpReg);
void PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg);
void PopDynamicRegsAndLR();
void PushCalleeSavedRegisters();
@@ -134,13 +127,14 @@ protected:
// Callee Saved:
// - X9-X15, X19-X31
// - Low 128-bits of v8-v31
void SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs = true);
void SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true);
void FillForPreserveAllABICall(bool FPRs = true);
void SpillForABICall(bool SupportsPreserveAllABI, ARMEmitter::Register TmpReg, bool FPRs = true) {
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
if (SupportsPreserveAllABI) {
SpillForPreserveAllABICall(TmpReg, FPRs);
} else {
}
else {
SpillStaticRegs(TmpReg, FPRs);
PushDynamicRegsAndLR(TmpReg);
}
@@ -149,7 +143,8 @@ protected:
void FillForABICall(bool SupportsPreserveAllABI, bool FPRs = true) {
if (SupportsPreserveAllABI) {
FillForPreserveAllABICall(FPRs);
} else {
}
else {
PopDynamicRegsAndLR();
FillStaticRegs(FPRs);
}
@@ -171,7 +166,8 @@ protected:
template<typename R, typename... P>
void GenerateRuntimeCall(R (*Function)(P...)) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
uintptr_t FunctionAddress = reinterpret_cast<uintptr_t>(Function);
@@ -189,7 +185,8 @@ protected:
template<typename R, typename... P>
void GenerateIndirectRuntimeCall(ARMEmitter::Register Reg) {
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
@@ -205,8 +202,8 @@ protected:
template<>
void GenerateIndirectRuntimeCall<float, __uint128_t>(ARMEmitter::Register Reg) {
uintptr_t SimulatorWrapperAddress =
reinterpret_cast<uintptr_t>(&(vixl::aarch64::Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
@@ -233,17 +230,15 @@ protected:
#ifdef VIXL_SIMULATOR
vixl::aarch64::Decoder SimDecoder;
vixl::aarch64::Simulator Simulator;
constexpr static size_t SimulatorStackSize = 8 * 1024 * 1024;
#endif
#ifdef VIXL_DISASSEMBLER
fextl::vector<char> DisasmBuffer;
constexpr static int DISASM_BUFFER_SIZE {256};
fextl::unique_ptr<vixl::aarch64::Disassembler> Disasm;
vixl::aarch64::Disassembler Disasm;
fextl::unique_ptr<vixl::aarch64::Decoder> DisasmDecoder;
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
#endif
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
};
} // namespace FEXCore::CPU
}
File diff suppressed because it is too large. Load diff
@@ -8,25 +8,23 @@ public:
public:
// Conditional branch immediate
///< Branch conditional
void b(ARMEmitter::Condition Cond, uint32_t Imm) {
void b(FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm);
}
void b(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
void b(FEXCore::ARMEmitter::Condition Cond, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void b(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
}
void b(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
void b(FEXCore::ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
b(Cond, &Label->Backward);
}
@@ -36,26 +34,24 @@ public:
}
///< Branch consistent conditional
void bc(ARMEmitter::Condition Cond, uint32_t Imm) {
void bc(FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm);
}
void bc(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
void bc(FEXCore::ARMEmitter::Condition Cond, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bc(ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void bc(FEXCore::ARMEmitter::Condition Cond, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
}
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
void bc(FEXCore::ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
bc(Cond, &Label->Backward);
}
@@ -65,7 +61,7 @@ public:
}
// Unconditional branch register
void br(ARMEmitter::Register rn) {
void br(FEXCore::ARMEmitter::Register rn) {
constexpr uint32_t Op = 0b1101011 << 25 |
0b0'000 << 21 | // opc
0b1'1111 << 16 | // op2
@@ -74,7 +70,7 @@ public:
UnconditionalBranch(Op, rn);
}
void blr(ARMEmitter::Register rn) {
void blr(FEXCore::ARMEmitter::Register rn) {
constexpr uint32_t Op = 0b1101011 << 25 |
0b0'001 << 21 | // opc
0b1'1111 << 16 | // op2
@@ -83,7 +79,7 @@ public:
UnconditionalBranch(Op, rn);
}
void ret(ARMEmitter::Register rn = ARMEmitter::Reg::r30) {
void ret(FEXCore::ARMEmitter::Register rn = FEXCore::ARMEmitter::Reg::r30) {
constexpr uint32_t Op = 0b1101011 << 25 |
0b0'010 << 21 | // opc
0b1'1111 << 16 | // op2
@@ -106,10 +102,8 @@ public:
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
void b(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
@@ -137,10 +131,8 @@ public:
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bl(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
void bl(ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::B });
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
@@ -156,13 +148,13 @@ public:
}
// Compare and branch
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, Imm);
}
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
@@ -171,17 +163,15 @@ public:
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, 0);
}
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
cbz(s, rt, &Label->Backward);
}
@@ -190,13 +180,13 @@ public:
}
}
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, Imm);
}
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
@@ -205,17 +195,15 @@ public:
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::BC });
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, 0);
}
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
cbnz(s, rt, &Label->Backward);
}
@@ -225,12 +213,12 @@ public:
}
// Test and branch immediate
void tbz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, Imm);
}
void tbz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
@@ -238,18 +226,15 @@ public:
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, 0);
}
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
tbz(rt, Bit, &Label->Backward);
}
@@ -258,12 +243,12 @@ public:
}
}
void tbnz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, Imm);
}
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
@@ -271,17 +256,14 @@ public:
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbnz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, ForwardLabel *Label) {
Label->Insts.emplace_back(ForwardLabel::Instructions{ .Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::Instructions::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
}
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
tbnz(rt, Bit, &Label->Backward);
}
@@ -292,7 +274,7 @@ public:
private:
// Conditional branch immediate
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, ARMEmitter::Condition Cond, uint32_t Imm) {
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
uint32_t Instr = Op;
Instr |= Op1 << 24;
@@ -304,7 +286,7 @@ private:
}
// Unconditional branch register
void UnconditionalBranch(uint32_t Op, ARMEmitter::Register rn) {
void UnconditionalBranch(uint32_t Op, FEXCore::ARMEmitter::Register rn) {
uint32_t Instr = Op;
Instr |= Encode_rn(rn);
dc32(Instr);
@@ -318,8 +300,8 @@ private:
}
// Compare and branch
void CompareAndBranch(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
void CompareAndBranch(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
@@ -330,7 +312,7 @@ private:
}
// Test and branch - immediate
void TestAndBranch(uint32_t Op, ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
void TestAndBranch(uint32_t Op, FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
uint32_t Instr = Op;
Instr |= (Bit >> 5) << 31;
@@ -0,0 +1,106 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <cstddef>
#include <cstdint>
#include <cstring>
namespace FEXCore::ARMEmitter {
class Buffer {
public:
Buffer() {
SetBuffer(nullptr, 0);
}
Buffer(uint8_t* Base, uint64_t BaseSize) {
SetBuffer(Base, BaseSize);
}
void SetBuffer(uint8_t* Base, uint64_t BaseSize) {
BufferBase = Base;
CurrentOffset = BufferBase;
Size = BaseSize;
}
void dc8(uint8_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc16(uint16_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc32(uint32_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc64(uint64_t Data) {
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void EmitString(const char *String) {
const auto StringLength = strlen(String);
memcpy(CurrentOffset, String, StringLength);
CurrentOffset += StringLength;
}
void Align() {
// Align the buffer to instruction size
auto CurrentAlignment = reinterpret_cast<uint64_t>(CurrentOffset) & 0b11;
if (!CurrentAlignment) {
return;
}
CurrentOffset += 4 - CurrentAlignment;
}
template<typename T>
T GetCursorAddress() const {
return reinterpret_cast<T>(CurrentOffset);
}
static void ClearICache(void* Begin, std::size_t Length) {
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
}
size_t GetCursorOffset() const {
return static_cast<size_t>(CurrentOffset - BufferBase);
}
uint8_t *GetBufferBase() const {
return BufferBase;
}
void CursorIncrement(size_t Size) {
CurrentOffset += Size;
}
void SetCursorOffset(size_t Offset) {
CurrentOffset = BufferBase + Offset;
}
uint64_t GetBufferSize() const {
return Size;
}
template<typename T>
size_t GetCursorOffsetFromAddress(const T* Address) const {
return static_cast<size_t>(reinterpret_cast<const uint8_t*>(Address) - BufferBase);
}
protected:
void ResetBuffer() {
CurrentOffset = BufferBase;
}
uint8_t* BufferBase;
uint8_t* CurrentOffset;
uint64_t Size;
};
}
@@ -0,0 +1,834 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Interface/Core/ArchHelpers/CodeEmitter/Buffer.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
#include <FEXCore/Utils/BitUtils.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/fextl/vector.h>
#include <aarch64/assembler-aarch64.h>
#include <array>
#include <cstdint>
#include <utility>
#include <type_traits>
/*
* Welcome to FEX-Emu's custom AArch64 emitter.
* This was written specifically to avoid the performance cost of the vixl emitter.
*
* There are some specific design constraints in this design to target a couple features:
* - High performance
* - Low CPU cache performance hit
* - Significantly reduced code footprint
* - Low number of branches
*
* These requirements are mostly achieved by removing a bunch of developer conveniences
* that vixl provides. The developer needs to take a lot of care to not shoot themselves in the foot.
*
* Misc design decisions:
* - Registers are encoded as basic uint32_t enums.
* - Converting between different registers is zero-cost.
* - Passing around as arguments are as cheap as registers
* - Contrast to vixl where every register requires living on the stack.
* - Registers can get encoded in to instructions with a simple `BFM` instruction.
*
* - Instructions are very simply emitted, allowing direct inlining most of the time.
* - These are simple enough that multiple back-to-back instructions get optimized to 128-bit load-store operations.
* - Contrast to vixl where pretty much no instruction emitter gets inlined.
*
* - Instruction emitters are /mostly/ unsized. Most instructions take a size argument first, which gets encoded
* directly in to the instruction.
* - Contrast to vixl where the register arguments are how the instructions determine operating size.
* - Size argument allows FEX to use `CSEL` to select a size at runtime, instead of branching.
* - Some instructions are explicitly sized based on register type. Read comments in the respective `inl` files to
* see why.
* Some scalar/vector operations are an example of this.
*
* - Almost zero helper functions.
* - Primary exception to this rule is load-store operations. These will use a helper to make
* it easier to select the correct load-store instruction. Mostly because these are a nightmare selecting
* the right instruction.
*/
namespace FEXCore::ARMEmitter {
/*
* This `Size` enum is used for most ALU operations.
* These follow the AArch64 encoding style in most cases.
*/
enum class Size : uint32_t {
i32Bit = 0,
i64Bit,
};
// This allows us to get the `Size` enum in bits.
[[nodiscard]]
constexpr size_t RegSizeInBits(Size size) {
return size_t{32} << FEXCore::ToUnderlying(size);
}
/* This `SubRegSize` enum is used for most ASIMD operations.
* These follow the AArch64 encoding style in most cases.
*/
enum class SubRegSize : uint32_t {
i8Bit = 0b00,
i16Bit = 0b01,
i32Bit = 0b10,
i64Bit = 0b11,
i128Bit = 0b100,
};
// This allows us to get the `SubRegSize` in bits.
[[nodiscard]]
constexpr size_t SubRegSizeInBits(SubRegSize size) {
return size_t{8} << FEXCore::ToUnderlying(size);
}
/* This `ScalarRegSize` enum is used for most scalar float
* operations.
*
* This is specifically duplicated from `SubRegSize` to have strongly
* typed functions.
*
* `ScalarRegSize` specifically doesn't have `i128Bit` because scalar operations
* can't operate at 128-bit.
*/
enum class ScalarRegSize : uint32_t {
i8Bit = 0b00,
i16Bit = 0b01,
i32Bit = 0b10,
i64Bit = 0b11,
};
// This allows us to get the `ScalarRegSize` in bits.
[[nodiscard]]
constexpr size_t ScalarRegSizeInBits(ScalarRegSize size) {
return size_t{8} << FEXCore::ToUnderlying(size);
}
/* This `VectorRegSizePair` union allows us to have an overlapping type
* to select a scalar operation or a vector depending on which operation
* we pass in.
* Useful in FEX's vector operations that behave as scalar or vector
* depending on various factors. But since the operation will have the sa,e
* element size, we want to choose the operation more easily
*/
union VectorRegSizePair {
ScalarRegSize Scalar;
SubRegSize Vector;
};
// This allows us to create a `VectorRegSizePair` union.
[[nodiscard]]
constexpr VectorRegSizePair ToVectorSizePair(SubRegSize size) {
return VectorRegSizePair {.Vector = size};
}
[[nodiscard]]
constexpr VectorRegSizePair ToVectorSizePair(ScalarRegSize size) {
return VectorRegSizePair {.Scalar = size};
}
// This `ShiftType` enum is used for ALU shift-register encoded instructions.
enum class ShiftType : uint32_t {
LSL = 0,
LSR,
ASR,
ROR,
};
// This `ExtendedType` enum is used for ALU extended-register encoded instructions.
enum class ExtendedType : uint32_t {
UXTB = 0b000,
UXTH = 0b001,
UXTW = 0b010,
UXTX = 0b011,
SXTB = 0b100,
SXTH = 0b101,
SXTW = 0b110,
SXTX = 0b111,
LSL_32 = UXTW,
LSL_64 = UXTX,
};
// This `Condition` enum is used for various conditional instructions.
enum class Condition : uint32_t {
// Meaning: Int - Float
CC_EQ = 0, // Equal - Equal
CC_NE, // Not Eq - Not Eq or unordered
CC_CS, // Carry set - Greater than, equal, or unordered
CC_CC, // Carry clear - Less than
CC_MI, // Minus/Negative - Less than
CC_PL, // Plus, positive or zero - GT, equal, or unordered
CC_VS, // Overflow - Unordered
CC_VC, // No Overflow - Ordered
CC_HI, // Unsigned higher - GT, or unordered
CC_LS, // Unsigned lower or same - LT or EQ
CC_GE, // Signed GT or EQ - GT or EQ
CC_LT, // Signed LT - LT or Unordered
CC_GT, // Signed GT - GT
CC_LE, // Signed LT or EQ - LT, EQ, or Unordered
CC_AL, // Always - Always
CC_NV, // Always - Always
// Aliases
CC_HS = CC_CS,
CC_LO = CC_CC,
};
/*
* This `StatusFlags` enum is used for conditional compare encoded instructions.
* These directly encode to the `nzcv` flags.
*/
enum class StatusFlags : uint32_t {
None = 0,
Flag_V = 0b0001,
Flag_C = 0b0010,
Flag_Z = 0b0100,
Flag_N = 0b1000,
Flag_NZCV = Flag_N | Flag_Z | Flag_C | Flag_V,
};
/*
* This `IndexType` enum is used for load-store instructions.
* Not all load-store instructions use this, so the user needs to be careful.
*/
enum class IndexType {
POST,
OFFSET,
PRE,
UNPRIVILEGED,
};
// Used with adr and scalar + vector load/store variants to denote
// a modifier operation.
enum class SVEModType : uint8_t {
MOD_UXTW,
MOD_SXTW,
MOD_LSL,
MOD_NONE,
};
/* This `SVEMemOperand` class is used for the helper SVE load-store instructions.
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
*/
class SVEMemOperand final {
public:
enum class Type {
ScalarPlusScalar,
ScalarPlusImm,
ScalarPlusVector,
VectorPlusImm,
};
SVEMemOperand(XRegister rn, XRegister rm = XReg::zr)
: rn {rn}
, MemType{Type::ScalarPlusScalar}
, MetaType {
.ScalarScalarType {
.rm = rm,
}
} {}
SVEMemOperand(XRegister rn, int32_t imm = 0)
: rn {rn}
, MemType{Type::ScalarPlusImm}
, MetaType {
.ScalarImmType {
.Imm = imm,
}
} {}
SVEMemOperand(XRegister rn, ZRegister zm, SVEModType mod = SVEModType::MOD_NONE, uint8_t scale = 0)
: rn{rn}
, MemType{Type::ScalarPlusVector}
, MetaType {
.ScalarVectorType {
.zm = zm,
.mod = mod,
.scale = scale,
}
} {}
SVEMemOperand(ZRegister zn, uint32_t imm)
: rn{Register{zn.Idx()}}
, MemType{Type::VectorPlusImm}
, MetaType {
.VectorImmType{
.Imm = imm,
}
} {}
[[nodiscard]] bool IsScalarPlusScalar() const {
return MemType == Type::ScalarPlusScalar;
}
[[nodiscard]] bool IsScalarPlusImm() const {
return MemType == Type::ScalarPlusImm;
}
[[nodiscard]] bool IsScalarPlusVector() const {
return MemType == Type::ScalarPlusVector;
}
[[nodiscard]] bool IsVectorPlusImm() const {
return MemType == Type::VectorPlusImm;
}
union Data {
struct {
Register rm;
} ScalarScalarType;
struct {
int32_t Imm;
} ScalarImmType;
struct {
ZRegister zm;
SVEModType mod;
uint8_t scale;
} ScalarVectorType;
struct {
// rn will be a ZRegister
uint32_t Imm;
} VectorImmType;
};
Register rn;
Type MemType;
Data MetaType;
};
/* This `ExtendedMemOperand` class is used for the helper load-store instructions.
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
*/
class ExtendedMemOperand final {
public:
ExtendedMemOperand(XRegister rn, XRegister rm = XReg::zr, ExtendedType Option = ExtendedType::LSL_64, uint32_t Shift = 0)
: rn {rn}
, MetaType {
.ExtendedType {
.Header = { .MemType = TYPE_EXTENDED },
.rm = rm,
.Option = Option,
.Shift = Shift,
}
} {}
ExtendedMemOperand(XRegister rn, IndexType Index = IndexType::OFFSET, int32_t Imm = 0)
: rn {rn}
, MetaType {
.ImmType {
.Header = { .MemType = TYPE_IMM },
.Index = Index,
.Imm = Imm,
}
} {}
Register rn;
enum Type {
TYPE_EXTENDED,
TYPE_IMM,
};
struct HeaderStruct {
Type MemType;
};
union {
HeaderStruct Header;
struct {
HeaderStruct Header;
Register rm;
ExtendedType Option;
uint32_t Shift;
} ExtendedType;
struct {
HeaderStruct Header;
IndexType Index;
int32_t Imm;
} ImmType;
} MetaType;
};
template<uint32_t op0, uint32_t op1, uint32_t CRn, uint32_t CRm, uint32_t op2>
constexpr uint32_t GenSystemReg() {
return op0 << 19 |
op1 << 16 |
CRn << 12 |
CRm << 8 |
op2 << 5;
};
// This `SystemRegister` enum is used for the mrs/msr instructions.
enum class SystemRegister : uint32_t {
CTR_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b001>(),
DCZID_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b111>(),
TPIDR_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b010>(),
RNDR = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b000>(),
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>(),
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>(),
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>(),
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>(),
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>(),
};
template<uint32_t op1, uint32_t CRm, uint32_t op2>
constexpr uint32_t GenDCReg() {
return op1 << 16 |
CRm << 8 |
op2 << 5;
};
// This `DataCacheOperation` enum is used for the dc instruction.
enum class DataCacheOperation : uint32_t {
IVAC = GenDCReg<0b000, 0b0110, 0b001>(),
ISW = GenDCReg<0b000, 0b0110, 0b010>(),
CSW = GenDCReg<0b000, 0b1010, 0b010>(),
CISW = GenDCReg<0b000, 0b1110, 0b010>(),
ZVA = GenDCReg<0b011, 0b0100, 0b001>(),
CVAC = GenDCReg<0b011, 0b1010, 0b001>(),
CVAU = GenDCReg<0b011, 0b1011, 0b001>(),
CIVAC = GenDCReg<0b011, 0b1110, 0b001>(),
// MTE2
IGVAC = GenDCReg<0b000, 0b0110, 0b011>(),
IGSW = GenDCReg<0b000, 0b0110, 0b100>(),
IGDVAC = GenDCReg<0b000, 0b0110, 0b101>(),
IGDSW = GenDCReg<0b000, 0b0110, 0b110>(),
CGSW = GenDCReg<0b000, 0b1010, 0b100>(),
CGDSW = GenDCReg<0b000, 0b1010, 0b110>(),
CIGSW = GenDCReg<0b000, 0b1110, 0b100>(),
CIGDSW = GenDCReg<0b000, 0b1110, 0b110>(),
// MTE
GVA = GenDCReg<0b011, 0b0100, 0b011>(),
GZVA = GenDCReg<0b011, 0b0100, 0b100>(),
CGVAC = GenDCReg<0b011, 0b1010, 0b011>(),
CGDVAC = GenDCReg<0b011, 0b1010, 0b101>(),
CGVAP = GenDCReg<0b011, 0b1100, 0b011>(),
CGDVAP = GenDCReg<0b011, 0b1100, 0b101>(),
CGVADP = GenDCReg<0b011, 0b1101, 0b011>(),
CGDVADP = GenDCReg<0b011, 0b1101, 0b101>(),
CIGVAC = GenDCReg<0b011, 0b1110, 0b011>(),
CIGDVAC = GenDCReg<0b011, 0b1110, 0b101>(),
// DPB
CVAP = GenDCReg<0b011, 0b1100, 0b001>(),
// DPB2
CVADP = GenDCReg<0b011, 0b1101, 0b001>(),
};
template<uint32_t CRm, uint32_t op2>
constexpr uint32_t GenHintBarrierReg() {
return CRm << 8 |
op2 << 5;
}
// This `HintRegister` enum is used for the hint instruction.
enum class HintRegister : uint32_t {
NOP = GenHintBarrierReg<0b0000, 0b000>(),
YIELD = GenHintBarrierReg<0b0000, 0b001>(),
WFE = GenHintBarrierReg<0b0000, 0b010>(),
WFI = GenHintBarrierReg<0b0000, 0b011>(),
SEV = GenHintBarrierReg<0b0000, 0b100>(),
SEVL = GenHintBarrierReg<0b0000, 0b101>(),
DGH = GenHintBarrierReg<0b0000, 0b110>(),
CSDB = GenHintBarrierReg<0b0010, 0b100>(),
};
// This `BarrierRegister` enum is used for the various barrier instructions.
enum class BarrierRegister : uint32_t {
CLREX = GenHintBarrierReg<0b0000, 0b010>(),
TCOMMIT = GenHintBarrierReg<0b0000, 0b011>(),
DSB = GenHintBarrierReg<0b0000, 0b100>(),
DMB = GenHintBarrierReg<0b0000, 0b101>(),
ISB = GenHintBarrierReg<0b0000, 0b110>(),
SB = GenHintBarrierReg<0b0000, 0b111>(),
};
// This `BarrierScope` enum is used for the dsb/dmb instructions.
enum class BarrierScope : uint32_t {
// Outer shareable
OSHLD = 0b0001,
OSHST = 0b0010,
OSH = 0b0011,
// Non shareable
NSHLD = 0b0101,
NSHST = 0b0110,
NSH = 0b0111,
// Inner shareable
ISHLD = 0b1001,
ISHST = 0b1010,
ISH = 0b1011,
// Full System visibility
LD = 0b1101,
ST = 0b1110,
SY = 0b1111,
};
// This `Prefetch` enum is used for prefetch instructions.
enum class Prefetch : uint32_t {
// Prefetch for load
PLDL1KEEP = 0b00000,
PLDL1STRM = 0b00001,
PLDL2KEEP = 0b00010,
PLDL2STRM = 0b00011,
PLDL3KEEP = 0b00100,
PLDL3STRM = 0b00101,
// Preload instructions
PLIL1KEEP = 0b01000,
PLIL1STRM = 0b01001,
PLIL2KEEP = 0b01010,
PLIL2STRM = 0b01011,
PLIL3KEEP = 0b01100,
PLIL3STRM = 0b01101,
// Preload for store
PSTL1KEEP = 0b10000,
PSTL1STRM = 0b10001,
PSTL2KEEP = 0b10010,
PSTL2STRM = 0b10011,
PSTL3KEEP = 0b10100,
PSTL3STRM = 0b10101,
};
// This `PredicatePattern` enun is used for some SVE instructions.
enum class PredicatePattern : uint32_t {
SVE_POW2 = 0b00000,
SVE_VL1 = 0b00001,
SVE_VL2 = 0b00010,
SVE_VL3 = 0b00011,
SVE_VL4 = 0b00100,
SVE_VL5 = 0b00101,
SVE_VL6 = 0b00110,
SVE_VL7 = 0b00111,
SVE_VL8 = 0b01000,
SVE_VL16 = 0b01001,
SVE_VL32 = 0b01010,
SVE_VL64 = 0b01011,
SVE_VL128 = 0b01100,
SVE_VL256 = 0b01101,
SVE_MUL4 = 0b11101,
SVE_MUL3 = 0b11110,
SVE_ALL = 0b11111,
};
// Used with SVE FP immediate arithmetic instructions
enum class SVEFAddSubImm : uint32_t {
_0_5,
_1_0,
};
enum class SVEFMulImm : uint32_t {
_0_5,
_2_0,
};
enum class SVEFMaxMinImm : uint32_t {
_0_0,
_1_0,
};
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is logically `below` an instruction that uses it.
* Which means that a branch would jump backwards.
*/
struct BackwardLabel {
uint8_t *Location{};
};
/* This `ForwardLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is logically `above` an instruction that uses it.
* Which means that a branch would jump forwards.
*
* This can be bound to multiple instructions, so it needs a vector for each bind instruction type.
*/
struct ForwardLabel {
struct Instructions {
enum class InstType {
ADR,
ADRP,
B,
BC,
TEST_BRANCH,
RELATIVE_LOAD,
LONG_ADDRESS_GEN,
};
uint8_t *Location{};
InstType Type;
};
fextl::vector<Instructions> Insts{};
};
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is in either direction of an instruction that uses it.
* Which means a branch could jump backwards or forwards depending on situation.
*/
struct BiDirectionalLabel {
BackwardLabel Backward;
ForwardLabel Forward;
};
// Some FCMA ASIMD instructions support a rotation argument.
enum class Rotation : uint32_t {
ROTATE_0 = 0b00,
ROTATE_90 = 0b01,
ROTATE_180 = 0b10,
ROTATE_270 = 0b11,
};
// Concept for contraining some instructions to accept only an XRegister or WRegister.
// Particularly for operations that differ encodings depending on which one is used.
template <typename T>
concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegister>;
// Whether or not a given set of vector registers are sequential
// in increasing order as far as the register file is concerned (modulo its size)
//
// For example, a set of registers like:
//
// v1, v2, v3 and
// v31, v0, v1
//
// would both be considered sequential sequences, and some instructions in particular
// limit register lists to these kind of sequences.
//
template <typename T, typename... Args>
constexpr bool AreVectorsSequential(T first, const Args&... args) {
// Ensure we always have a pair of registers to compare against.
static_assert(sizeof...(args) >= 1, "Number of arguments must be greater than 1");
const auto fn = [](auto& lhs, const auto& rhs) {
const auto result = ((lhs.Idx() + 1) % 32) == rhs.Idx();
lhs = rhs;
return result;
};
return (fn(first, args) && ...);
}
// This is an emitter that is designed around the smallest code bloat as possible.
// Eschewing most developer convenience in order to keep code as small as possible.
// Choices:
// - Size of ops passed as an argument rather than template to let the compiler use csel instead of branching.
// - Registers are unsized so they can be passed in a GPR and not need conversion operations
class Emitter : public FEXCore::ARMEmitter::Buffer {
public:
Emitter() = default;
Emitter(uint8_t* Base, uint64_t BaseSize)
: Buffer (Base, BaseSize) {
}
// Bind a backward label to an address.
// Address that is bound is the current emitter location.
void Bind(BackwardLabel *Label) {
LOGMAN_THROW_AA_FMT(Label->Location == nullptr, "Trying to bind a label twice");
Label->Location = GetCursorAddress<uint8_t*>();
}
// Bind a forward label to a location.
// This walks all the instructions in the label's vector.
// Then backpatching all instructions that have used the label.
template<bool WarnAboutEmpty = false>
void Bind(ForwardLabel *Label) {
if constexpr (WarnAboutEmpty) {
LOGMAN_THROW_A_FMT(Label->Insts.empty() == false, "Binding forward label that didn't have any instructions using it");
}
uint8_t *CurrentAddress = GetCursorAddress<uint8_t*>();
for (const auto &Inst : Label->Insts) {
// Patch up the instructions
switch (Inst.Type) {
case ForwardLabel::Instructions::InstType::ADR: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::ADRP: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::B: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= Offset;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::TEST_BRANCH: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::BC:
case ForwardLabel::Instructions::InstType::RELATIVE_LOAD: {
uint32_t *Instruction = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case ForwardLabel::Instructions::InstType::LONG_ADDRESS_GEN: {
uint32_t *Instructions = reinterpret_cast<uint32_t*>(Inst.Location);
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
auto OriginalOffset = GetCursorOffset();
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
SetCursorOffset(InstOffset);
// We encoded the destination register in to the first instruction space.
// Read it back.
ARMEmitter::Register DestReg(Instructions[0]);
if (IsADRRange(ImmInstTwo)) {
// If within ADR range from the second instruction, then we can emit NOP+ADR
nop();
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
}
else if (IsADRPRange(ImmInstOne)) {
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
// First check if we are in non-offset range for second instruction.
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
// We can emit nop + adrp
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
}
else {
// Not aligned, need adrp + add
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
}
}
else {
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
FEX_UNREACHABLE;
}
SetCursorOffset(OriginalOffset);
break;
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
}
}
// Bind a bidirectional location to a location.
// Binds both forwards and backwards depending on how the label was used.
void Bind(BiDirectionalLabel *Label) {
if (!Label->Backward.Location) {
Bind(&Label->Backward);
}
Bind<false>(&Label->Forward);
}
public:
// TODO: Implement SME when it matters.
#include "Interface/Core/ArchHelpers/CodeEmitter/ALUOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/BranchOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/LoadstoreOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/SystemOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/ScalarOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/ASIMDOps.inl"
#include "Interface/Core/ArchHelpers/CodeEmitter/SVEOps.inl"
private:
template<typename T>
uint32_t Encode_ra(T Reg) const {
return Reg.Idx() << 10;
}
uint32_t Encode_ra(uint32_t Reg) const {
return Reg << 10;
}
template<typename T>
uint32_t Encode_rt2(T Reg) const {
return Reg.Idx() << 10;
}
template<>
uint32_t Encode_rt2(uint32_t Reg) const {
return Reg << 10;
}
template<typename T>
uint32_t Encode_rm(T Reg) const {
return Reg.Idx() << 16;
}
uint32_t Encode_rm(uint32_t Reg) const {
return Reg << 16;
}
template<typename T>
uint32_t Encode_rs(T Reg) const {
return Reg.Idx() << 16;
}
uint32_t Encode_rs(uint32_t Reg) const {
return Reg << 16;
}
template<typename T>
uint32_t Encode_rn(T Reg) const {
return Reg.Idx() << 5;
}
uint32_t Encode_rn(uint32_t Reg) const {
return Reg << 5;
}
template<typename T>
uint32_t Encode_rd(T Reg) const {
return Reg.Idx();
}
uint32_t Encode_rd(uint32_t Reg) const {
return Reg;
}
template<typename T>
uint32_t Encode_rt(T Reg) const {
return Reg.Idx();
}
template<>
uint32_t Encode_rt(Prefetch Reg) const {
return FEXCore::ToUnderlying(Reg);
}
uint32_t Encode_rt(uint32_t Reg) const {
return Reg;
}
template<typename T>
uint32_t Encode_pd(T Reg) const {
return FEXCore::ToUnderlying(Reg);
}
};
}
File diff suppressed because it is too large. Load diff
@@ -1506,14 +1506,13 @@ public:
}
// SVE broadcast floating-point immediate (unpredicated)
void fdup(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, float Value) {
LOGMAN_THROW_AA_FMT(size == ARMEmitter::SubRegSize::i16Bit ||
size == ARMEmitter::SubRegSize::i32Bit ||
size == ARMEmitter::SubRegSize::i64Bit, "Unsupported fmov size");
void fdup(FEXCore::ARMEmitter::SubRegSize size, FEXCore::ARMEmitter::ZRegister zd, float Value) {
LOGMAN_THROW_AA_FMT(size == FEXCore::ARMEmitter::SubRegSize::i16Bit ||
size == FEXCore::ARMEmitter::SubRegSize::i32Bit ||
size == FEXCore::ARMEmitter::SubRegSize::i64Bit, "Unsupported fmov size");
uint32_t Imm{};
if (size == SubRegSize::i16Bit) {
LOGMAN_MSG_A_FMT("Unsupported");
FEX_UNREACHABLE;
Imm = FP16ToImm8(vixl::Float16(Value));
} else if (size == SubRegSize::i32Bit) {
Imm = FP32ToImm8(Value);
} else if (size == SubRegSize::i64Bit) {
@@ -1522,7 +1521,7 @@ public:
SVEBroadcastFloatImmUnpredicated(0b00, 0, Imm, size, zd);
}
void fmov(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, float Value) {
void fmov(FEXCore::ARMEmitter::SubRegSize size, FEXCore::ARMEmitter::ZRegister zd, float Value) {
fdup(size, zd, Value);
}
@@ -3321,18 +3320,7 @@ public:
// SVE Memory - Contiguous Store with Immediate Offset
// SVE contiguous non-temporal store (scalar plus immediate)
void stnt1b(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
SVEContiguousNontemporalStore(0b00, zt, pg, rn, Imm);
}
void stnt1h(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
SVEContiguousNontemporalStore(0b01, zt, pg, rn, Imm);
}
void stnt1w(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
SVEContiguousNontemporalStore(0b10, zt, pg, rn, Imm);
}
void stnt1d(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
SVEContiguousNontemporalStore(0b11, zt, pg, rn, Imm);
}
// XXX:
// SVE store multiple structures (scalar plus immediate)
void st2b(ZRegister zt1, ZRegister zt2, PRegister pg, Register rn, int32_t Imm = 0) {
@@ -3525,8 +3513,7 @@ private:
size == SubRegSize::i64Bit, "Unsupported fcpy/fmov size");
uint32_t imm{};
if (size == SubRegSize::i16Bit) {
LOGMAN_MSG_A_FMT("Unsupported");
FEX_UNREACHABLE;
imm = FP16ToImm8(vixl::Float16(value));
} else if (size == SubRegSize::i32Bit) {
imm = FP32ToImm8(value);
} else if (size == SubRegSize::i64Bit) {
@@ -3730,7 +3717,7 @@ private:
// SVE bitwise logical operations (predicated)
void SVEBitwiseLogicalPredicated(uint32_t opc, SubRegSize size, PRegister pg, ZRegister zdn, ZRegister zm, ZRegister zd) {
LOGMAN_THROW_AA_FMT(size != ARMEmitter::SubRegSize::i128Bit, "Can't use 128-bit size");
LOGMAN_THROW_AA_FMT(size != FEXCore::ARMEmitter::SubRegSize::i128Bit, "Can't use 128-bit size");
LOGMAN_THROW_A_FMT(zd == zdn, "zd needs to equal zdn");
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
@@ -4492,22 +4479,6 @@ private:
dc32(Instr);
}
// SVE contiguous non-temporal store (scalar plus immediate)
void SVEContiguousNontemporalStore(uint32_t msz, ZRegister zt, PRegister pg, Register rn, int32_t imm) {
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_AA_FMT(imm >= -8 && imm <= 7,
"Invalid loadstore offset ({}). Must be between [-8, 7]", imm);
const auto imm4 = static_cast<uint32_t>(imm) & 0xF;
uint32_t Instr = 0b1110'0100'0001'0000'1110'0000'0000'0000;
Instr |= msz << 23;
Instr |= imm4 << 16;
Instr |= pg.Idx() << 10;
Instr |= Encode_rn(rn);
Instr |= zt.Idx();
dc32(Instr);
}
void SVEContiguousLoadImm(bool is_store, uint32_t dtype, int32_t imm, PRegister pg, Register rn, ZRegister zt) {
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
LOGMAN_THROW_AA_FMT(imm >= -8 && imm <= 7,
@@ -4772,7 +4743,7 @@ private:
dc32(Instr);
}
void SVEPermuteVector(uint32_t op0, ARMEmitter::ZRegister zd, ARMEmitter::ZRegister zm, uint32_t Imm) {
void SVEPermuteVector(uint32_t op0, FEXCore::ARMEmitter::ZRegister zd, FEXCore::ARMEmitter::ZRegister zm, uint32_t Imm) {
constexpr uint32_t Op = 0b0000'0101'0010'0000'000 << 13;
uint32_t Instr = Op;
@@ -5257,14 +5228,15 @@ private:
// Alias that returns the equivalently sized unsigned type for a floating-point type T.
template <typename T>
requires(std::is_same_v<T, float> || std::is_same_v<T, double>)
using FloatToEquivalentUInt = std::conditional_t<std::is_same_v<T, float>, uint32_t, uint64_t>;
requires(std::is_same_v<T, float> || std::is_same_v<T, double> || std::is_same_v<T, vixl::Float16>)
using FloatToEquivalentUInt = std::conditional_t<std::is_same_v<T, vixl::Float16>, uint16_t,
std::conditional_t<std::is_same_v<T, float>, uint32_t, uint64_t>>;
// Determines if a floating-point value is capable of being converted
// into an 8-bit immediate. See pseudocode definition of VFPExpandImm
// in ARM A-profile reference manual for a general overview of how this was derived.
template <typename T>
requires(std::is_same_v<T, float> || std::is_same_v<T, double>)
requires(std::is_same_v<T, float> || std::is_same_v<T, double> || std::is_same_v<T, vixl::Float16>)
[[nodiscard, maybe_unused]] static bool IsValidFPValueForImm8(T value) {
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
@@ -5305,6 +5277,18 @@ private:
return true;
}
static uint32_t FP16ToImm8(vixl::Float16 value) {
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value),
"Value cannot be encoded into an 8-bit immediate");
const uint32_t bits = vixl::Float16ToRawbits(value);
const uint32_t sign = (bits & 0x8000) >> 8;
const uint32_t expb2 = (bits & 0x2000) >> 7;
const uint32_t b5_to_0 = (bits >> 6) & 0x3F;
return sign | expb2 | b5_to_0;
}
static uint32_t FP32ToImm8(float value) {
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value),
"Value ({}) cannot be encoded into an 8-bit immediate", value);
@@ -33,7 +33,7 @@ public:
ASIMDScalarCopy(Op, 1, imm5, 0b0000, rd, rn);
}
void mov(ARMEmitter::ScalarRegSize size, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn, uint32_t Index) {
void mov(FEXCore::ARMEmitter::ScalarRegSize size, FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn, uint32_t Index) {
dup(size, rd, rn, Index);
}
@@ -1052,21 +1052,21 @@ public:
}
// Floating-point immediate
void fmov(ARMEmitter::ScalarRegSize size, ARMEmitter::VRegister rd, float Value) {
void fmov(FEXCore::ARMEmitter::ScalarRegSize size, FEXCore::ARMEmitter::VRegister rd, float Value) {
uint32_t M = 0;
uint32_t S = 0;
uint32_t ptype;
uint32_t imm8;
uint32_t imm5 = 0b0'0000;
if (size == ARMEmitter::ScalarRegSize::i16Bit) {
LOGMAN_MSG_A_FMT("Unsupported");
FEX_UNREACHABLE;
if (size == FEXCore::ARMEmitter::ScalarRegSize::i16Bit) {
ptype = 0b11;
imm8 = FP16ToImm8(vixl::Float16(Value));
}
else if (size == ARMEmitter::ScalarRegSize::i32Bit) {
else if (size == FEXCore::ARMEmitter::ScalarRegSize::i32Bit) {
ptype = 0b00;
imm8 = FP32ToImm8(Value);
}
else if (size == ARMEmitter::ScalarRegSize::i64Bit) {
else if (size == FEXCore::ARMEmitter::ScalarRegSize::i64Bit) {
ptype = 0b01;
imm8 = FP64ToImm8(Value);
}
@@ -1077,7 +1077,7 @@ public:
FloatScalarImmediate(M, S, ptype, imm8, imm5, rd);
}
void FloatScalarImmediate(uint32_t M, uint32_t S, uint32_t ptype, uint32_t imm8, uint32_t imm5, ARMEmitter::VRegister rd) {
void FloatScalarImmediate(uint32_t M, uint32_t S, uint32_t ptype, uint32_t imm8, uint32_t imm5, FEXCore::ARMEmitter::VRegister rd) {
constexpr uint32_t Op = 0b0001'1110'0010'0000'0001'00 << 10;
uint32_t Instr = Op;
@@ -1286,7 +1286,7 @@ public:
private:
// Advanced SIMD scalar copy
void ASIMDScalarCopy(uint32_t Op, uint32_t Q, uint32_t imm5, uint32_t imm4, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn) {
void ASIMDScalarCopy(uint32_t Op, uint32_t Q, uint32_t imm5, uint32_t imm4, FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn) {
uint32_t Instr = Op;
Instr |= Q << 30;
@@ -11,7 +11,7 @@ public:
// TODO: AT
// TODO: CFP
// TODO: CPP
void dc(ARMEmitter::DataCacheOperation DCOp, ARMEmitter::Register rt) {
void dc(FEXCore::ARMEmitter::DataCacheOperation DCOp, FEXCore::ARMEmitter::Register rt) {
constexpr uint32_t Op = 0b1101'0101'0000'1000'0111 << 12;
SystemInstruction(Op, 0, FEXCore::ToUnderlying(DCOp), rt);
}
@@ -48,67 +48,67 @@ public:
ExceptionGeneration(0b101, 0b000, 0b11, Imm);
}
// System instructions with register argument
void wfet(ARMEmitter::Register rt) {
void wfet(FEXCore::ARMEmitter::Register rt) {
SystemInstructionWithReg(0b0000, 0b000, rt);
}
void wfit(ARMEmitter::Register rt) {
void wfit(FEXCore::ARMEmitter::Register rt) {
SystemInstructionWithReg(0b0000, 0b001, rt);
}
// Hints
void nop() {
Hint(ARMEmitter::HintRegister::NOP);
Hint(FEXCore::ARMEmitter::HintRegister::NOP);
}
void yield() {
Hint(ARMEmitter::HintRegister::YIELD);
Hint(FEXCore::ARMEmitter::HintRegister::YIELD);
}
void wfe() {
Hint(ARMEmitter::HintRegister::WFE);
Hint(FEXCore::ARMEmitter::HintRegister::WFE);
}
void wfi() {
Hint(ARMEmitter::HintRegister::WFI);
Hint(FEXCore::ARMEmitter::HintRegister::WFI);
}
void sev() {
Hint(ARMEmitter::HintRegister::SEV);
Hint(FEXCore::ARMEmitter::HintRegister::SEV);
}
void sevl() {
Hint(ARMEmitter::HintRegister::SEVL);
Hint(FEXCore::ARMEmitter::HintRegister::SEVL);
}
void dgh() {
Hint(ARMEmitter::HintRegister::DGH);
Hint(FEXCore::ARMEmitter::HintRegister::DGH);
}
void csdb() {
Hint(ARMEmitter::HintRegister::CSDB);
Hint(FEXCore::ARMEmitter::HintRegister::CSDB);
}
// Barriers
void clrex(uint32_t imm = 15) {
LOGMAN_THROW_AA_FMT(imm < 16, "Immediate out of range");
Barrier(ARMEmitter::BarrierRegister::CLREX, imm);
Barrier(FEXCore::ARMEmitter::BarrierRegister::CLREX, imm);
}
void dsb(ARMEmitter::BarrierScope Scope) {
Barrier(ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
void dsb(FEXCore::ARMEmitter::BarrierScope Scope) {
Barrier(FEXCore::ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
}
void dmb(ARMEmitter::BarrierScope Scope) {
Barrier(ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
void dmb(FEXCore::ARMEmitter::BarrierScope Scope) {
Barrier(FEXCore::ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
}
void isb() {
Barrier(ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(ARMEmitter::BarrierScope::SY));
Barrier(FEXCore::ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(FEXCore::ARMEmitter::BarrierScope::SY));
}
void sb() {
Barrier(ARMEmitter::BarrierRegister::SB, 0);
Barrier(FEXCore::ARMEmitter::BarrierRegister::SB, 0);
}
void tcommit() {
Barrier(ARMEmitter::BarrierRegister::TCOMMIT, 0);
Barrier(FEXCore::ARMEmitter::BarrierRegister::TCOMMIT, 0);
}
// System register move
void msr(ARMEmitter::SystemRegister reg, ARMEmitter::Register rt) {
void msr(FEXCore::ARMEmitter::SystemRegister reg, FEXCore::ARMEmitter::Register rt) {
constexpr uint32_t Op = 0b1101'0101'0001 << 20;
SystemRegisterMove(Op, rt, reg);
}
void mrs(ARMEmitter::Register rd, ARMEmitter::SystemRegister reg) {
void mrs(FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::SystemRegister reg) {
constexpr uint32_t Op = 0b1101'0101'0011 << 20;
SystemRegisterMove(Op, rd, reg);
}
@@ -130,7 +130,7 @@ private:
}
// System instructions with register argument
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, ARMEmitter::Register rt) {
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, FEXCore::ARMEmitter::Register rt) {
uint32_t Instr = 0b1101'0101'0000'0011'0001 << 12;
Instr |= CRm << 8;
@@ -140,13 +140,13 @@ private:
}
// Hints
void Hint(ARMEmitter::HintRegister Reg) {
void Hint(FEXCore::ARMEmitter::HintRegister Reg) {
uint32_t Instr = 0b1101'0101'0000'0011'0010'0000'0001'1111U;
Instr |= FEXCore::ToUnderlying(Reg);
dc32(Instr);
}
// Barriers
void Barrier(ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
void Barrier(FEXCore::ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
uint32_t Instr = 0b1101'0101'0000'0011'0011'0000'0001'1111U;
Instr |= CRm << 8;
Instr |= FEXCore::ToUnderlying(Reg);
@@ -154,7 +154,7 @@ private:
}
// System Instruction
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, ARMEmitter::Register rt) {
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, FEXCore::ARMEmitter::Register rt) {
uint32_t Instr = Op;
Instr |= L << 21;
@@ -165,7 +165,7 @@ private:
}
// System register move
void SystemRegisterMove(uint32_t Op, ARMEmitter::Register rt, ARMEmitter::SystemRegister reg) {
void SystemRegisterMove(uint32_t Op, FEXCore::ARMEmitter::Register rt, FEXCore::ARMEmitter::SystemRegister reg) {
uint32_t Instr = Op;
Instr |= FEXCore::ToUnderlying(reg);
@@ -6,46 +6,48 @@
#include <utility>
namespace FEXCore {
void BlockSamplingData::DumpBlockData() {
std::fstream Output;
Output.open("output.csv", std::fstream::out | std::fstream::binary);
void BlockSamplingData::DumpBlockData() {
std::fstream Output;
Output.open("output.csv", std::fstream::out | std::fstream::binary);
if (!Output.is_open()) {
return;
}
if (!Output.is_open())
return;
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
for (auto it : SamplingMap) {
if (!it.second->TotalCalls) {
continue;
for (auto it : SamplingMap) {
if (!it.second->TotalCalls)
continue;
Output << "0x" << std::hex << it.first
<< ", " << std::dec << it.second->Min
<< ", " << std::dec << it.second->Max
<< ", " << std::dec << it.second->TotalTime
<< ", " << std::dec << it.second->TotalCalls
<< ", " << std::dec << ((double)it.second->TotalTime / (double)it.second->TotalCalls)
<< std::endl;
}
Output << "0x" << std::hex << it.first << ", " << std::dec << it.second->Min << ", " << std::dec << it.second->Max << ", " << std::dec
<< it.second->TotalTime << ", " << std::dec << it.second->TotalCalls << ", " << std::dec
<< ((double)it.second->TotalTime / (double)it.second->TotalCalls) << std::endl;
Output.close();
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
}
Output.close();
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
}
BlockSamplingData::BlockData* BlockSamplingData::GetBlockData(uint64_t RIP) {
auto it = SamplingMap.find(RIP);
if (it != SamplingMap.end()) {
return it->second;
BlockSamplingData::BlockData *BlockSamplingData::GetBlockData(uint64_t RIP) {
auto it = SamplingMap.find(RIP);
if (it != SamplingMap.end()) {
return it->second;
}
BlockData *NewData = new BlockData{};
memset(NewData, 0, sizeof(BlockData));
NewData->Min = ~0ULL;
SamplingMap[RIP] = NewData;
return NewData;
}
BlockData* NewData = new BlockData {};
memset(NewData, 0, sizeof(BlockData));
NewData->Min = ~0ULL;
SamplingMap[RIP] = NewData;
return NewData;
}
BlockSamplingData::~BlockSamplingData() {
DumpBlockData();
for (auto it : SamplingMap) {
delete it.second;
BlockSamplingData::~BlockSamplingData() {
DumpBlockData();
for (auto it : SamplingMap) {
delete it.second;
}
SamplingMap.clear();
}
SamplingMap.clear();
}
} // namespace FEXCore
@@ -14,7 +14,7 @@ public:
uint64_t TotalCalls;
};
BlockData* GetBlockData(uint64_t RIP);
BlockData *GetBlockData(uint64_t RIP);
~BlockSamplingData();
void DumpBlockData();
@@ -22,4 +22,4 @@ public:
private:
std::unordered_map<uint64_t, BlockData*> SamplingMap;
};
} // namespace FEXCore
}
+345 -356
View File
@@ -2,392 +2,381 @@
#include "FEXCore/IR/IR.h"
#include "FEXCore/Utils/AllocatorHooks.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#ifndef _WIN32
#include <sys/prctl.h>
#endif
#include <FEXCore/Core/CPUBackend.h>
namespace FEXCore {
namespace CPU {
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_PSUBADDPS_INVERT
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_PSUBADDPS_INVERT_UPPER
{0x0000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_PSUBADDPD_INVERT
{0x0000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_PSUBADDPD_INVERT_UPPER
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
{0x0706'0504'0302'0100ULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0111B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1001B
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB_UPPER
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_ONE
{0xD49A'784B'CD1B'8AFEULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_LOG2_10
{0xB8AA'3B29'5C17'F0BCULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_LOG2_E
{0xC90F'DAA2'2168'C235ULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_PI
{0x9A20'9A84'FBCF'F799ULL, 0x0000'0000'0000'3FFDULL}, // NAMED_VECTOR_X87_LOG10_2
{0xB172'17F7'D1CF'79ACULL, 0x0000'0000'0000'3FFEULL}, // NAMED_VECTOR_X87_LOG_2
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
{0x0706'0504'0302'0100ULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0111B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1001B
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
};
constexpr static auto PSHUFLW_LUT {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFLW with ARM's TBL (single register) instruction
// PSHUFLW behaviour:
// 16-bit words in [63:48], [47:32], [31:16], [15:0] are selected using the 8-bit Index.
// For 128-bit PSHUFLW, bits [127:64] are identity copied.
constexpr uint64_t IdentityCopyUpper = 0x0f'0e'0d'0c'0b'0a'09'08;
std::array<LUTType, 256> TotalLUT{};
uint64_t WordSelection[4] = {
0x01'00,
0x03'02,
0x05'04,
0x07'06,
};
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] =
(WordSelection[Word0] << 0) |
(WordSelection[Word1] << 16) |
(WordSelection[Word2] << 32) |
(WordSelection[Word3] << 48);
LUT.Val[1] = IdentityCopyUpper;
}
return TotalLUT;
}()
};
constexpr static auto PSHUFHW_LUT {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFHW with ARM's TBL (single register) instruction
// PSHUFHW behaviour:
// 16-bit words in [127:112], [111:96], [95:80], [79:64] are selected using the 8-bit Index.
// Incoming words come from bits [127:64] of the source.
// Bits [63:0] are identity copied.
constexpr uint64_t IdentityCopyLower = 0x07'06'05'04'03'02'01'00;
std::array<LUTType, 256> TotalLUT{};
uint64_t WordSelection[4] = {
0x09'08,
0x0b'0a,
0x0d'0c,
0x0f'0e,
};
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] = IdentityCopyLower;
LUT.Val[1] =
(WordSelection[Word0] << 0) |
(WordSelection[Word1] << 16) |
(WordSelection[Word2] << 32) |
(WordSelection[Word3] << 48);
}
return TotalLUT;
}()
};
constexpr static auto PSHUFD_LUT {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFD with ARM's TBL (single register) instruction
// PSHUFD behaviour:
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
std::array<LUTType, 256> TotalLUT{};
uint64_t WordSelection[4] = {
0x03'02'01'00,
0x07'06'05'04,
0x0b'0a'09'08,
0x0f'0e'0d'0c,
};
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] =
(WordSelection[Word0] << 0) |
(WordSelection[Word1] << 32);
LUT.Val[1] =
(WordSelection[Word2] << 0) |
(WordSelection[Word3] << 32);
}
return TotalLUT;
}()
};
constexpr static auto SHUFPS_LUT {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
// Expectation for this LUT is to simulate SHUFPS with ARM's TBL (two register) instruction.
// SHUFPS behaviour:
// Two 32-bits words from each source are selected from each source in the lower and upper halves of the 128-bit destination.
// Dest[31:0] = Src1[<Word0>]
// Dest[63:32] = Src1[<Word1>]
// Dest[95:64] = Src2[<Word2>]
// Dest[127:96] = Src2[<Word3>]
std::array<LUTType, 256> TotalLUT{};
const uint64_t WordSelectionSrc1[4] = {
0x03'02'01'00,
0x07'06'05'04,
0x0b'0a'09'08,
0x0f'0e'0d'0c,
};
constexpr static auto PSHUFLW_LUT {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFLW with ARM's TBL (single register) instruction
// PSHUFLW behaviour:
// 16-bit words in [63:48], [47:32], [31:16], [15:0] are selected using the 8-bit Index.
// For 128-bit PSHUFLW, bits [127:64] are identity copied.
constexpr uint64_t IdentityCopyUpper = 0x0f'0e'0d'0c'0b'0a'09'08;
std::array<LUTType, 256> TotalLUT {};
uint64_t WordSelection[4] = {
0x01'00,
0x03'02,
0x05'04,
0x07'06,
};
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
// Src2 needs to offset each byte index by 16-bytes to pull from the second source.
const uint64_t WordSelectionSrc2[4] = {
0x03'02'01'00 + (0x10101010),
0x07'06'05'04 + (0x10101010),
0x0b'0a'09'08 + (0x10101010),
0x0f'0e'0d'0c + (0x10101010),
};
LUT.Val[0] = (WordSelection[Word0] << 0) | (WordSelection[Word1] << 16) | (WordSelection[Word2] << 32) | (WordSelection[Word3] << 48);
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[1] = IdentityCopyUpper;
}
return TotalLUT;
}()};
LUT.Val[0] =
(WordSelectionSrc1[Word0] << 0) |
(WordSelectionSrc1[Word1] << 32);
constexpr static auto PSHUFHW_LUT {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFHW with ARM's TBL (single register) instruction
// PSHUFHW behaviour:
// 16-bit words in [127:112], [111:96], [95:80], [79:64] are selected using the 8-bit Index.
// Incoming words come from bits [127:64] of the source.
// Bits [63:0] are identity copied.
constexpr uint64_t IdentityCopyLower = 0x07'06'05'04'03'02'01'00;
std::array<LUTType, 256> TotalLUT {};
uint64_t WordSelection[4] = {
0x09'08,
0x0b'0a,
0x0d'0c,
0x0f'0e,
};
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[1] =
(WordSelectionSrc2[Word2] << 0) |
(WordSelectionSrc2[Word3] << 32);
}
return TotalLUT;
}()
};
LUT.Val[0] = IdentityCopyLower;
constexpr static auto DPPS_MASK {
[]() consteval {
struct LUTType {
uint32_t Val[4];
};
LUT.Val[1] = (WordSelection[Word0] << 0) | (WordSelection[Word1] << 16) | (WordSelection[Word2] << 32) | (WordSelection[Word3] << 48);
}
return TotalLUT;
}()};
constexpr static auto PSHUFD_LUT {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// Expectation for this LUT is to simulate PSHUFD with ARM's TBL (single register) instruction
// PSHUFD behaviour:
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
std::array<LUTType, 256> TotalLUT {};
uint64_t WordSelection[4] = {
0x03'02'01'00,
0x07'06'05'04,
0x0b'0a'09'08,
0x0f'0e'0d'0c,
};
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] = (WordSelection[Word0] << 0) | (WordSelection[Word1] << 32);
LUT.Val[1] = (WordSelection[Word2] << 0) | (WordSelection[Word3] << 32);
}
return TotalLUT;
}()};
constexpr static auto SHUFPS_LUT {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
// Expectation for this LUT is to simulate SHUFPS with ARM's TBL (two register) instruction.
// SHUFPS behaviour:
// Two 32-bits words from each source are selected from each source in the lower and upper halves of the 128-bit destination.
// Dest[31:0] = Src1[<Word0>]
// Dest[63:32] = Src1[<Word1>]
// Dest[95:64] = Src2[<Word2>]
// Dest[127:96] = Src2[<Word3>]
std::array<LUTType, 256> TotalLUT {};
const uint64_t WordSelectionSrc1[4] = {
0x03'02'01'00,
0x07'06'05'04,
0x0b'0a'09'08,
0x0f'0e'0d'0c,
};
// Src2 needs to offset each byte index by 16-bytes to pull from the second source.
const uint64_t WordSelectionSrc2[4] = {
0x03'02'01'00 + (0x10101010),
0x07'06'05'04 + (0x10101010),
0x0b'0a'09'08 + (0x10101010),
0x0f'0e'0d'0c + (0x10101010),
};
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] = (WordSelectionSrc1[Word0] << 0) | (WordSelectionSrc1[Word1] << 32);
LUT.Val[1] = (WordSelectionSrc2[Word2] << 0) | (WordSelectionSrc2[Word3] << 32);
}
return TotalLUT;
}()};
constexpr static auto DPPS_MASK {[]() consteval {
struct LUTType {
uint32_t Val[4];
};
std::array<LUTType, 16> TotalLUT {};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto& LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1U;
}
return 0U;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
LUT.Val[2] = GetLUT(i, 2);
LUT.Val[3] = GetLUT(i, 3);
}
return TotalLUT;
}()};
constexpr static auto DPPD_MASK {[]() consteval {
struct LUTType {
uint64_t Val[2];
};
std::array<LUTType, 4> TotalLUT {};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto& LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1ULL;
}
return 0ULL;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
}
return TotalLUT;
}()};
constexpr static auto PBLENDW_LUT {[]() consteval {
struct LUTType {
uint16_t Val[8];
};
// 16-bit words in [127:112], [111:96], [95:80], [79:64], [63:48], [47:32], [31:16], [15:0] are selected using 8-bit swizzle.
// Expectation for this LUT is to simulate PBLENDW with ARM's TBX (one register) instruction.
// PBLENDW behaviour:
// 16-bit words from the source is moved in to the destination based on the bit in the swizzle.
// Dest[15:0] = Swizzle[0] ? Src[15:0] : Dest[15:0]
// Dest[31:16] = Swizzle[1] ? Src[31:16] : Dest[31:16]
// Dest[47:32] = Swizzle[2] ? Src[47:32] : Dest[47:32]
// Dest[63:48] = Swizzle[3] ? Src[63:48] : Dest[63:48]
// Dest[79:64] = Swizzle[4] ? Src[79:64] : Dest[79:64]
// Dest[95:80] = Swizzle[5] ? Src[95:80] : Dest[95:80]
// Dest[111:96] = Swizzle[6] ? Src[111:96] : Dest[111:96]
// Dest[127:112] = Swizzle[7] ? Src[127:112] : Dest[127:112]
std::array<LUTType, 256> TotalLUT {};
const uint16_t WordSelectionSrc[8] = {
0x01'00, 0x03'02, 0x05'04, 0x07'06, 0x09'08, 0x0B'0A, 0x0D'0C, 0x0F'0E,
};
constexpr uint16_t OriginalDest = 0xFF'FF;
for (size_t i = 0; i < 256; ++i) {
auto& LUT = TotalLUT[i];
for (size_t j = 0; j < 8; ++j) {
LUT.Val[j] = ((i >> j) & 1) ? WordSelectionSrc[j] : OriginalDest;
std::array<LUTType, 16> TotalLUT{};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto &LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1U;
}
return 0U;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
LUT.Val[2] = GetLUT(i, 2);
LUT.Val[3] = GetLUT(i, 3);
}
return TotalLUT;
}()
};
constexpr static auto DPPD_MASK {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
std::array<LUTType, 4> TotalLUT{};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto &LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1ULL;
}
return 0ULL;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
}
return TotalLUT;
}()
};
constexpr static auto PBLENDW_LUT {
[]() consteval {
struct LUTType {
uint16_t Val[8];
};
// 16-bit words in [127:112], [111:96], [95:80], [79:64], [63:48], [47:32], [31:16], [15:0] are selected using 8-bit swizzle.
// Expectation for this LUT is to simulate PBLENDW with ARM's TBX (one register) instruction.
// PBLENDW behaviour:
// 16-bit words from the source is moved in to the destination based on the bit in the swizzle.
// Dest[15:0] = Swizzle[0] ? Src[15:0] : Dest[15:0]
// Dest[31:16] = Swizzle[1] ? Src[31:16] : Dest[31:16]
// Dest[47:32] = Swizzle[2] ? Src[47:32] : Dest[47:32]
// Dest[63:48] = Swizzle[3] ? Src[63:48] : Dest[63:48]
// Dest[79:64] = Swizzle[4] ? Src[79:64] : Dest[79:64]
// Dest[95:80] = Swizzle[5] ? Src[95:80] : Dest[95:80]
// Dest[111:96] = Swizzle[6] ? Src[111:96] : Dest[111:96]
// Dest[127:112] = Swizzle[7] ? Src[127:112] : Dest[127:112]
std::array<LUTType, 256> TotalLUT{};
const uint16_t WordSelectionSrc[8] = {
0x01'00,
0x03'02,
0x05'04,
0x07'06,
0x09'08,
0x0B'0A,
0x0D'0C,
0x0F'0E,
};
constexpr uint16_t OriginalDest = 0xFF'FF;
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
for (size_t j = 0; j < 8; ++j) {
LUT.Val[j] = ((i >> j) & 1) ? WordSelectionSrc[j] : OriginalDest;
}
return TotalLUT;
}()};
}
return TotalLUT;
}()
};
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
: ThreadState(ThreadState)
, InitialCodeSize(InitialCodeSize)
, MaxCodeSize(MaxCodeSize) {
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
// Initialize named vector constants.
for (size_t i = 0; i < FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX; ++i) {
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
}
// Initialize named vector constants.
for (size_t i = 0; i < FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX; ++i) {
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
}
// Copy named vector constants.
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
// Copy named vector constants.
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
// Initialize Indexed named vector constants.
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] =
reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] =
reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] =
reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] =
reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] =
reinterpret_cast<uint64_t>(DPPS_MASK.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] =
reinterpret_cast<uint64_t>(DPPD_MASK.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] =
reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
// Initialize Indexed named vector constants.
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] = reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] = reinterpret_cast<uint64_t>(DPPS_MASK.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] = reinterpret_cast<uint64_t>(DPPD_MASK.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] = reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
#ifndef FEX_DISABLE_TELEMETRY
// Fill in telemetry values
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
auto& Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(Telem.GetAddr());
}
// Fill in telemetry values
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
auto &Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(Telem.GetAddr());
}
#endif
}
CPUBackend::~CPUBackend() {
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
}
CPUBackend::~CPUBackend() {
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
}
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
if (CodeBuffers.empty()) {
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
} else {
if (CodeBuffers.size() > 1) {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
for (size_t i = 1; i < CodeBuffers.size(); i++) {
FreeCodeBuffer(CodeBuffers[i]);
}
CodeBuffers.resize(1);
}
// Set the current code buffer to the initial
CurrentCodeBuffer = &CodeBuffers[0];
if (CurrentCodeBuffer->Size != MaxCodeSize) {
FreeCodeBuffer(*CurrentCodeBuffer);
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer->Size *= 1.5;
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
}
}
} else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
if (CodeBuffers.empty()) {
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
}
} else {
if (CodeBuffers.size() > 1) {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
for (size_t i = 1; i < CodeBuffers.size(); i++) {
FreeCodeBuffer(CodeBuffers[i]);
}
CodeBuffers.resize(1);
}
// Set the current code buffer to the initial
CurrentCodeBuffer = &CodeBuffers[0];
return CurrentCodeBuffer;
}
if (CurrentCodeBuffer->Size != MaxCodeSize) {
FreeCodeBuffer(*CurrentCodeBuffer);
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
#ifndef _WIN32
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
//
// MDWE prevents applications from creating RWX memory mappings.
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
//
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
//
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
//
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
// -1: The kernel doesn't support MDWE
// 0: MDWE is supported but disabled
// >0: MDWE is enabled, hence prohibiting RWX mappings
#ifndef PR_GET_MDWE
#define PR_GET_MDWE 66
#endif
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
if (MDWE != -1 && MDWE != 0) {
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
}
#endif
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer->Size *= 1.5;
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
for (auto& Buffer : CodeBuffers) {
auto start = (uintptr_t)Buffer.Ptr;
auto end = start + Buffer.Size;
if (Address >= start && Address < end) {
return true;
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
}
}
return false;
} else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
}
} // namespace CPU
} // namespace FEXCore
return CurrentCodeBuffer;
}
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t *>(
FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
for (auto &Buffer: CodeBuffers) {
auto start = (uintptr_t)Buffer.Ptr;
auto end = start + Buffer.Size;
if (Address >= start && Address < end) {
return true;
}
}
return false;
}
}
}
File diff suppressed because it is too large. Load diff
+82 -94
View File
@@ -14,8 +14,6 @@ namespace Context {
class ContextImpl;
}
uint32_t GetCycleCounterFrequency();
// Debugging define to switch what family of CPU we execute as.
// Might be useful if an application makes an assumption about a CPU.
// #define CPUID_AMD
@@ -30,12 +28,12 @@ private:
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
public:
CPUIDEmu(const FEXCore::Context::ContextImpl* ctx);
// X86 cacheline size effectively has to be hardcoded to 64
// if we report anything differently then applications are likely to break
constexpr static uint64_t CACHELINE_SIZE = 64;
void Init(FEXCore::Context::ContextImpl *ctx);
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) const {
if (Function < Primary.size()) {
const auto Handler = Primary[Function];
@@ -58,13 +56,12 @@ public:
}
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) const {
if (Function == 0x8000'0002U) {
if (Function == 0x8000'0002U)
return Function_8000_0002h(Leaf, CPU % PerCPUData.size());
} else if (Function == 0x8000'0003U) {
else if (Function == 0x8000'0003U)
return Function_8000_0003h(Leaf, CPU % PerCPUData.size());
} else {
else
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
}
}
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) const {
@@ -114,12 +111,10 @@ public:
}
private:
const FEXCore::Context::ContextImpl* CTX;
bool Hybrid {};
uint32_t Cores {};
FEXCore::Context::ContextImpl *CTX;
bool Hybrid{};
uint32_t Cores{};
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
// XFEATURE_ENABLED_MASK
// Mask that configures what features are enabled on the CPU.
@@ -141,17 +136,11 @@ private:
constexpr static uint64_t XCR0_SSE = 1ULL << 1;
constexpr static uint64_t XCR0_AVX = 1ULL << 2;
struct FeaturesConfig {
uint64_t SHA : 1;
uint64_t _pad : 63;
uint64_t XCR0 {
XCR0_X87 |
XCR0_SSE
};
FeaturesConfig Features {
.SHA = 1,
};
uint64_t XCR0 {XCR0_X87 | XCR0_SSE};
uint32_t SupportsAVX() const {
return (XCR0 & XCR0_AVX) ? 1 : 0;
}
@@ -159,13 +148,13 @@ private:
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf) const;
struct CPUData {
const char* ProductName {};
const char *ProductName{};
#ifdef _M_ARM_64
uint32_t MIDR {};
uint32_t MIDR{};
#endif
bool IsBig {};
bool IsBig{};
};
fextl::vector<CPUData> PerCPUData {};
fextl::vector<CPUData> PerCPUData{};
// Functions
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf) const;
@@ -200,7 +189,6 @@ private:
FEXCore::CPUID::XCRResults XCRFunction_0h() const;
void SetupHostHybridFlag();
void SetupFeatures();
static constexpr size_t PRIMARY_FUNCTION_COUNT = 27;
static constexpr size_t HYPERVISOR_FUNCTION_COUNT = 2;
static constexpr size_t EXTENDED_FUNCTION_COUNT = 32;
@@ -276,74 +264,74 @@ private:
static constexpr std::array<FunctionConstant, PRIMARY_FUNCTION_COUNT> Primary_Constant = {{
// 0: Highest function parameter and ID
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
// 1: Processor info
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 2: Cache and TLB info
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 3: Serial Number(previously), now reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifndef CPUID_AMD
// 4: Deterministic cache parameters for each level
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
#else
// 4: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
// 5: Monitor/mwait
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 6: Thermal and power management
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 7: Extended feature flags
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
// 0x08: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 9: Direct Cache Access information
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0A: Architectural performance monitoring
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0B: Extended topology enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0C: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0D: Processor extended state enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
// 0x0E: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x0F: Intel RDT monitoring
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x10: Intel RDT allocation enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x12: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x12: Intel SGX capability enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x13: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x14: Intel Processor trace
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifndef CPUID_AMD
// 0x15: Timestamp counter information
// Doesn't exist on AMD hardware
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#else
// 0x15: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
// 0x16: Processor frequency information
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x17: SoC vendor attribute enumeration
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x18: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x19: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifndef CPUID_AMD
// 0x1A: Hybrid Information Sub-leaf
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#else
// 0x1A: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
}};
@@ -356,9 +344,9 @@ private:
static constexpr std::array<FunctionConstant, HYPERVISOR_FUNCTION_COUNT> Hypervisor_Constant = {{
// Hypervisor CPUID information leaf
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// FEX-Emu specific leaf
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
}};
static constexpr std::array<FunctionHandler, EXTENDED_FUNCTION_COUNT> Extended = {
@@ -438,79 +426,79 @@ private:
static constexpr std::array<FunctionConstant, EXTENDED_FUNCTION_COUNT> Extended_Constant = {{
// Largest extended function number
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// Processor vendor
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// Processor brand string
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// Processor brand string continued
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// Processor brand string continued
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifdef CPUID_AMD
// 0x8000'0005: L1 Cache and TLB identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#else
// 0x8000'0005: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
// 0x8000'0006: L2 Cache identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0007: Advanced power management information
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0008: Virtual and physical address sizes
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0009: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000A: SVM Revision
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000B: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000C: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000D: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000E: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'000F: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0010: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0011: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0012: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0013: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0014: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0015: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0016: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0017: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0018: Reserved?
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'0019: TLB 1GB page identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'001A: Performance optimization identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'001B: Instruction based sampling identifiers
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'001C: Lightweight profiling capabilities
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#ifdef CPUID_AMD
// 0x8000'001D: Cache properties
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
#else
// 0x8000'001D: Reserved
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
#endif
// 0x8000'001E: Extended APIC ID
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
// 0x8000'001F: AMD Secure Encryption
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
}};
};
} // namespace FEXCore
}
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,6 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <stdint.h>
namespace FEXCore::CPU {
}
@@ -1,23 +1,20 @@
// SPDX-License-Identifier: MIT
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/X86HelperGen.h"
#include "Utils/MemberFunctionToPointer.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <CodeEmitter/Emitter.h>
#include <atomic>
#include <condition_variable>
#include <csignal>
@@ -26,15 +23,42 @@
namespace FEXCore::CPU {
static void SleepThread(FEXCore::Context::ContextImpl* CTX, FEXCore::Core::CpuStateFrame* Frame) {
CTX->SyscallHandler->SleepThread(CTX, Frame);
void Dispatcher::SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame) {
auto Thread = Frame->Thread;
--ctx->IdleWaitRefCount;
ctx->IdleWaitCV.notify_all();
Thread->RunningEvents.ThreadSleeping = true;
// Go to sleep
Thread->StartRunning.Wait();
Thread->RunningEvents.Running = true;
++ctx->IdleWaitRefCount;
Thread->RunningEvents.ThreadSleeping = false;
ctx->IdleWaitCV.notify_all();
}
uint64_t Dispatcher::GetCompileBlockPtr() {
using ClassPtrType = void (FEXCore::Context::ContextImpl::*)(FEXCore::Core::CpuStateFrame *, uint64_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
PtrCast CompileBlockPtr;
CompileBlockPtr.ClassPtr = &FEXCore::Context::ContextImpl::CompileBlockJit;
return CompileBlockPtr.Data;
}
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl* ctx)
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
, CTX {ctx} {
, CTX {ctx}
, config {config} {
EmitDispatcher();
}
@@ -61,16 +85,8 @@ void Dispatcher::EmitDispatcher() {
// }
ARMEmitter::ForwardLabel l_CTX;
ARMEmitter::SingleUseForwardLabel l_Sleep;
#ifdef _M_ARM_64EC
// These structures are not included in the standard Windows headers, define them here
static constexpr size_t TEBCPUAreaOffset = 0x1788;
static constexpr size_t CPUAreaInSyscallCallbackOffset = 0x1;
static constexpr size_t CPUAreaEmulatorStackLimitOffset = 0x8;
static constexpr size_t CPUAreaEmulatorDataOffset = 0x30;
ARMEmitter::SingleUseForwardLabel ExitEC;
#endif
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
ARMEmitter::ForwardLabel l_Sleep;
ARMEmitter::ForwardLabel l_CompileBlock;
// Push all the register we need to save
PushCalleeSavedRegisters();
@@ -87,137 +103,101 @@ void Dispatcher::EmitDispatcher() {
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
FillStaticRegs();
ARMEmitter::BiDirectionalLabel LoopTop {};
#ifdef _M_ARM_64EC
b(&LoopTop);
AbsoluteLoopTopAddressEnterECFillSRA = GetCursorAddress<uint64_t>();
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPUAreaEmulatorDataOffset);
FillStaticRegs();
// Enter JIT
b(&LoopTop);
AbsoluteLoopTopAddressEnterEC = GetCursorAddress<uint64_t>();
// Load ThreadState and write the target PC there
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPUAreaEmulatorDataOffset);
str(EC_CALL_CHECKER_PC_REG, STATE_PTR(CpuStateFrame, State.rip));
// Swap stacks to the emulator stack
ldr(TMP1, EC_ENTRY_CPUAREA_REG, CPUAreaEmulatorStackLimitOffset);
add(ARMEmitter::Size::i64Bit, StaticRegisters[X86State::REG_RSP], ARMEmitter::Reg::rsp, 0);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, TMP1, 0);
if (EmitterCTX->HostFeatures.SupportsSVE128) {
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
if (config.StaticRegisterAllocation) {
FillStaticRegs();
}
// Enter JIT
#endif
// We want to ensure that we are 16 byte aligned at the top of this loop
Align16B();
ARMEmitter::BiDirectionalLabel FullLookup {};
ARMEmitter::BiDirectionalLabel CallBlock {};
ARMEmitter::BiDirectionalLabel FullLookup{};
ARMEmitter::BiDirectionalLabel CallBlock{};
ARMEmitter::BackwardLabel LoopTop{};
Bind(&LoopTop);
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
// Load in our RIP
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
auto RipReg = TMP3;
// Don't modify x2 since it contains our RIP once the block doesn't exist
auto RipReg = ARMEmitter::XReg::x2;
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
// L1 Cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP1, TMP1, 0);
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
sub(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, RipReg);
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &FullLookup);
br(TMP4);
br(ARMEmitter::Reg::r3);
// L1C check failed, do a full lookup
Bind(&FullLookup);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), VirtualMemorySize - 1);
}
else {
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
}
ARMEmitter::ForwardLabel NoBlock;
{
// Offset the address and add to our page pointer
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
lsr(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 12);
// Load the pointer from the offset
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ExtendedType::LSL_64, 3);
// If page pointer is zero then we have no block
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
#ifdef _M_ARM_64EC
// The LSB of an L2 page entry indicates if this page contains EC code
tbnz(TMP1, 0, &ExitEC);
#endif
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &NoBlock);
// Steal the page offset
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::r3, 0x0FFF);
// Shift the offset by the size of the block cache entry
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
// The the full LookupCacheEntry with a single LDP.
// Check the guest address first to ensure it maps to the address we are currently at.
// This fixes aliasing problems
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
// If the guest address doesn't match, Compile the block.
sub(TMP2, TMP2, RipReg);
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
sub(ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, RipReg);
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, &NoBlock);
// Check the host address to see if it matches, else compile the block.
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
stp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP3, TMP1);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::ShiftType::LSL, 4);
stp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x2, ARMEmitter::Reg::r0);
// Jump to the block
br(TMP4);
br(ARMEmitter::Reg::r3);
}
}
#ifdef _M_ARM_64EC
{
Bind(&ExitEC);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
mov(EC_CALL_CHECKER_PC_REG, RipReg);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
}
#endif
{
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
@@ -230,96 +210,71 @@ void Dispatcher::EmitDispatcher() {
{
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
#ifdef _M_ARM_64EC
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPUAreaInSyscallCallbackOffset);
#endif
mov(ARMEmitter::XReg::x0, STATE);
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
} else {
GenerateIndirectRuntimeCall<uintptr_t, void *, void *>(ARMEmitter::Reg::r2);
}
else {
blr(ARMEmitter::Reg::r2);
}
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
}
if (config.StaticRegisterAllocation)
FillStaticRegs();
FillStaticRegs();
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
strb(ARMEmitter::WReg::zr, TMP2, CPUAreaInSyscallCallbackOffset);
#endif
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
// Trigger segfault if any deferred signals are pending
strb(ARMEmitter::XReg::zr, STATE,
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
str(ARMEmitter::XReg::zr, ARMEmitter::XReg::x1, 0);
br(TMP1);
br(ARMEmitter::Reg::r0);
}
// Need to create the block
{
Bind(&NoBlock);
SpillStaticRegs(TMP1);
if (!TMP_ABIARGS) {
mov(ARMEmitter::XReg::x2, TMP3);
}
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
#ifdef _M_ARM_64EC
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPUAreaInSyscallCallbackOffset);
#endif
ldr(ARMEmitter::XReg::x0, &l_CTX);
mov(ARMEmitter::XReg::x1, STATE);
// x2 contains guest RIP
mov(ARMEmitter::XReg::x3, 0);
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
ldr(ARMEmitter::XReg::x3, &l_CompileBlock);
// X2 contains our guest RIP
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
} else {
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
GenerateIndirectRuntimeCall<void, void *, uint64_t, void *>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
}
FillStaticRegs();
if (config.StaticRegisterAllocation)
FillStaticRegs();
#ifdef _M_ARM_64EC
ldr(TMP1, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
strb(ARMEmitter::WReg::zr, TMP1, CPUAreaInSyscallCallbackOffset);
#endif
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
// Trigger segfault if any deferred signals are pending
strb(ARMEmitter::XReg::zr, STATE,
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
str(ARMEmitter::XReg::zr, TMP1, 0);
b(&LoopTop);
}
@@ -345,7 +300,8 @@ void Dispatcher::EmitDispatcher() {
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
hlt(0);
}
@@ -353,9 +309,10 @@ void Dispatcher::EmitDispatcher() {
{
// Guest SIGTRAP handler
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
brk(0);
}
@@ -365,7 +322,8 @@ void Dispatcher::EmitDispatcher() {
// Needs to be distinct from the SignalHandlerReturnAddress
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
// hlt/udf = SIGILL
// brk = SIGTRAP
@@ -376,7 +334,8 @@ void Dispatcher::EmitDispatcher() {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::r0, 0);
PopCalleeSavedRegisters();
ret();
} else {
}
else {
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
}
@@ -384,7 +343,8 @@ void Dispatcher::EmitDispatcher() {
{
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
SpillStaticRegs(TMP1);
if (config.StaticRegisterAllocation)
SpillStaticRegs(TMP1);
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
// We are pausing, this means the frontend should be waiting for this thread to idle
@@ -395,8 +355,9 @@ void Dispatcher::EmitDispatcher() {
mov(ARMEmitter::XReg::x1, STATE);
ldr(ARMEmitter::XReg::x2, &l_Sleep);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
GenerateIndirectRuntimeCall<void, void *, void *>(ARMEmitter::Reg::r2);
}
else {
blr(ARMEmitter::Reg::r2);
}
@@ -450,58 +411,112 @@ void Dispatcher::EmitDispatcher() {
str(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, State.rip));
// load static regs
FillStaticRegs();
if (config.StaticRegisterAllocation)
FillStaticRegs();
// Now go back to the regular dispatcher loop
b(&LoopTop);
}
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
auto Address = GetCursorAddress<uint64_t>();
{
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
if (!TMP_ABIARGS) {
mov(ARMEmitter::XReg::x0, TMP1);
mov(ARMEmitter::XReg::x1, TMP2);
mov(ARMEmitter::XReg::x2, TMP3);
}
ldr(ARMEmitter::XReg::x3, R, Offset);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
} else {
}
else {
blr(ARMEmitter::Reg::r3);
}
// Result is now in x0
if (!TMP_ABIARGS) {
mov(TMP1, ARMEmitter::XReg::x0);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
return Address;
};
}
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
LUREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
LREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
{
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
{
LREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
SpillStaticRegs(ARMEmitter::Reg::r3);
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
// Result is now in x0
// Fix the stack and any values that were stepped on
PopDynamicRegsAndLR();
// Go back to our code block
ret();
}
Bind(&l_CTX);
dc64(reinterpret_cast<uintptr_t>(CTX));
Bind(&l_Sleep);
dc64(reinterpret_cast<uint64_t>(SleepThread));
Bind(&l_CompileBlock);
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::Context::ContextImpl::CompileBlock);
dc64(PMF.GetConvertedPointer());
dc64(GetCompileBlockPtr());
Start = reinterpret_cast<uint64_t>(DispatchPtr);
End = GetCursorAddress<uint64_t>();
@@ -520,7 +535,7 @@ void Dispatcher::EmitDispatcher() {
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
for (auto PCToDecode = DisasmBegin; PCToDecode < DisasmEnd; PCToDecode += 4) {
DisasmDecoder->Decode(PCToDecode);
auto Output = Disasm->GetOutput();
auto Output = Disasm.GetOutput();
LogMan::Msg::IFmt("{}", Output);
}
}
@@ -528,28 +543,26 @@ void Dispatcher::EmitDispatcher() {
}
#ifdef VIXL_SIMULATOR
void Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
void Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.RunFrom(reinterpret_cast< const vixl::aarch64::Instruction*>(DispatchPtr));
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(DispatchPtr));
}
void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
Simulator.WriteXRegister(1, RIP);
Simulator.RunFrom(reinterpret_cast< const vixl::aarch64::Instruction*>(CallbackPtr));
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(CallbackPtr));
}
#endif
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread) {
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto& Common = Thread->CurrentFrame->Pointers.Common;
auto &Common = Thread->CurrentFrame->Pointers.Common;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Common.DispatcherLoopTopEnterEC = AbsoluteLoopTopAddressEnterEC;
Common.DispatcherLoopTopEnterECFillSRA = AbsoluteLoopTopAddressEnterECFillSRA;
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
@@ -559,7 +572,7 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
Common.SignalReturnHandler = SignalHandlerReturnAddress;
Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
auto& AArch64 = Thread->CurrentFrame->Pointers.AArch64;
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
AArch64.LUDIVHandler = LUDIVHandlerAddress;
AArch64.LDIVHandler = LDIVHandlerAddress;
AArch64.LUREMHandler = LUREMHandlerAddress;
@@ -567,8 +580,8 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
}
}
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl* CTX) {
return fextl::make_unique<Dispatcher>(CTX);
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config) {
return fextl::make_unique<Dispatcher>(CTX, Config);
}
} // namespace FEXCore::CPU
}
@@ -2,8 +2,8 @@
#pragma once
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/fextl/memory.h>
#ifdef VIXL_SIMULATOR
@@ -23,7 +23,7 @@ struct GuestSigAction;
namespace FEXCore::Core {
struct CpuStateFrame;
struct InternalThreadState;
} // namespace FEXCore::Core
}
namespace FEXCore::Context {
class ContextImpl;
@@ -31,52 +31,55 @@ class ContextImpl;
namespace FEXCore::CPU {
#define STATE_PTR(STATE_TYPE, FIELD) STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
struct DispatcherConfig {
bool StaticRegisterAllocation = false;
};
#define STATE_PTR(STATE_TYPE, FIELD) \
STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
class Dispatcher final : public Arm64Emitter {
public:
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl* CTX);
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
Dispatcher(FEXCore::Context::ContextImpl* ctx);
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config);
~Dispatcher();
/**
* @name Dispatch Helper functions
* @{ */
uint64_t ThreadStopHandlerAddress {};
uint64_t ThreadStopHandlerAddressSpillSRA {};
uint64_t AbsoluteLoopTopAddress {};
uint64_t AbsoluteLoopTopAddressFillSRA {};
uint64_t AbsoluteLoopTopAddressEnterEC {};
uint64_t AbsoluteLoopTopAddressEnterECFillSRA {};
uint64_t ThreadPauseHandlerAddress {};
uint64_t ThreadPauseHandlerAddressSpillSRA {};
uint64_t ExitFunctionLinkerAddress {};
uint64_t SignalHandlerReturnAddress {};
uint64_t SignalHandlerReturnAddressRT {};
uint64_t GuestSignal_SIGILL {};
uint64_t GuestSignal_SIGTRAP {};
uint64_t GuestSignal_SIGSEGV {};
uint64_t IntCallbackReturnAddress {};
uint64_t ThreadStopHandlerAddress{};
uint64_t ThreadStopHandlerAddressSpillSRA{};
uint64_t AbsoluteLoopTopAddress{};
uint64_t AbsoluteLoopTopAddressFillSRA{};
uint64_t ThreadPauseHandlerAddress{};
uint64_t ThreadPauseHandlerAddressSpillSRA{};
uint64_t ExitFunctionLinkerAddress{};
uint64_t SignalHandlerReturnAddress{};
uint64_t SignalHandlerReturnAddressRT{};
uint64_t GuestSignal_SIGILL{};
uint64_t GuestSignal_SIGTRAP{};
uint64_t GuestSignal_SIGSEGV{};
uint64_t IntCallbackReturnAddress{};
uint64_t PauseReturnInstruction {};
uint64_t PauseReturnInstruction{};
/** @} */
uint64_t Start {};
uint64_t End {};
uint64_t Start{};
uint64_t End{};
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread);
#ifdef VIXL_SIMULATOR
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) ;
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
#else
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
DispatchPtr(Frame);
}
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
CallbackPtr(Frame, RIP);
}
#endif
@@ -103,22 +106,29 @@ public:
}
}
protected:
FEXCore::Context::ContextImpl* CTX;
const DispatcherConfig& GetConfig() const { return config; }
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame);
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
protected:
FEXCore::Context::ContextImpl *CTX;
DispatcherConfig config;
static void SleepThread(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::CpuStateFrame *Frame);
static uint64_t GetCompileBlockPtr();
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
AsmDispatch DispatchPtr;
JITCallback CallbackPtr;
private:
// Long division helpers
uint64_t LUDIVHandlerAddress {};
uint64_t LDIVHandlerAddress {};
uint64_t LUREMHandlerAddress {};
uint64_t LREMHandlerAddress {};
uint64_t LUDIVHandlerAddress{};
uint64_t LDIVHandlerAddress{};
uint64_t LUREMHandlerAddress{};
uint64_t LREMHandlerAddress{};
void EmitDispatcher();
};
} // namespace FEXCore::CPU
}
File diff suppressed because it is too large. Load diff
+24 -31
View File
@@ -21,10 +21,10 @@ class Decoder final {
public:
// New Frontend decoding
struct DecodedBlocks final {
uint64_t Entry {};
uint64_t NumInstructions {};
FEXCore::X86Tables::DecodedInst* DecodedInstructions;
bool HasInvalidInstruction {};
uint64_t Entry{};
uint64_t NumInstructions{};
FEXCore::X86Tables::DecodedInst *DecodedInstructions;
bool HasInvalidInstruction{};
};
struct DecodedBlockInformation final {
@@ -32,24 +32,19 @@ public:
fextl::vector<DecodedBlocks> Blocks;
};
Decoder(FEXCore::Context::ContextImpl* ctx);
Decoder(FEXCore::Context::ContextImpl *ctx);
~Decoder();
void DecodeInstructionsAtEntry(const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst,
std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, uint64_t MaxInst, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
const DecodedBlockInformation* GetDecodedBlockInfo() const {
DecodedBlockInformation const *GetDecodedBlockInfo() const {
return &BlockInfo;
}
uint64_t DecodedMinAddress {};
uint64_t DecodedMaxAddress {~0ULL};
void SetSectionMaxAddress(uint64_t v) {
SectionMaxAddress = v;
}
void SetExternalBranches(fextl::set<uint64_t>* v) {
ExternalBranches = v;
}
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
void SetExternalBranches(fextl::set<uint64_t> *v) { ExternalBranches = v; }
void DelayedDisownBuffer() {
PoolObject.DelayedDisownBuffer();
@@ -64,8 +59,8 @@ private:
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
};
FEXCore::Context::ContextImpl* CTX;
const FEXCore::HLE::SyscallOSABI OSABI {};
FEXCore::Context::ContextImpl *CTX;
const FEXCore::HLE::SyscallOSABI OSABI{};
bool DecodeInstruction(uint64_t PC);
@@ -75,24 +70,22 @@ private:
uint8_t ReadByte();
uint8_t PeekByte(uint8_t Offset) const;
uint64_t ReadData(uint8_t Size);
void SkipBytes(uint8_t Size) {
InstructionSize += Size;
}
void SkipBytes(uint8_t Size) { InstructionSize += Size; }
bool NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options = {});
bool NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
size_t DecodedSize {};
const uint8_t* InstStream;
uint8_t const *InstStream;
static constexpr size_t MAX_INST_SIZE = 15;
uint8_t InstructionSize;
std::array<uint8_t, MAX_INST_SIZE> Instruction;
FEXCore::X86Tables::DecodedInst* DecodeInst;
FEXCore::X86Tables::DecodedInst *DecodeInst;
// This is for multiblock data tracking
bool SymbolAvailable {false};
@@ -106,21 +99,21 @@ private:
DecodedBlockInformation BlockInfo;
fextl::set<uint64_t> BlocksToDecode;
fextl::set<uint64_t> HasBlocks;
fextl::set<uint64_t>* ExternalBranches {nullptr};
fextl::set<uint64_t> *ExternalBranches {nullptr};
// ModRM rm decoding
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
void DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp {
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp{
&FEXCore::Frontend::Decoder::DecodeModRM_64,
&FEXCore::Frontend::Decoder::DecodeModRM_16,
};
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
const uint8_t *AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP);
FEXCORE_TELEMETRY_INIT(VEXOpTelem, TYPE_USES_VEX_OPS);
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
};
} // namespace FEXCore::Frontend
}
+158 -107
View File
@@ -9,6 +9,14 @@
#ifdef _M_X86_64
#define XBYAK64
#define XBYAK_CUSTOM_ALLOC
#define XBYAK_CUSTOM_MALLOC FEXCore::Allocator::malloc
#define XBYAK_CUSTOM_FREE FEXCore::Allocator::free
#define XBYAK_CUSTOM_SETS
#define XBYAK_STD_UNORDERED_SET fextl::unordered_set
#define XBYAK_STD_UNORDERED_MAP fextl::unordered_map
#define XBYAK_STD_UNORDERED_MULTIMAP fextl::unordered_multimap
#define XBYAK_STD_LIST fextl::list
#define XBYAK_NO_EXCEPTION
#include <FEXCore/fextl/list.h>
#include <FEXCore/fextl/unordered_map.h>
@@ -28,29 +36,24 @@ namespace FEXCore {
[[maybe_unused]] constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
#ifdef _M_ARM_64
[[maybe_unused]]
static uint32_t GetDCZID() {
uint64_t Result {};
__asm("mrs %[Res], DCZID_EL0" : [Res] "=r"(Result));
[[maybe_unused]] static uint32_t GetDCZID() {
uint64_t Result{};
__asm("mrs %[Res], DCZID_EL0"
: [Res] "=r" (Result));
return Result;
}
static uint32_t GetFPCR() {
uint64_t Result {};
__asm("mrs %[Res], FPCR" : [Res] "=r"(Result));
uint64_t Result{};
__asm ("mrs %[Res], FPCR"
: [Res] "=r" (Result));
return Result;
}
static void SetFPCR(uint64_t Value) {
__asm("msr FPCR, %[Value]" ::[Value] "r"(Value));
__asm ("msr FPCR, %[Value]"
:: [Value] "r" (Value));
}
static uint32_t GetMIDR() {
uint64_t Result {};
__asm("mrs %[Res], MIDR_EL1" : [Res] "=r"(Result));
return Result;
}
#else
static uint32_t GetDCZID() {
// Return unsupported
@@ -58,7 +61,7 @@ static uint32_t GetDCZID() {
}
#endif
static void OverrideFeatures(HostFeatures* Features, uint64_t ForceSVEWidth) {
static void OverrideFeatures(HostFeatures *Features) {
// Override features if the user has specifically called for it.
FEX_CONFIG_OPT(HostFeatures, HOSTFEATURES);
if (!HostFeatures()) {
@@ -66,57 +69,132 @@ static void OverrideFeatures(HostFeatures* Features, uint64_t ForceSVEWidth) {
return;
}
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
do { \
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
const bool AlreadyEnabled = Features->FeatureName; \
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
Features->FeatureName = Result; \
} while (0)
#define ENABLE_DISABLE_OPTION(name, enum_name) \
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
#define GET_SINGLE_OPTION(name, enum_name) \
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
ENABLE_DISABLE_OPTION(SupportsAVX, AVX, AVX);
ENABLE_DISABLE_OPTION(SupportsSVE128, SVE, SVE);
ENABLE_DISABLE_OPTION(SupportsAFP, AFP, AFP);
ENABLE_DISABLE_OPTION(SupportsRCPC, LRCPC, LRCPC);
ENABLE_DISABLE_OPTION(SupportsTSOImm9, LRCPC2, LRCPC2);
ENABLE_DISABLE_OPTION(SupportsCSSC, CSSC, CSSC);
ENABLE_DISABLE_OPTION(SupportsPMULL_128Bit, PMULL128, PMULL128);
ENABLE_DISABLE_OPTION(SupportsRAND, RNG, RNG);
ENABLE_DISABLE_OPTION(SupportsCLZERO, CLZERO, CLZERO);
ENABLE_DISABLE_OPTION(SupportsAtomics, Atomics, ATOMICS);
ENABLE_DISABLE_OPTION(SupportsFCMA, FCMA, FCMA);
ENABLE_DISABLE_OPTION(SupportsFlagM, FlagM, FLAGM);
ENABLE_DISABLE_OPTION(SupportsFlagM2, FlagM2, FLAGM2);
ENABLE_DISABLE_OPTION(SupportsRPRES, RPRES, RPRES);
ENABLE_DISABLE_OPTION(SupportsPreserveAllABI, PRESERVEALLABI, PRESERVEALLABI);
GET_SINGLE_OPTION(Crypto, CRYPTO);
ENABLE_DISABLE_OPTION(AVX, AVX);
ENABLE_DISABLE_OPTION(AVX2, AVX2);
ENABLE_DISABLE_OPTION(SVE, SVE);
ENABLE_DISABLE_OPTION(AFP, AFP);
ENABLE_DISABLE_OPTION(LRCPC, LRCPC);
ENABLE_DISABLE_OPTION(LRCPC2, LRCPC2);
ENABLE_DISABLE_OPTION(CSSC, CSSC);
ENABLE_DISABLE_OPTION(PMULL128, PMULL128);
ENABLE_DISABLE_OPTION(RNG, RNG);
ENABLE_DISABLE_OPTION(CLZERO, CLZERO);
ENABLE_DISABLE_OPTION(Atomics, ATOMICS);
ENABLE_DISABLE_OPTION(FCMA, FCMA);
ENABLE_DISABLE_OPTION(FlagM, FLAGM);
ENABLE_DISABLE_OPTION(FlagM2, FLAGM2);
ENABLE_DISABLE_OPTION(Crypto, CRYPTO);
ENABLE_DISABLE_OPTION(RPRES, RPRES);
#undef ENABLE_DISABLE_OPTION
#undef GET_SINGLE_OPTION
if (EnableAVX) {
Features->SupportsAVX = true;
}
else if (DisableAVX) {
Features->SupportsAVX = false;
}
if (EnableAVX2) {
Features->SupportsAVX2 = true;
}
else if (DisableAVX2) {
Features->SupportsAVX2 = false;
}
if (EnableSVE) {
Features->SupportsSVE = true;
}
else if (DisableSVE) {
Features->SupportsSVE = false;
}
if (EnableAFP) {
Features->SupportsAFP = true;
}
else if (DisableAFP) {
Features->SupportsAFP = false;
}
if (EnableLRCPC) {
Features->SupportsRCPC = true;
}
else if (DisableLRCPC) {
Features->SupportsRCPC = false;
}
if (EnableLRCPC2) {
Features->SupportsTSOImm9 = true;
}
else if (DisableLRCPC2) {
Features->SupportsTSOImm9 = false;
}
if (EnableCSSC) {
Features->SupportsCSSC = true;
}
else if (DisableCSSC) {
Features->SupportsCSSC = false;
}
if (EnablePMULL128) {
Features->SupportsPMULL_128Bit = true;
}
else if (DisablePMULL128) {
Features->SupportsPMULL_128Bit = false;
}
if (EnableRNG) {
Features->SupportsRAND = true;
}
else if (DisableRNG) {
Features->SupportsRAND = false;
}
if (EnableCLZERO) {
Features->SupportsCLZERO = true;
}
else if (DisableCLZERO) {
Features->SupportsCLZERO = false;
}
if (EnableAtomics) {
Features->SupportsAtomics = true;
}
else if (DisableAtomics) {
Features->SupportsAtomics = false;
}
if (EnableFCMA) {
Features->SupportsFCMA = true;
}
else if (DisableFCMA) {
Features->SupportsFCMA = false;
}
if (EnableFlagM) {
Features->SupportsFlagM = true;
}
else if (DisableFlagM) {
Features->SupportsFlagM = false;
}
if (EnableFlagM2) {
Features->SupportsFlagM2 = true;
}
else if (DisableFlagM2) {
Features->SupportsFlagM2 = false;
}
if (EnableCrypto) {
Features->SupportsAES = true;
Features->SupportsCRC = true;
Features->SupportsSHA = true;
Features->SupportsPMULL_128Bit = true;
Features->SupportsAES256 = true;
} else if (DisableCrypto) {
}
else if (DisableCrypto) {
Features->SupportsAES = false;
Features->SupportsCRC = false;
Features->SupportsSHA = false;
Features->SupportsPMULL_128Bit = false;
Features->SupportsAES256 = false;
}
///< Only force enable SVE256 if SVE is already enabled and ForceSVEWidth is set to >= 256.
Features->SupportsSVE256 = ForceSVEWidth && ForceSVEWidth >= 256;
if (EnableRPRES) {
Features->SupportsRPRES = true;
}
else if (DisableRPRES) {
Features->SupportsRPRES = false;
}
}
HostFeatures::HostFeatures() {
@@ -133,12 +211,10 @@ HostFeatures::HostFeatures() {
auto Features = vixl::CPUFeatures::InferFromIDRegisters();
#endif
FEX_CONFIG_OPT(ForceSVEWidth, FORCESVEWIDTH);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) && Features.Has(vixl::CPUFeatures::Feature::kSHA2);
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) &&
Features.Has(vixl::CPUFeatures::Feature::kSHA2);
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
@@ -155,23 +231,26 @@ HostFeatures::HostFeatures() {
Supports3DNow = true;
SupportsSSE4A = true;
#ifdef VIXL_SIMULATOR
// Hardcode enable SVE with 256-bit wide registers.
SupportsSVE128 = ForceSVEWidth() ? ForceSVEWidth() >= 128 : true;
SupportsSVE256 = ForceSVEWidth() ? ForceSVEWidth() >= 256 : true;
#else
SupportsSVE128 = Features.Has(vixl::CPUFeatures::Feature::kSVE2);
SupportsSVE256 = Features.Has(vixl::CPUFeatures::Feature::kSVE2) && vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
#endif
SupportsSVE = true;
SupportsAVX = true;
SupportsAES256 = SupportsAVX && SupportsAES;
#else
SupportsSVE = Features.Has(vixl::CPUFeatures::Feature::kSVE);
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
#endif
// TODO: AVX2 is currently unsupported. Disable until the remaining features are implemented.
SupportsAVX2 = false;
SupportsBMI1 = true;
SupportsBMI2 = true;
SupportsCLWB = true;
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
SupportsAFP = false;
// RPRES has a dependency on AFP. Disable it until AFP is enabled.
SupportsRPRES = false;
if (!SupportsAtomics) {
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
}
@@ -180,19 +259,21 @@ HostFeatures::HostFeatures() {
// We need to get the CPU's cache line size
// We expect sane targets that have correct cacheline sizes across clusters
uint64_t CTR;
__asm volatile("mrs %[ctr], ctr_el0" : [ctr] "=r"(CTR));
__asm volatile ("mrs %[ctr], ctr_el0"
: [ctr] "=r"(CTR));
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
ICacheLineSize = 4 << (CTR & 0xF);
// Test if this CPU supports float exception trapping by attempting to enable
// On unsupported these bits are architecturally defined as RAZ/WI
constexpr uint32_t ExceptionEnableTraps = (1U << 8) | // Invalid Operation float exception trap enable
(1U << 9) | // Divide by zero float exception trap enable
(1U << 10) | // Overflow float exception trap enable
(1U << 11) | // Underflow float exception trap enable
(1U << 12) | // Inexact float exception trap enable
(1U << 15); // Input Denormal float exception trap enable
constexpr uint32_t ExceptionEnableTraps =
(1U << 8) | // Invalid Operation float exception trap enable
(1U << 9) | // Divide by zero float exception trap enable
(1U << 10) | // Overflow float exception trap enable
(1U << 11) | // Underflow float exception trap enable
(1U << 12) | // Inexact float exception trap enable
(1U << 15); // Input Denormal float exception trap enable
uint32_t OriginalFPCR = GetFPCR();
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
@@ -202,24 +283,6 @@ HostFeatures::HostFeatures() {
// Set FPCR back to original just in case anything changed
SetFPCR(OriginalFPCR);
if (SupportsRAND) {
const auto MIDR = GetMIDR();
constexpr uint32_t Implementer_QCOM = 0x51;
constexpr uint32_t PartNum_Oryon1 = 0x001;
const uint32_t MIDR_Implementer = (MIDR >> 24) & 0xFF;
const uint32_t MIDR_PartNum = (MIDR >> 4) & 0xFFF;
if (MIDR_Implementer == Implementer_QCOM && MIDR_PartNum == PartNum_Oryon1) {
// Work around an errata in Qualcomm's Oryon.
// While this CPU implements the RAND extension:
// - The RNDR register works.
// - The RNDRRS register will never read a random number. (Always return failure)
// This is contrary to x86 RNG behaviour where it allows spurious failure with RDSEED, but guarantees eventual success.
// This manifested itself on Linux when an x86 processor failed to guarantee forward progress and boot of services would infinite
// loop. Just disable this extension if this CPU is detected.
SupportsRAND = false;
}
}
#endif
#ifdef VIXL_SIMULATOR
@@ -245,7 +308,7 @@ HostFeatures::HostFeatures() {
ICacheLineSize = 64U;
#if !defined(VIXL_SIMULATOR)
Xbyak::util::Cpu X86Features {};
Xbyak::util::Cpu X86Features{};
SupportsAES = X86Features.has(Xbyak::util::Cpu::tAESNI);
SupportsCRC = X86Features.has(Xbyak::util::Cpu::tSSE42);
SupportsRAND = X86Features.has(Xbyak::util::Cpu::tRDRAND) && X86Features.has(Xbyak::util::Cpu::tRDSEED);
@@ -254,12 +317,12 @@ HostFeatures::HostFeatures() {
Supports3DNow = X86Features.has(Xbyak::util::Cpu::t3DN) && X86Features.has(Xbyak::util::Cpu::tE3DN);
SupportsSSE4A = X86Features.has(Xbyak::util::Cpu::tSSE4a);
SupportsAVX = true;
SupportsAVX2 = true;
SupportsSHA = X86Features.has(Xbyak::util::Cpu::tSHA);
SupportsBMI1 = X86Features.has(Xbyak::util::Cpu::tBMI1);
SupportsBMI2 = X86Features.has(Xbyak::util::Cpu::tBMI2);
SupportsCLWB = X86Features.has(Xbyak::util::Cpu::tCLWB);
SupportsPMULL_128Bit = X86Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
SupportsAES256 = SupportsAES && X86Features.has(Xbyak::util::Cpu::tVAES);
// xbyak doesn't know how to check for CLZero
// First ensure we support a new enough extended CPUID function range
@@ -276,18 +339,6 @@ HostFeatures::HostFeatures() {
SupportsFloatExceptions = true;
#endif
#endif
SupportsPreserveAllABI = FEXCORE_HAS_PRESERVE_ALL_ATTR;
if (!Is64BitMode()) {
///< Always disable AVX and AVX2 in 32-bit mode.
// When AVX256 is enabled, signal frames start using significantly more stack space.
// - 16bytes * 16 registers = 256 bytes for XMM registers.
// - 32bytes * 16 registers = 512 bytes for YMM registers.
// There are known game failures on real x86 hardware where a 32-bit game is running up against the wall on stack space on non-AVX
// hardware and then explodes when run on AVX hardware. This is to guard against that.
SupportsAVX = false;
}
OverrideFeatures(this, ForceSVEWidth());
OverrideFeatures(this);
}
}
} // namespace FEXCore
@@ -0,0 +1,2 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/Debug/InternalThreadState.h>
@@ -2,36 +2,48 @@
#include "Common/SoftFloat.h"
#include "Common/SoftFloat-3e/softfloat.h"
#include <FEXCore/IR/IR.h>
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
#include "Interface/IR/IR.h"
namespace FEXCore::CPU {
FEXCORE_PRESERVE_ALL_ATTR static void LoadDeferredFCW(uint16_t NewFCW) {
FEXCORE_PRESERVE_ALL_ATTR
static void LoadDeferredFCW(uint16_t NewFCW) {
auto PC = (NewFCW >> 8) & 3;
switch (PC) {
case 0: extF80_roundingPrecision = 32; break;
case 2: extF80_roundingPrecision = 64; break;
case 3: extF80_roundingPrecision = 80; break;
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
switch(PC) {
case 0: extF80_roundingPrecision = 32; break;
case 2: extF80_roundingPrecision = 64; break;
case 3: extF80_roundingPrecision = 80; break;
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
}
auto RC = (NewFCW >> 10) & 3;
switch (RC) {
case 0: softfloat_roundingMode = softfloat_round_near_even; break;
case 1: softfloat_roundingMode = softfloat_round_min; break;
case 2: softfloat_roundingMode = softfloat_round_max; break;
case 3: softfloat_roundingMode = softfloat_round_minMag; break;
switch(RC) {
case 0:
softfloat_roundingMode = softfloat_round_near_even;
break;
case 1:
softfloat_roundingMode = softfloat_round_min;
break;
case 2:
softfloat_roundingMode = softfloat_round_max;
break;
case 3:
softfloat_roundingMode = softfloat_round_minMag;
break;
}
}
template<>
struct OpHandlers<IR::OP_F80CVTTO> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t NewFCW, float src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle4(uint16_t NewFCW, float src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle8(uint16_t NewFCW, double src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle8(uint16_t NewFCW, double src) {
LoadDeferredFCW(NewFCW);
return src;
}
@@ -40,20 +52,24 @@ struct OpHandlers<IR::OP_F80CVTTO> {
template<>
struct OpHandlers<IR::OP_F80CMP> {
template<uint32_t Flags>
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static uint64_t handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
bool eq, lt, nan;
uint64_t ResultFlags = 0;
X80SoftFloat::FCMP(Src1, Src2, &eq, &lt, &nan);
if (Flags & (1 << IR::FCMP_FLAG_LT) && lt) {
if (Flags & (1 << IR::FCMP_FLAG_LT) &&
lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) && nan) {
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
nan) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
if (Flags & (1 << IR::FCMP_FLAG_EQ) && eq) {
if (Flags & (1 << IR::FCMP_FLAG_EQ) &&
eq) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
return ResultFlags;
@@ -62,12 +78,14 @@ struct OpHandlers<IR::OP_F80CMP> {
template<>
struct OpHandlers<IR::OP_F80CVT> {
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static float handle4(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static double handle8(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
@@ -75,22 +93,26 @@ struct OpHandlers<IR::OP_F80CVT> {
template<>
struct OpHandlers<IR::OP_F80CVTINT> {
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int16_t handle2(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t handle4(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int64_t handle8(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int16_t handle2t(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
@@ -103,12 +125,14 @@ struct OpHandlers<IR::OP_F80CVTINT> {
}
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t handle4t(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return extF80_to_i32(src, softfloat_round_minMag, false);
}
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t NewFCW, X80SoftFloat src) {
FEXCORE_PRESERVE_ALL_ATTR
static int64_t handle8t(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return extF80_to_i64(src, softfloat_round_minMag, false);
}
@@ -116,12 +140,14 @@ struct OpHandlers<IR::OP_F80CVTINT> {
template<>
struct OpHandlers<IR::OP_F80CVTTOINT> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle2(uint16_t NewFCW, int16_t src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle2(uint16_t NewFCW, int16_t src) {
LoadDeferredFCW(NewFCW);
return src;
}
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t NewFCW, int32_t src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle4(uint16_t NewFCW, int32_t src) {
LoadDeferredFCW(NewFCW);
return src;
}
@@ -129,7 +155,8 @@ struct OpHandlers<IR::OP_F80CVTTOINT> {
template<>
struct OpHandlers<IR::OP_F80ROUND> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FRNDINT(Src1);
}
@@ -137,7 +164,8 @@ struct OpHandlers<IR::OP_F80ROUND> {
template<>
struct OpHandlers<IR::OP_F80F2XM1> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::F2XM1(Src1);
}
@@ -145,7 +173,8 @@ struct OpHandlers<IR::OP_F80F2XM1> {
template<>
struct OpHandlers<IR::OP_F80TAN> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FTAN(Src1);
}
@@ -153,7 +182,8 @@ struct OpHandlers<IR::OP_F80TAN> {
template<>
struct OpHandlers<IR::OP_F80SQRT> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSQRT(Src1);
}
@@ -161,7 +191,8 @@ struct OpHandlers<IR::OP_F80SQRT> {
template<>
struct OpHandlers<IR::OP_F80SIN> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSIN(Src1);
}
@@ -169,7 +200,8 @@ struct OpHandlers<IR::OP_F80SIN> {
template<>
struct OpHandlers<IR::OP_F80COS> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FCOS(Src1);
}
@@ -177,7 +209,8 @@ struct OpHandlers<IR::OP_F80COS> {
template<>
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FXTRACT_EXP(Src1);
}
@@ -185,7 +218,8 @@ struct OpHandlers<IR::OP_F80XTRACT_EXP> {
template<>
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FXTRACT_SIG(Src1);
}
@@ -193,7 +227,8 @@ struct OpHandlers<IR::OP_F80XTRACT_SIG> {
template<>
struct OpHandlers<IR::OP_F80ADD> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FADD(Src1, Src2);
}
@@ -201,7 +236,8 @@ struct OpHandlers<IR::OP_F80ADD> {
template<>
struct OpHandlers<IR::OP_F80SUB> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSUB(Src1, Src2);
}
@@ -209,7 +245,8 @@ struct OpHandlers<IR::OP_F80SUB> {
template<>
struct OpHandlers<IR::OP_F80MUL> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FMUL(Src1, Src2);
}
@@ -217,7 +254,8 @@ struct OpHandlers<IR::OP_F80MUL> {
template<>
struct OpHandlers<IR::OP_F80DIV> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FDIV(Src1, Src2);
}
@@ -225,7 +263,8 @@ struct OpHandlers<IR::OP_F80DIV> {
template<>
struct OpHandlers<IR::OP_F80FYL2X> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FYL2X(Src1, Src2);
}
@@ -233,7 +272,8 @@ struct OpHandlers<IR::OP_F80FYL2X> {
template<>
struct OpHandlers<IR::OP_F80ATAN> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FATAN(Src1, Src2);
}
@@ -241,7 +281,8 @@ struct OpHandlers<IR::OP_F80ATAN> {
template<>
struct OpHandlers<IR::OP_F80FPREM1> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FREM1(Src1, Src2);
}
@@ -249,7 +290,8 @@ struct OpHandlers<IR::OP_F80FPREM1> {
template<>
struct OpHandlers<IR::OP_F80FPREM> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FREM(Src1, Src2);
}
@@ -257,7 +299,8 @@ struct OpHandlers<IR::OP_F80FPREM> {
template<>
struct OpHandlers<IR::OP_F80SCALE> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSCALE(Src1, Src2);
}
@@ -331,14 +374,15 @@ template<>
struct OpHandlers<IR::OP_F64SCALE> {
static double handle(uint16_t NewFCW, double src1, double src2) {
LoadDeferredFCW(NewFCW);
double trunc = (double)(int64_t)(src2); // truncate
double trunc = (double)(int64_t)(src2); //truncate
return src1 * exp2(trunc);
}
};
template<>
struct OpHandlers<IR::OP_F80BCDSTORE> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
bool Negative = Src1.Sign;
@@ -349,7 +393,7 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
uint64_t Tmp = Src1;
X80SoftFloat Rv;
uint8_t* BCD = reinterpret_cast<uint8_t*>(&Rv);
uint8_t *BCD = reinterpret_cast<uint8_t*>(&Rv);
memset(BCD, 0, 10);
for (size_t i = 0; i < 9; ++i) {
@@ -379,10 +423,11 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
template<>
struct OpHandlers<IR::OP_F80BCDLOAD> {
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src) {
FEXCORE_PRESERVE_ALL_ATTR
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src) {
LoadDeferredFCW(NewFCW);
uint8_t* Src1 = reinterpret_cast<uint8_t*>(&Src);
uint64_t BCD {};
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
// Each 4bit split is a digit
// Only 0-9 is supported, A-F results in undefined data
@@ -68,7 +68,8 @@ namespace FEXCore::CPU {
//
// 5. Done.
//
template<IR::IROps Op>
struct OpHandlers {};
template <IR::IROps Op>
struct OpHandlers {
};
} // namespace FEXCore::CPU
@@ -10,23 +10,23 @@
namespace FEXCore::CPU {
template<typename R, typename... Args>
static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_UNKNOWN, (void*)fn, HandlerIndex, false};
}
template<>
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_I16_F64, (void*)fn, HandlerIndex, false};
}
template<>
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_I16_F64_F64, (void*)fn, HandlerIndex, false};
}
void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4);
Info[Core::OPINDEX_F80CVT_8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8);
Info[Core::OPINDEX_F80CVTINT_2] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2);
@@ -55,8 +55,8 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
Info[Core::OPINDEX_F80COS] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80COS>::handle);
Info[Core::OPINDEX_F80XTRACT_EXP] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle);
Info[Core::OPINDEX_F80XTRACT_SIG] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle);
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle);
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle);
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle);
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle);
// Binary
Info[Core::OPINDEX_F80ADD] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80ADD>::handle);
@@ -85,123 +85,126 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
Info[Core::OPINDEX_VPCMPISTRX] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle);
}
bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::IROp_Header* IROp, FallbackInfo* Info) {
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
uint8_t OpSize = IROp->Size;
switch (IROp->Op) {
case IR::OP_F80CVTTO: {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch(IROp->Op) {
case IR::OP_F80CVTTO: {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch (Op->SrcSize) {
case 4: {
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
return true;
}
case 8: {
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CVT: {
switch (OpSize) {
case 4: {
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
return true;
}
case 8: {
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CVTINT: {
auto Op = IROp->C<IR::IROp_F80CVTInt>();
switch (OpSize) {
case 2: {
if (Op->Truncate) {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
SupportsPreserveAllABI};
} else {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, SupportsPreserveAllABI};
switch (Op->SrcSize) {
case 4: {
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
case 8: {
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
return true;
break;
}
case 4: {
if (Op->Truncate) {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
SupportsPreserveAllABI};
} else {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, SupportsPreserveAllABI};
case IR::OP_F80CVT: {
switch (OpSize) {
case 4: {
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
case 8: {
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
return true;
break;
}
case 8: {
if (Op->Truncate) {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
SupportsPreserveAllABI};
} else {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, SupportsPreserveAllABI};
case IR::OP_F80CVTINT: {
auto Op = IROp->C<IR::IROp_F80CVTInt>();
switch (OpSize) {
case 2: {
if (Op->Truncate) {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
else {
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
return true;
}
case 4: {
if (Op->Truncate) {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
else {
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
return true;
}
case 8: {
if (Op->Truncate) {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
else {
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, FEXCORE_HAS_PRESERVE_ALL_ATTR};
}
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
case IR::OP_F80CMP: {
auto Op = IROp->C<IR::IROp_F80Cmp>();
static constexpr std::array handlers{
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
};
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags), FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
case IR::OP_F80CVTTOINT: {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->SrcSize) {
case 2: {
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
case 4: {
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
}
break;
}
break;
}
case IR::OP_F80CMP: {
auto Op = IROp->C<IR::IROp_F80Cmp>();
static constexpr std::array handlers {
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
};
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags),
SupportsPreserveAllABI};
return true;
}
case IR::OP_F80CVTTOINT: {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->SrcSize) {
case 2: {
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
return true;
#define COMMON_UNARY_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, FEXCORE_HAS_PRESERVE_ALL_ATTR}; \
return true; \
}
case 4: {
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
return true;
#define COMMON_BINARY_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, FEXCORE_HAS_PRESERVE_ALL_ATTR}; \
return true; \
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
#define COMMON_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
return true; \
}
break;
}
#define COMMON_UNARY_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
return true; \
}
#define COMMON_BINARY_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
return true; \
}
#define COMMON_F64_OP(OP) \
case IR::OP_F64##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
return true; \
}
// Unary
COMMON_UNARY_X87_OP(ROUND)
@@ -239,20 +242,20 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
COMMON_F64_OP(FPREM)
COMMON_F64_OP(SCALE)
// SSE4.2 Fallbacks
case IR::OP_VPCMPESTRX:
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX,
SupportsPreserveAllABI};
return true;
case IR::OP_VPCMPISTRX:
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
return true;
// SSE4.2 Fallbacks
case IR::OP_VPCMPESTRX:
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
case IR::OP_VPCMPISTRX:
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, FEXCORE_HAS_PRESERVE_ALL_ATTR};
return true;
default: break;
default:
break;
}
return false;
}
} // namespace FEXCore::CPU
}
@@ -15,9 +15,9 @@ namespace FEXCore::CPU {
template<>
struct OpHandlers<IR::OP_VPCMPESTRX> {
enum class AggregationOp {
EqualAny = 0b00,
Ranges = 0b01,
EqualEach = 0b10,
EqualAny = 0b00,
Ranges = 0b01,
EqualEach = 0b10,
EqualOrdered = 0b11,
};
@@ -35,7 +35,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
NegativeMasked,
};
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(uint64_t RAX, uint64_t RDX, __uint128_t lhs, __uint128_t rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t handle(uint64_t RAX, uint64_t RDX, __uint128_t lhs, __uint128_t rhs, uint16_t control) {
// Subtract by 1 in order to make validity limits 0-based
const auto valid_lhs = GetExplicitLength(RAX, control) - 1;
const auto valid_rhs = GetExplicitLength(RDX, control) - 1;
@@ -44,31 +45,33 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
}
// Main PCMPXSTRX algorithm body. Allows for reuse with both implicit and explicit length variants.
FEXCORE_PRESERVE_ALL_ATTR static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
const uint32_t aggregation = PerformAggregation(lhs, valid_lhs, rhs, valid_rhs, control);
const int32_t upper_limit = (16 >> (control & 1)) - 1;
// Bits are arranged as:
// Bit #: 3 2 1 0
// [SF | ZF | CF | OF]
// [OF | CF | SF | ZF]
uint32_t flags = 0;
flags |= (valid_rhs < upper_limit) ? 0b0100 : 0b0000;
flags |= (valid_lhs < upper_limit) ? 0b1000 : 0b0000;
flags |= (valid_rhs < upper_limit) ? 0b01 : 0b00;
flags |= (valid_lhs < upper_limit) ? 0b10 : 0b00;
const uint32_t result = HandlePolarity(aggregation, control, upper_limit, valid_rhs);
if (result != 0) {
flags |= 0b0010;
flags |= 0b0100;
}
if ((result & 1) != 0) {
flags |= 0b0001;
flags |= 0b1000;
}
// We track the flags in the usual NZCV bit position so we can msr them
// later. Avoids handling flags natively in JIT.
return result | (flags << 28);
// We tack the flags on top of the result to avoid needing to handle
// multiple return values in the JITs.
return result | (flags << 16);
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetExplicitLength(uint64_t reg, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t GetExplicitLength(uint64_t reg, uint16_t control) {
// Bit 8 controls whether or not the reg value is 64-bit or 32-bit.
int64_t value = 0;
if (((control >> 8) & 1) != 0) {
@@ -91,50 +94,62 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
return std::abs(static_cast<int>(value));
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetElement(const __uint128_t& vec, int32_t index, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t GetElement(const __uint128_t& vec, int32_t index, uint16_t control) {
const auto* vec_ptr = reinterpret_cast<const uint8_t*>(&vec);
// Control bits [1:0] define the data type being dealt with.
switch (static_cast<SourceData>(control & 0b11)) {
case SourceData::U8: return static_cast<int32_t>(vec_ptr[index]);
case SourceData::U8:
return static_cast<int32_t>(vec_ptr[index]);
case SourceData::U16: {
uint16_t value {};
uint16_t value{};
std::memcpy(&value, vec_ptr + (sizeof(uint16_t) * static_cast<size_t>(index)), sizeof(value));
return value;
}
case SourceData::S8: return static_cast<int8_t>(vec_ptr[index]);
case SourceData::S8:
return static_cast<int8_t>(vec_ptr[index]);
case SourceData::S16:
default: {
int16_t value {};
int16_t value{};
std::memcpy(&value, vec_ptr + (sizeof(int16_t) * static_cast<size_t>(index)), sizeof(value));
return value;
}
}
}
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
PerformAggregation(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t PerformAggregation(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
switch (static_cast<AggregationOp>((control >> 2) & 0b11)) {
case AggregationOp::EqualAny: return HandleEqualAny(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::Ranges: return HandleRanges(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::EqualEach: return HandleEqualEach(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::EqualAny:
return HandleEqualAny(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::Ranges:
return HandleRanges(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::EqualEach:
return HandleEqualEach(lhs, valid_lhs, rhs, valid_rhs, control);
case AggregationOp::EqualOrdered:
default: return HandleEqualOrdered(lhs, valid_lhs, rhs, valid_rhs, control);
default:
return HandleEqualOrdered(lhs, valid_lhs, rhs, valid_rhs, control);
}
}
FEXCORE_PRESERVE_ALL_ATTR static uint32_t HandlePolarity(uint32_t value, uint16_t control, int upper_limit, int valid_rhs) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandlePolarity(uint32_t value, uint16_t control, int upper_limit, int valid_rhs) {
switch (static_cast<Polarity>((control >> 4) & 0b11)) {
case Polarity::Negative: return value ^ ((2U << upper_limit) - 1);
case Polarity::NegativeMasked: return value ^ ((1U << (valid_rhs + 1)) - 1);
case Polarity::Positive:
case Polarity::PositiveMasked:
default:
// Both positive masking and positive polarity are documented
// as both being equivalent to "IntRes2 = IntRes1", where IntRes1
// is our 'value' parameter, so we don't need to do anything in
// these cases except return the same value.
return value;
case Polarity::Negative:
return value ^ ((2U << upper_limit) - 1);
case Polarity::NegativeMasked:
return value ^ ((1U << (valid_rhs + 1)) - 1);
case Polarity::Positive:
case Polarity::PositiveMasked:
default:
// Both positive masking and positive polarity are documented
// as both being equivalent to "IntRes2 = IntRes1", where IntRes1
// is our 'value' parameter, so we don't need to do anything in
// these cases except return the same value.
return value;
}
}
@@ -160,8 +175,10 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
// │
// 'c' match ────────┘
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
HandleEqualAny(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandleEqualAny(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
uint32_t result = 0;
for (int j = valid_rhs; j >= 0; j--) {
@@ -205,8 +222,10 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
// │
// 'Z' >= 'z' && 'A' <= 'z' ──────────┘
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
HandleRanges(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandleRanges(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
uint32_t result = 0;
for (int j = valid_rhs; j >= 0; j--) {
@@ -256,8 +275,10 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
// │
// 'a' == 'a' ──────────┘
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
HandleEqualEach(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandleEqualEach(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
const auto upper_limit = (16 >> (control & 1)) - 1;
const auto max_valid = std::max(valid_lhs, valid_rhs);
const auto min_valid = std::min(valid_lhs, valid_rhs);
@@ -309,8 +330,10 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
// │
// At index 0 ──────────┘
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
HandleEqualOrdered(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t HandleEqualOrdered(const __uint128_t& lhs, int32_t valid_lhs,
const __uint128_t& rhs, int32_t valid_rhs,
uint16_t control) {
const auto upper_limit = (16 >> (control & 1)) - 1;
// Edge case!
@@ -322,7 +345,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
}
uint32_t result = 0;
const int initial = valid_rhs == upper_limit ? valid_rhs : valid_rhs - valid_lhs;
const int initial = valid_rhs == upper_limit ? valid_rhs
: valid_rhs - valid_lhs;
for (int j = initial; j >= 0; j--) {
result <<= 1;
@@ -355,7 +379,8 @@ struct OpHandlers<IR::OP_VPCMPISTRX> {
// to be the max length possible for the given character size specified
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
//
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
// Subtract by 1 in order to make validity limits 0-based
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
@@ -363,7 +388,8 @@ struct OpHandlers<IR::OP_VPCMPISTRX> {
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
}
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
FEXCORE_PRESERVE_ALL_ATTR
static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
const auto is_using_words = (control & 1) != 0;
@@ -373,7 +399,7 @@ struct OpHandlers<IR::OP_VPCMPISTRX> {
const auto get_word = [data_u8](int32_t index) {
const auto* src = data_u8 + (index * sizeof(uint16_t));
uint16_t element {};
uint16_t element{};
std::memcpy(&element, src, sizeof(uint16_t));
return element;
};
@@ -7,43 +7,44 @@
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::IR {
class IRListView;
struct IROp_Header;
} // namespace FEXCore::IR
class IRListView;
struct IROp_Header;
}
namespace FEXCore::CPU {
enum FallbackABI {
FABI_UNKNOWN,
FABI_F80_I16_F32,
FABI_F80_I16_F64,
FABI_F80_I16_I16,
FABI_F80_I16_I32,
FABI_F32_I16_F80,
FABI_F64_I16_F80,
FABI_F64_I16_F64,
FABI_F64_I16_F64_F64,
FABI_I16_I16_F80,
FABI_I32_I16_F80,
FABI_I64_I16_F80,
FABI_I64_I16_F80_F80,
FABI_F80_I16_F80,
FABI_F80_I16_F80_F80,
FABI_I32_I64_I64_I128_I128_I16,
FABI_I32_I128_I128_I16,
};
enum FallbackABI {
FABI_UNKNOWN,
FABI_F80_I16_F32,
FABI_F80_I16_F64,
FABI_F80_I16_I16,
FABI_F80_I16_I32,
FABI_F32_I16_F80,
FABI_F64_I16_F80,
FABI_F64_I16_F64,
FABI_F64_I16_F64_F64,
FABI_I16_I16_F80,
FABI_I32_I16_F80,
FABI_I64_I16_F80,
FABI_I64_I16_F80_F80,
FABI_F80_I16_F80,
FABI_F80_I16_F80_F80,
FABI_I32_I64_I64_I128_I128_I16,
FABI_I32_I128_I128_I16,
};
struct FallbackInfo {
FallbackABI ABI;
void* fn;
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
bool SupportsPreserveAllABI;
};
struct FallbackInfo {
FallbackABI ABI;
void *fn;
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
bool SupportsPreserveAllABI;
};
class InterpreterOps {
public:
static void FillFallbackIndexPointers(uint64_t* Info);
static bool GetFallbackHandler(bool SupportsPreserveAllABI, const IR::IROp_Header* IROp, FallbackInfo* Info);
};
class InterpreterOps {
public:
static void FillFallbackIndexPointers(uint64_t *Info);
static bool GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info);
};
} // namespace FEXCore::CPU
File diff suppressed because it is too large. Load diff
@@ -13,19 +13,21 @@ namespace FEXCore::CPU {
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
switch (Op) {
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
break;
default:
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
break;
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op)); break;
}
return ~0ULL;
}
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum) {
Relocation MoveABI {};
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum &Sum) {
Relocation MoveABI{};
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
auto CurrentCursor = GetCursorAddress<uint8_t *>();
MoveABI.NamedThunkMove.Offset = CurrentCursor - CodeData.BlockBegin;
MoveABI.NamedThunkMove.Symbol = Sum;
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
@@ -41,25 +43,22 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
Arm64JITCore::NamedSymbolLiteralPair Lit {
.Lit = Pointer,
.MoveABI =
{
.NamedSymbolLiteral =
{
.Header =
{
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
},
.Symbol = Op,
.Offset = 0,
},
.MoveABI = {
.NamedSymbolLiteral = {
.Header = {
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
},
.Symbol = Op,
.Offset = 0,
},
},
};
return Lit;
}
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
auto CurrentCursor = GetCursorAddress<uint8_t *>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
Bind(&Lit.Loc);
@@ -68,10 +67,10 @@ void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
}
void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant) {
Relocation MoveABI {};
Relocation MoveABI{};
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint8_t*>();
auto CurrentCursor = GetCursorAddress<uint8_t *>();
MoveABI.GuestRIPMove.Offset = CurrentCursor - CodeData.BlockBegin;
MoveABI.GuestRIPMove.GuestRIP = Constant;
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
@@ -80,54 +79,54 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
Relocations.emplace_back(MoveABI);
}
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
const char* EntryRelocations) {
size_t DataIndex {};
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
size_t DataIndex{};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
dc64(Pointer);
// Generate a literal so we can place it
dc64(Pointer);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
// XXX: Reenable once the JIT Object Cache is upstream
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
return false;
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
// XXX: Reenable once the JIT Object Cache is upstream
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
}
}
return true;
}
} // namespace FEXCore::CPU
}
@@ -6,11 +6,12 @@ $end_info$
*/
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CASPair>();
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Wrong element size");
@@ -28,20 +29,13 @@ DEF_OP(CASPair) {
caspal(EmitSize, TMP3, TMP4, Desired.first, Desired.second, MemSrc);
mov(EmitSize, Dst.first, TMP3.R());
mov(EmitSize, Dst.second, TMP4.R());
} else {
// Save NZCV so we don't have to mark this op as clobbering NZCV (the
// SupportsAtomics does not clobber atomics and this !SupportsAtomics path
// is so slow it's not worth the complexity of splitting the IR op.). We
// clobber NZCV inside the hot loop and we can't replace cmp/ccmp/b.ne with
// something NZCV-preserving without requiring an extra instruction.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
}
else {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
ARMEmitter::SingleUseForwardLabel LoopExpected;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
Bind(&LoopTop);
// This instruction sequence must be synced with HandleCASPAL_Armv8.
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
cmp(EmitSize, TMP2, Expected.first);
ccmp(EmitSize, TMP3, Expected.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
@@ -53,23 +47,20 @@ DEF_OP(CASPair) {
b(&LoopExpected);
Bind(&LoopNotExpected);
mov(EmitSize, Dst.first, TMP2.R());
mov(EmitSize, Dst.second, TMP3.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopNotExpected);
mov(EmitSize, Dst.first, TMP2.R());
mov(EmitSize, Dst.second, TMP3.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopExpected);
// Restore
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
}
}
DEF_OP(CAS) {
auto Op = IROp->C<IR::IROp_CAS>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
// DataSrc = *Src1
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
// This will write to memory! Careful!
@@ -78,21 +69,30 @@ DEF_OP(CAS) {
auto Desired = GetReg(Op->Desired.ID());
auto MemSrc = GetReg(Op->Addr.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
mov(EmitSize, TMP2, Expected);
casal(SubEmitSize, TMP2, Desired, MemSrc);
mov(EmitSize, GetReg(Node), TMP2.R());
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
ARMEmitter::SingleUseForwardLabel LoopExpected;
ARMEmitter::ForwardLabel LoopNotExpected;
ARMEmitter::ForwardLabel LoopExpected;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
if (IROp->Size == 1) {
if (OpSize == 1) {
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
} else if (IROp->Size == 2) {
}
else if (OpSize == 2) {
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTH, 0);
} else {
}
else {
cmp(EmitSize, TMP2, Expected);
}
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
@@ -101,26 +101,33 @@ DEF_OP(CAS) {
mov(EmitSize, GetReg(Node), Expected);
b(&LoopExpected);
Bind(&LoopNotExpected);
mov(EmitSize, GetReg(Node), TMP2.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopNotExpected);
mov(EmitSize, GetReg(Node), TMP2.R());
// exclusive monitor needs to be cleared here
// Might have hit the case where ldaxr was hit but stlxr wasn't
clrex();
Bind(&LoopExpected);
}
}
DEF_OP(AtomicAdd) {
auto Op = IROp->C<IR::IROp_AtomicAdd>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
staddl(SubEmitSize, Src, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -132,16 +139,23 @@ DEF_OP(AtomicAdd) {
DEF_OP(AtomicSub) {
auto Op = IROp->C<IR::IROp_AtomicSub>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
neg(EmitSize, TMP2, Src);
staddl(SubEmitSize, TMP2, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -153,16 +167,23 @@ DEF_OP(AtomicSub) {
DEF_OP(AtomicAnd) {
auto Op = IROp->C<IR::IROp_AtomicAnd>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
mvn(EmitSize, TMP2, Src);
stclrl(SubEmitSize, TMP2, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -174,15 +195,22 @@ DEF_OP(AtomicAnd) {
DEF_OP(AtomicCLR) {
auto Op = IROp->C<IR::IROp_AtomicCLR>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
stclrl(SubEmitSize, Src, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -194,15 +222,22 @@ DEF_OP(AtomicCLR) {
DEF_OP(AtomicOr) {
auto Op = IROp->C<IR::IROp_AtomicOr>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
stsetl(SubEmitSize, Src, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -214,15 +249,22 @@ DEF_OP(AtomicOr) {
DEF_OP(AtomicXor) {
auto Op = IROp->C<IR::IROp_AtomicXor>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
steorl(SubEmitSize, Src, MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -234,11 +276,17 @@ DEF_OP(AtomicXor) {
DEF_OP(AtomicNeg) {
auto Op = IROp->C<IR::IROp_AtomicNeg>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -255,16 +303,17 @@ DEF_OP(AtomicSwap) {
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = ConvertSize(IROp);
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ARMEmitter::SubRegSize::i8Bit;
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
mov(EmitSize, TMP2, Src);
ldswpal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -276,15 +325,22 @@ DEF_OP(AtomicSwap) {
DEF_OP(AtomicFetchAdd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -297,16 +353,23 @@ DEF_OP(AtomicFetchAdd) {
DEF_OP(AtomicFetchSub) {
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
neg(EmitSize, TMP2, Src);
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -319,16 +382,23 @@ DEF_OP(AtomicFetchSub) {
DEF_OP(AtomicFetchAnd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
mvn(EmitSize, TMP2, Src);
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -341,15 +411,22 @@ DEF_OP(AtomicFetchAnd) {
DEF_OP(AtomicFetchCLR) {
auto Op = IROp->C<IR::IROp_AtomicFetchCLR>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -362,15 +439,22 @@ DEF_OP(AtomicFetchCLR) {
DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -383,15 +467,22 @@ DEF_OP(AtomicFetchOr) {
DEF_OP(AtomicFetchXor) {
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -404,11 +495,17 @@ DEF_OP(AtomicFetchXor) {
DEF_OP(AtomicFetchNeg) {
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
const auto EmitSize = ConvertSize(IROp);
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
@@ -431,7 +528,8 @@ DEF_OP(TelemetrySetValue) {
if (CTX->HostFeatures.SupportsAtomics) {
stsetl(ARMEmitter::SubRegSize::i64Bit, TMP1, TMP2);
} else {
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
@@ -443,4 +541,5 @@ DEF_OP(TelemetrySetValue) {
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -7,18 +7,19 @@ $end_info$
#include "Interface/Context/Context.h"
#include "FEXCore/IR/IR.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "Interface/Core/InternalThreadState.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/MathUtils.h>
#include <Interface/HLE/Thunks/Thunks.h>
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(CallbackReturn) {
// spill back to CTX
@@ -52,40 +53,33 @@ DEF_OP(ExitFunction) {
uint64_t NewRIP;
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
#ifdef _M_ARM_64EC
if (RtlIsEcCode(NewRIP)) {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
LoadConstant(ARMEmitter::Size::i64Bit, EC_CALL_CHECKER_PC_REG, NewRIP);
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
br(TMP2);
} else {
#endif
ARMEmitter::SingleUseForwardLabel l_BranchHost;
ldr(TMP1, &l_BranchHost);
blr(TMP1);
ARMEmitter::ForwardLabel l_BranchHost;
ARMEmitter::ForwardLabel l_BranchGuest;
ldr(ARMEmitter::XReg::x0, &l_BranchHost);
blr(ARMEmitter::Reg::r0);
Bind(&l_BranchHost);
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
Bind(&l_BranchGuest);
dc64(NewRIP);
Bind(&l_BranchHost);
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
dc64(NewRIP);
#ifdef _M_ARM_64EC
}
#endif
} else {
ARMEmitter::SingleUseForwardLabel FullLookup;
ARMEmitter::ForwardLabel FullLookup;
auto RipReg = GetReg(Op->NewRIP.ID());
// L1 Cache
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg, LookupCache::L1_ENTRIES_MASK);
add(TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x3, ARMEmitter::ShiftType::LSL, 4);
// Note: sub+cbnz used over cmp+br to preserve flags.
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
sub(TMP1, TMP1, RipReg.X());
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x1, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
sub(TMP1, ARMEmitter::XReg::x0, RipReg.X());
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
br(TMP2);
br(ARMEmitter::Reg::r1);
Bind(&FullLookup);
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
@@ -101,27 +95,56 @@ DEF_OP(Jump) {
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
}
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
default:
LOGMAN_MSG_A_FMT("Unsupported compare type");
return ARMEmitter::Condition::CC_NV;
}
}
DEF_OP(CondJump) {
auto Op = IROp->C<IR::IROp_CondJump>();
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
if (Op->FromNZCV) {
b(MapCC(Op->Cond), TrueTargetLabel);
b(MapBranchCC(Op->Cond), TrueTargetLabel);
} else {
[[maybe_unused]] uint64_t Const;
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
uint64_t Const;
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ || Op->Cond.Val == FEXCore::IR::COND_NEQ, "CondJump: Expected simple "
"condition");
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ ||
Op->Cond.Val == FEXCore::IR::COND_NEQ,
"CondJump: Expected simple condition");
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
} else {
} else {
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
}
@@ -161,9 +184,7 @@ DEF_OP(Syscall) {
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
continue;
}
if (Op->Header.Args[i].IsInvalid()) continue;
str(GetReg(Op->Header.Args[i].ID()).X(), ARMEmitter::Reg::rsp, i * 8);
}
@@ -175,7 +196,8 @@ DEF_OP(Syscall) {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
} else {
}
else {
blr(ARMEmitter::Reg::r3);
}
@@ -213,19 +235,20 @@ DEF_OP(InlineSyscall) {
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS - 1> RegArgs = {
{ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5}};
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS-1> RegArgs = {{
ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5
}};
bool Intersects {};
bool Intersects{};
// We always need to spill x8 since we can't know if it is live at this SSA location
uint32_t SpillMask = 1U << 8;
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
break;
}
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
auto Reg = GetReg(Op->Header.Args[i].ID());
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
if (Reg == ARMEmitter::Reg::r8 ||
Reg == ARMEmitter::Reg::r4 ||
Reg == ARMEmitter::Reg::r5) {
SpillMask |= (1U << Reg.Idx());
Intersects = true;
@@ -249,10 +272,8 @@ DEF_OP(InlineSyscall) {
const auto EmitSize = CTX->Config.Is64BitMode() ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto EmitSubSize = CTX->Config.Is64BitMode() ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
if (Intersects) {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
break;
}
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
auto Reg = GetReg(Op->Header.Args[i].ID());
// In the case of intersection with x4, x5, or x8 then these are currently SRA
@@ -260,19 +281,21 @@ DEF_OP(InlineSyscall) {
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
if (Reg == ARMEmitter::Reg::r8) {
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP]));
} else if (Reg == ARMEmitter::Reg::r4) {
}
else if (Reg == ARMEmitter::Reg::r4) {
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX]));
} else if (Reg == ARMEmitter::Reg::r5) {
}
else if (Reg == ARMEmitter::Reg::r5) {
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RCX]));
} else {
}
else {
mov(EmitSize, RegArgs[i].R(), Reg);
}
}
} else {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
if (Op->Header.Args[i].IsInvalid()) {
break;
}
}
else {
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i].ID()));
}
@@ -313,7 +336,8 @@ DEF_OP(Thunk) {
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
}
else {
blr(ARMEmitter::Reg::r2);
}
@@ -324,65 +348,70 @@ DEF_OP(Thunk) {
DEF_OP(ValidateCode) {
auto Op = IROp->C<IR::IROp_ValidateCode>();
const auto* OldCode = (const uint8_t*)&Op->CodeOriginalLow;
const auto *OldCode = (const uint8_t *)&Op->CodeOriginalLow;
int len = Op->CodeLength;
int idx = 0;
LoadConstant(ARMEmitter::Size::i64Bit, GetReg(Node), 0);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Entry + Op->Offset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, 1);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Entry + Op->Offset);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 1);
const auto Dst = GetReg(Node);
while (len >= 8) {
ldr(ARMEmitter::XReg::x2, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i64Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
while (len >= 8)
{
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 8;
idx += 8;
}
while (len >= 4) {
ldr(ARMEmitter::WReg::w2, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
while (len >= 4)
{
ldr(ARMEmitter::WReg::w2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint32_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 4;
idx += 4;
}
while (len >= 2) {
ldrh(TMP3, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint16_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
while (len >= 2)
{
ldrh(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint16_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 2;
idx += 2;
}
while (len >= 1) {
ldrb(TMP3, TMP1, idx);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint8_t*)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
while (len >= 1)
{
ldrb(ARMEmitter::Reg::r2, ARMEmitter::Reg::r0, idx);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, *(const uint8_t *)(OldCode + idx));
cmp(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3);
csel(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::r1, ARMEmitter::Condition::CC_EQ);
len -= 1;
idx += 1;
}
}
DEF_OP(ThreadRemoveCodeEntry) {
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
// Arguments are passed as follows:
// X0: Thread
// X1: RIP
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
PushDynamicRegsAndLR(TMP1);
SpillStaticRegs(TMP1);
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
} else {
}
else {
blr(ARMEmitter::Reg::r2);
}
FillStaticRegs();
@@ -394,32 +423,21 @@ DEF_OP(ThreadRemoveCodeEntry) {
DEF_OP(CPUID) {
auto Op = IROp->C<IR::IROp_CPUID>();
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function.ID()));
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf.ID()));
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
PushDynamicRegsAndLR(TMP1);
SpillStaticRegs(TMP1);
// x0 = CPUID Handler
// x1 = CPUID Function
// x2 = CPUID Leaf
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction));
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, TMP2);
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, TMP3);
}
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, GetReg(Op->Leaf.ID()));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<__uint128_t, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
} else {
blr(ARMEmitter::Reg::r3);
}
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
mov(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Reg::r1);
else {
blr(ARMEmitter::Reg::r3);
}
FillStaticRegs();
@@ -429,30 +447,26 @@ DEF_OP(CPUID) {
// Results are in x0, x1
// Results want to be in a i64v2 vector
auto Dst = GetRegPair(Node);
mov(ARMEmitter::Size::i64Bit, Dst.first, TMP1);
mov(ARMEmitter::Size::i64Bit, Dst.second, TMP2);
mov(ARMEmitter::Size::i64Bit, Dst.first, ARMEmitter::Reg::r0);
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
}
DEF_OP(XGetBV) {
auto Op = IROp->C<IR::IROp_XGetBV>();
PushDynamicRegsAndLR(TMP4);
SpillStaticRegs(TMP4);
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
PushDynamicRegsAndLR(TMP1);
SpillStaticRegs(TMP1);
// x0 = CPUID Handler
// x1 = XCR Function
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
} else {
blr(ARMEmitter::Reg::r2);
}
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
else {
blr(ARMEmitter::Reg::r2);
}
FillStaticRegs();
@@ -462,9 +476,10 @@ DEF_OP(XGetBV) {
// Results are in x0
// Results want to be in a i32v2 vector
auto Dst = GetRegPair(Node);
mov(ARMEmitter::Size::i32Bit, Dst.first, TMP1);
lsr(ARMEmitter::Size::i64Bit, Dst.second, TMP1, 32);
mov(ARMEmitter::Size::i32Bit, Dst.first, ARMEmitter::Reg::r0);
lsr(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r0, 32);
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -5,10 +5,11 @@ tags: backend|arm64
$end_info$
*/
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(VInsGPR) {
const auto Op = IROp->C<IR::IROp_VInsGPR>();
const auto OpSize = IROp->Size;
@@ -17,7 +18,11 @@ DEF_OP(VInsGPR) {
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto SubEmitSize = ConvertSubRegSize8(IROp);
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
const auto ElementsPer128Bit = 16 / ElementSize;
const auto Dst = GetVReg(Node);
@@ -60,7 +65,7 @@ DEF_OP(VInsGPR) {
// Inserts the GPR value into the given V register.
// Also automatically adjusts the index in the case of using the
// moved upper lane.
const auto Insert = [&](const ARMEmitter::VRegister& reg, int index) {
const auto Insert = [&](const FEXCore::ARMEmitter::VRegister& reg, int index) {
if (InUpperLane) {
index -= ElementsPer128Bit;
}
@@ -89,17 +94,21 @@ DEF_OP(VCastFromGPR) {
auto Src = GetReg(Op->Src.ID());
switch (Op->Header.ElementSize) {
case 1:
uxtb(ARMEmitter::Size::i32Bit, TMP1, Src);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
break;
case 2:
uxth(ARMEmitter::Size::i32Bit, TMP1, Src);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
break;
case 4: fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src); break;
case 8: fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src); break;
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
case 1:
uxtb(ARMEmitter::Size::i32Bit, TMP1, Src);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
break;
case 2:
uxth(ARMEmitter::Size::i32Bit, TMP1, Src);
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
break;
case 4:
fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src);
break;
case 8:
fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src);
break;
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
}
}
@@ -111,7 +120,16 @@ DEF_OP(VDupFromGPR) {
const auto Src = GetReg(Op->Src.ID());
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto SubEmitSize = ConvertSubRegSize8(IROp);
const auto ElementSize = IROp->ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1,
"Unexpected {} element size: {}", __func__, ElementSize);
const auto SubEmitSize =
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (HostSupportsSVE256 && Is256Bit) {
dup(SubEmitSize, Dst.Z(), Src);
@@ -130,33 +148,34 @@ DEF_OP(Float_FromGPR_S) {
auto Src = GetReg(Op->Src.ID());
switch (Conv) {
case 0x0204: { // Half <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.H(), Src);
break;
}
case 0x0208: { // Half <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.H(), Src);
break;
}
case 0x0404: { // Float <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.S(), Src);
break;
}
case 0x0408: { // Float <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.S(), Src);
break;
}
case 0x0804: { // Double <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.D(), Src);
break;
}
case 0x0808: { // Double <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.D(), Src);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}", Conv, ElementSize, Op->SrcElementSize);
break;
case 0x0204: { // Half <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.H(), Src);
break;
}
case 0x0208: { // Half <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.H(), Src);
break;
}
case 0x0404: { // Float <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.S(), Src);
break;
}
case 0x0408: { // Float <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.S(), Src);
break;
}
case 0x0804: { // Double <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.D(), Src);
break;
}
case 0x0808: { // Double <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.D(), Src);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
Conv, ElementSize, Op->SrcElementSize);
break;
}
}
@@ -168,31 +187,31 @@ DEF_OP(Float_FToF) {
auto Src = GetVReg(Op->Scalar.ID());
switch (Conv) {
case 0x0204: { // Half <- Float
fcvt(Dst.H(), Src.S());
break;
}
case 0x0208: { // Half <- Double
fcvt(Dst.H(), Src.D());
break;
}
case 0x0402: { // Float <- Half
fcvt(Dst.S(), Src.H());
break;
}
case 0x0802: { // Double <- Half
fcvt(Dst.D(), Src.H());
break;
}
case 0x0804: { // Double <- Float
fcvt(Dst.D(), Src.S());
break;
}
case 0x0408: { // Float <- Double
fcvt(Dst.S(), Src.D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
case 0x0204: { // Half <- Float
fcvt(Dst.H(), Src.S());
break;
}
case 0x0208: { // Half <- Double
fcvt(Dst.H(), Src.D());
break;
}
case 0x0402: { // Float <- Half
fcvt(Dst.S(), Src.H());
break;
}
case 0x0802: { // Double <- Half
fcvt(Dst.D(), Src.H());
break;
}
case 0x0804: { // Double <- Float
fcvt(Dst.D(), Src.S());
break;
}
case 0x0408: { // Float <- Double
fcvt(Dst.S(), Src.D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
}
}
@@ -201,9 +220,13 @@ DEF_OP(Vector_SToF) {
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto SubEmitSize = ConvertSubRegSize248(IROp);
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
if (HostSupportsSVE256 && Is256Bit) {
@@ -213,15 +236,19 @@ DEF_OP(Vector_SToF) {
if (OpSize == ElementSize) {
if (ElementSize == 8) {
scvtf(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
} else if (ElementSize == 4) {
}
else if (ElementSize == 4) {
scvtf(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
} else {
}
else {
scvtf(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
}
} else {
}
else {
if (OpSize == 8) {
scvtf(SubEmitSize, Dst.D(), Vector.D());
} else {
}
else {
scvtf(SubEmitSize, Dst.Q(), Vector.Q());
}
}
@@ -233,9 +260,13 @@ DEF_OP(Vector_FToZS) {
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto SubEmitSize = ConvertSubRegSize248(IROp);
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
if (HostSupportsSVE256 && Is256Bit) {
@@ -245,15 +276,19 @@ DEF_OP(Vector_FToZS) {
if (OpSize == ElementSize) {
if (ElementSize == 8) {
fcvtzs(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
} else if (ElementSize == 4) {
}
else if (ElementSize == 4) {
fcvtzs(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
} else {
}
else {
fcvtzs(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
}
} else {
}
else {
if (OpSize == 8) {
fcvtzs(SubEmitSize, Dst.D(), Vector.D());
} else {
}
else {
fcvtzs(SubEmitSize, Dst.Q(), Vector.Q());
}
}
@@ -264,8 +299,13 @@ DEF_OP(Vector_FToS) {
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto SubEmitSize = ConvertSubRegSize248(IROp);
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
@@ -280,7 +320,8 @@ DEF_OP(Vector_FToS) {
if (OpSize == 8) {
frinti(SubEmitSize, Dst.D(), Vector.D());
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
} else {
}
else {
frinti(SubEmitSize, Dst.Q(), Vector.Q());
fcvtzs(SubEmitSize, Dst.Q(), Dst.Q());
}
@@ -292,10 +333,14 @@ DEF_OP(Vector_FToF) {
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto SubEmitSize = ConvertSubRegSize248(IROp);
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
@@ -316,75 +361,46 @@ DEF_OP(Vector_FToF) {
const auto Mask = PRED_TMP_32B.Merging();
switch (Conv) {
case 0x0402: { // Float <- Half
zip1(ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Vector.Z(), Vector.Z());
fcvtlt(ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Dst.Z());
break;
}
case 0x0804: { // Double <- Float
zip1(ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Vector.Z(), Vector.Z());
fcvtlt(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Dst.Z());
break;
}
case 0x0204: { // Half <- Float
fcvtnt(ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Mask, Vector.Z());
uzp2(ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
break;
}
case 0x0408: { // Float <- Double
fcvtnt(ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Vector.Z());
uzp2(ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
case 0x0402: { // Float <- Half
zip1(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Vector.Z(), Vector.Z());
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Dst.Z());
break;
}
case 0x0804: { // Double <- Float
zip1(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Vector.Z(), Vector.Z());
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Dst.Z());
break;
}
case 0x0204: { // Half <- Float
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Mask, Vector.Z());
uzp2(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
break;
}
case 0x0408: { // Float <- Double
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Vector.Z());
uzp2(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
break;
}
} else {
switch (Conv) {
case 0x0402: // Float <- Half
case 0x0804: { // Double <- Float
fcvtl(SubEmitSize, Dst.D(), Vector.D());
break;
case 0x0402: // Float <- Half
case 0x0804: { // Double <- Float
fcvtl(SubEmitSize, Dst.D(), Vector.D());
break;
}
case 0x0204: // Half <- Float
case 0x0408: { // Float <- Double
fcvtn(SubEmitSize, Dst.D(), Vector.D());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
break;
}
case 0x0204: // Half <- Float
case 0x0408: { // Float <- Double
fcvtn(SubEmitSize, Dst.D(), Vector.D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
}
}
}
DEF_OP(VFCVTL2) {
const auto Op = IROp->C<IR::IROp_VFCVTL2>();
const auto SubEmitSize = ConvertSubRegSize248(IROp);
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
fcvtl2(SubEmitSize, Dst.D(), Vector.D());
}
DEF_OP(VFCVTN2) {
const auto Op = IROp->C<IR::IROp_VFCVTN2>();
const auto SubEmitSize = ConvertSubRegSize248(IROp);
const auto Dst = GetVReg(Node);
const auto VectorLower = GetVReg(Op->VectorLower.ID());
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
auto Lower = VectorLower;
if (Dst != VectorLower) {
mov(VTMP1.Q(), VectorLower.Q());
Lower = VTMP1;
}
fcvtn2(SubEmitSize, Lower.Q(), VectorUpper.Q());
if (Dst != VectorLower) {
mov(Dst.Q(), Lower.Q());
}
}
@@ -393,8 +409,12 @@ DEF_OP(Vector_FToI) {
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto SubEmitSize = ConvertSubRegSize248(IROp);
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
@@ -403,51 +423,82 @@ DEF_OP(Vector_FToI) {
const auto Mask = PRED_TMP_32B.Merging();
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
case FEXCore::IR::Round_Nearest.Val:
frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
case FEXCore::IR::Round_Towards_Zero.Val:
frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
case FEXCore::IR::Round_Host.Val:
frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z());
break;
}
} else {
const auto IsScalar = ElementSize == OpSize;
if (IsScalar) {
// Since we have multiple overloads of the same name (e.g.
// frinti having AdvSIMD, AdvSIMD scalar, and an SVE version),
// we can't just use a lambda without some seriously ugly casting.
// This is fairly self-contained otherwise.
#define ROUNDING_FN(name) \
if (ElementSize == 2) { \
name(Dst.H(), Vector.H()); \
} else if (ElementSize == 4) { \
name(Dst.S(), Vector.S()); \
} else if (ElementSize == 8) { \
name(Dst.D(), Vector.D()); \
} else { \
FEX_UNREACHABLE; \
}
// Since we have multiple overloads of the same name (e.g.
// frinti having AdvSIMD, AdvSIMD scalar, and an SVE version),
// we can't just use a lambda without some seriously ugly casting.
// This is fairly self-contained otherwise.
#define ROUNDING_FN(name) \
if (ElementSize == 2) { \
name(Dst.H(), Vector.H()); \
} else if (ElementSize == 4) { \
name(Dst.S(), Vector.S()); \
} else if (ElementSize == 8) { \
name(Dst.D(), Vector.D()); \
} else { \
FEX_UNREACHABLE; \
}
switch (Op->Round) {
case IR::Round_Nearest.Val: ROUNDING_FN(frintn); break;
case IR::Round_Negative_Infinity.Val: ROUNDING_FN(frintm); break;
case IR::Round_Positive_Infinity.Val: ROUNDING_FN(frintp); break;
case IR::Round_Towards_Zero.Val: ROUNDING_FN(frintz); break;
case IR::Round_Host.Val: ROUNDING_FN(frinti); break;
case IR::Round_Nearest.Val:
ROUNDING_FN(frintn);
break;
case IR::Round_Negative_Infinity.Val:
ROUNDING_FN(frintm);
break;
case IR::Round_Positive_Infinity.Val:
ROUNDING_FN(frintp);
break;
case IR::Round_Towards_Zero.Val:
ROUNDING_FN(frintz);
break;
case IR::Round_Host.Val:
ROUNDING_FN(frinti);
break;
}
#undef ROUNDING_FN
#undef ROUNDING_FN
} else {
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
case FEXCore::IR::Round_Nearest.Val:
frintn(SubEmitSize, Dst.Q(), Vector.Q());
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
frintm(SubEmitSize, Dst.Q(), Vector.Q());
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
frintp(SubEmitSize, Dst.Q(), Vector.Q());
break;
case FEXCore::IR::Round_Towards_Zero.Val:
frintz(SubEmitSize, Dst.Q(), Vector.Q());
break;
case FEXCore::IR::Round_Host.Val:
frinti(SubEmitSize, Dst.Q(), Vector.Q());
break;
}
}
}
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -5,10 +5,12 @@ tags: backend|arm64
$end_info$
*/
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(VAESImc) {
auto Op = IROp->C<IR::IROp_VAESImc>();
@@ -24,7 +26,8 @@ DEF_OP(VAESEnc) {
const auto State = GetVReg(Op->State.ID());
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
if (Dst == State && Dst != Key) {
// Optimal case in which Dst already contains the starting state.
@@ -32,7 +35,8 @@ DEF_OP(VAESEnc) {
aese(Dst.Q(), ZeroReg.Q());
aesmc(Dst.Q(), Dst.Q());
eor(Dst.Q(), Dst.Q(), Key.Q());
} else {
}
else {
mov(VTMP1.Q(), State.Q());
aese(VTMP1, ZeroReg.Q());
aesmc(VTMP1, VTMP1);
@@ -49,14 +53,16 @@ DEF_OP(VAESEncLast) {
const auto State = GetVReg(Op->State.ID());
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
if (Dst == State && Dst != Key) {
// Optimal case in which Dst already contains the starting state.
// This matches the common case of XMM AES.
aese(Dst.Q(), ZeroReg.Q());
eor(Dst.Q(), Dst.Q(), Key.Q());
} else {
}
else {
mov(VTMP1.Q(), State.Q());
aese(VTMP1, ZeroReg.Q());
eor(Dst.Q(), VTMP1.Q(), Key.Q());
@@ -72,7 +78,8 @@ DEF_OP(VAESDec) {
const auto State = GetVReg(Op->State.ID());
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
if (Dst == State && Dst != Key) {
// Optimal case in which Dst already contains the starting state.
@@ -80,7 +87,8 @@ DEF_OP(VAESDec) {
aesd(Dst.Q(), ZeroReg.Q());
aesimc(Dst.Q(), Dst.Q());
eor(Dst.Q(), Dst.Q(), Key.Q());
} else {
}
else {
mov(VTMP1.Q(), State.Q());
aesd(VTMP1, ZeroReg.Q());
aesimc(VTMP1, VTMP1);
@@ -97,14 +105,16 @@ DEF_OP(VAESDecLast) {
const auto State = GetVReg(Op->State.ID());
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
if (Dst == State && Dst != Key) {
// Optimal case in which Dst already contains the starting state.
// This matches the common case of XMM AES.
aesd(Dst.Q(), ZeroReg.Q());
eor(Dst.Q(), Dst.Q(), Key.Q());
} else {
}
else {
mov(VTMP1.Q(), State.Q());
aesd(VTMP1, ZeroReg.Q());
eor(Dst.Q(), VTMP1.Q(), Key.Q());
@@ -139,7 +149,8 @@ DEF_OP(VAESKeyGenAssist) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->RCON) << 32);
dup(ARMEmitter::SubRegSize::i64Bit, VTMP2.Q(), TMP1);
eor(Dst.Q(), Dst.Q(), VTMP2.Q());
} else {
}
else {
tbl(Dst.Q(), Dst.Q(), Swizzle.Q());
}
}
@@ -152,11 +163,19 @@ DEF_OP(CRC32) {
const auto Src2 = GetReg(Op->Src2.ID());
switch (Op->SrcSize) {
case 1: crc32cb(Dst.W(), Src1.W(), Src2.W()); break;
case 2: crc32ch(Dst.W(), Src1.W(), Src2.W()); break;
case 4: crc32cw(Dst.W(), Src1.W(), Src2.W()); break;
case 8: crc32cx(Dst.X(), Src1.X(), Src2.X()); break;
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
case 1:
crc32cb(Dst.W(), Src1.W(), Src2.W());
break;
case 2:
crc32ch(Dst.W(), Src1.W(), Src2.W());
break;
case 4:
crc32cw(Dst.W(), Src1.W(), Src2.W());
break;
case 8:
crc32cx(Dst.X(), Src1.X(), Src2.X());
break;
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
}
}
@@ -178,10 +197,11 @@ DEF_OP(VSha256U0) {
if (Dst == Src1) {
sha256su0(Dst, Src2);
} else {
}
else {
mov(VTMP1.Q(), Src1.Q());
sha256su0(VTMP1, Src2);
mov(Dst.Q(), VTMP1.Q());
mov(Dst.Q(), Src1.Q());
}
}
@@ -189,14 +209,17 @@ DEF_OP(PCLMUL) {
const auto Op = IROp->C<IR::IROp_PCLMUL>();
const auto OpSize = IROp->Size;
const auto Dst = GetVReg(Node);
const auto Dst = GetVReg(Node);
const auto Src1 = GetVReg(Op->Src1.ID());
const auto Src2 = GetVReg(Op->Src2.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
"Currently only supports 128-bit operations.");
switch (Op->Selector) {
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
case 0b00000000:
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D());
break;
case 0b00000001:
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
@@ -205,10 +228,14 @@ DEF_OP(PCLMUL) {
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
break;
case 0b00010001: pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q()); break;
default: LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector); break;
case 0b00010001:
pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q());
break;
default:
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
break;
}
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -8,11 +8,12 @@ $end_info$
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(GetHostFlag) {
auto Op = IROp->C<IR::IROp_GetHostFlag>();
ubfx(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Value.ID()), Op->Flag, 1);
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
File diff suppressed because it is too large. Load diff
+103 -224
View File
@@ -8,84 +8,69 @@ $end_info$
#pragma once
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IntrusiveIRList.h"
#include "Interface/IR/RegisterAllocationData.h"
#include <aarch64/assembler-aarch64.h>
#include <aarch64/disasm-aarch64.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <CodeEmitter/Emitter.h>
#include <array>
#include <cstdint>
#include <utility>
#include <variant>
namespace FEXCore::Core {
struct InternalThreadState;
struct InternalThreadState;
}
namespace FEXCore::CPU {
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
public:
explicit Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
explicit Arm64JITCore(FEXCore::Context::ContextImpl *ctx,
FEXCore::Core::InternalThreadState *Thread);
~Arm64JITCore() override;
[[nodiscard]]
fextl::string GetName() override {
return "JIT";
}
[[nodiscard]] fextl::string GetName() override { return "JIT"; }
[[nodiscard]]
CPUBackend::CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
const FEXCore::IR::RegisterAllocationData* RAData) override;
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData) override;
[[nodiscard]]
void* MapRegion(void* HostPtr, uint64_t, uint64_t) override {
return HostPtr;
}
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
[[nodiscard]]
bool NeedsOpDispatch() override {
return true;
}
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
void ClearCache() override;
void ClearRelocations() override {
Relocations.clear();
}
void ClearRelocations() override { Relocations.clear(); }
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
const bool HostSupportsSVE128 {};
const bool HostSupportsSVE256 {};
const bool HostSupportsAVX256 {};
const bool HostSupportsRPRES {};
const bool HostSupportsAFP {};
const bool HostSupportsSVE128{};
const bool HostSupportsSVE256{};
const bool HostSupportsRPRES{};
const bool HostSupportsAFP{};
ARMEmitter::BiDirectionalLabel* PendingTargetLabel;
FEXCore::Context::ContextImpl* CTX;
const FEXCore::IR::IRListView* IR;
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
FEXCore::Context::ContextImpl *CTX;
FEXCore::IR::IRListView const *IR;
uint64_t Entry;
CPUBackend::CompiledCode CodeData {};
CPUBackend::CompiledCode CodeData{};
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
[[nodiscard]]
ARMEmitter::Register GetReg(IR::NodeID Node) const {
[[nodiscard]] FEXCore::ARMEmitter::Register GetReg(IR::NodeID Node) const {
const auto Reg = GetPhys(Node);
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
@@ -99,8 +84,7 @@ private:
FEX_UNREACHABLE;
}
[[nodiscard]]
ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
[[nodiscard]] FEXCore::ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
const auto Reg = GetPhys(Node);
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
@@ -114,20 +98,17 @@ private:
FEX_UNREACHABLE;
}
[[nodiscard]]
std::pair<ARMEmitter::Register, ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
[[nodiscard]] std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
const auto Reg = GetPhys(Node);
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
return std::make_pair(GeneralRegisters[Reg.Reg], GeneralRegisters[Reg.Reg + 1]);
return GeneralPairRegisters[Reg.Reg];
}
[[nodiscard]]
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
[[nodiscard]]
IR::PhysicalRegister GetPhys(IR::NodeID Node) const {
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const {
auto PhyReg = RAData->GetNodeRegister(Node);
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
@@ -135,144 +116,38 @@ private:
return PhyReg;
}
[[nodiscard]]
ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
uint64_t Const;
if (IsInlineConstant(Src, &Const)) {
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
return ARMEmitter::Reg::zr;
} else {
return GetReg(Src.ID());
}
}
// Converts IR-base shift type to ARMEmitter shift type.
// Will be a no-op, only a type conversion since the two definitions match.
[[nodiscard]]
ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
[[nodiscard]] ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
ARMEmitter::ShiftType::ROR;
ARMEmitter::ShiftType::ROR;
}
[[nodiscard]]
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
return Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
}
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
[[nodiscard]]
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
LOGMAN_THROW_AA_FMT(Op->Size == 4 || Op->Size == 8, "Invalid size");
return ConvertSize(Op);
}
[[nodiscard]]
ARMEmitter::SubRegSize ConvertSubRegSize16(uint8_t ElementSize) {
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
return ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
ARMEmitter::SubRegSize::i128Bit;
}
[[nodiscard]]
ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
return ConvertSubRegSize16(Op->ElementSize);
}
[[nodiscard]]
ARMEmitter::SubRegSize ConvertSubRegSize8(uint8_t ElementSize) {
LOGMAN_THROW_AA_FMT(ElementSize != 16, "Invalid size");
return ConvertSubRegSize16(ElementSize);
}
[[nodiscard]]
ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
return ConvertSubRegSize8(Op->ElementSize);
}
[[nodiscard]]
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
LOGMAN_THROW_AA_FMT(Op->ElementSize != 8, "Invalid size");
return ConvertSubRegSize8(Op);
}
[[nodiscard]]
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
LOGMAN_THROW_AA_FMT(Op->ElementSize != 1, "Invalid size");
return ConvertSubRegSize8(Op);
}
[[nodiscard]]
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
return ARMEmitter::ToVectorSizePair(ConvertSubRegSize16(Op));
}
[[nodiscard]]
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
LOGMAN_THROW_AA_FMT(Op->ElementSize != 16, "Invalid size");
return ConvertSubRegSizePair16(Op);
}
[[nodiscard]]
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
LOGMAN_THROW_AA_FMT(Op->ElementSize != 1, "Invalid size");
return ConvertSubRegSizePair8(Op);
}
[[nodiscard]]
ARMEmitter::Condition MapCC(IR::CondClassType Cond) {
switch (Cond.Val) {
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
case FEXCore::IR::COND_VS:
case FEXCore::IR::COND_VC:
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
}
}
[[nodiscard]]
bool IsFPR(IR::NodeID Node) const;
[[nodiscard]]
bool IsGPR(IR::NodeID Node) const;
[[nodiscard]]
bool IsGPRPair(IR::NodeID Node) const;
[[nodiscard]]
ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
[[nodiscard]] FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize,
FEXCore::ARMEmitter::Register Base,
IR::OrderedNodeWrapper Offset,
IR::MemOffsetType OffsetType,
uint8_t OffsetScale);
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
// the limits of the scalar plus immediate variant of SVE load/stores.
//
// TMP1 is safe to use again once this memory operand is used with its
// equivalent loads or stores that this was called for.
[[nodiscard]]
ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
[[nodiscard]] FEXCore::ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize,
FEXCore::ARMEmitter::Register Base,
IR::OrderedNodeWrapper Offset,
IR::MemOffsetType OffsetType,
uint8_t OffsetScale);
[[nodiscard]]
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
[[nodiscard]]
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
struct LiveRange {
uint32_t Begin;
@@ -281,85 +156,89 @@ private:
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
IR::RegisterAllocationPass* RAPass;
const IR::RegisterAllocationData* RAData;
FEXCore::Core::DebugData* DebugData;
IR::RegisterAllocationPass *RAPass;
IR::RegisterAllocationData *RAData;
FEXCore::Core::DebugData *DebugData;
void ResetStack();
/**
* @name Relocations
* @{ */
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief A literal pair relocation object for named symbol literals
*/
struct NamedSymbolLiteralPair {
ARMEmitter::ForwardLabel Loc;
uint64_t Lit;
Relocation MoveABI {};
};
/**
* @brief A literal pair relocation object for named symbol literals
*/
struct NamedSymbolLiteralPair {
ARMEmitter::ForwardLabel Loc;
uint64_t Lit;
Relocation MoveABI{};
};
/**
* @brief Inserts a thunk relocation
*
* @param Reg - The GPR to move the thunk handler in to
* @param Sum - The hash of the thunk
*/
void InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum);
/**
* @brief Inserts a thunk relocation
*
* @param Reg - The GPR to move the thunk handler in to
* @param Sum - The hash of the thunk
*/
void InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum &Sum);
/**
* @brief Inserts a guest GPR move relocation
*
* @param Reg - The GPR to move the guest RIP in to
* @param Constant - The guest RIP that will be relocated
*/
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
/**
* @brief Inserts a guest GPR move relocation
*
* @param Reg - The GPR to move the guest RIP in to
* @param Constant - The guest RIP that will be relocated
*/
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
/**
* @brief Inserts a named symbol as a literal in memory
*
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
*
* @param Op The named symbol to place
*
* @return A temporary `NamedSymbolLiteralPair`
*/
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief Inserts a named symbol as a literal in memory
*
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
*
* @param Op The named symbol to place
*
* @return A temporary `NamedSymbolLiteralPair`
*/
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief Place the named symbol literal relocation in memory
*
* @param Lit - Which literal to place
*/
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit);
/**
* @brief Place the named symbol literal relocation in memory
*
* @param Lit - Which literal to place
*/
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
fextl::vector<FEXCore::CPU::Relocation> Relocations;
fextl::vector<FEXCore::CPU::Relocation> Relocations;
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
/** @} */
uint32_t SpillSlots {};
using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::NodeID Node);
uint32_t SpillSlots{};
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
// Runtime selection;
// Load and store register style.
OpType RT_LoadRegister;
OpType RT_StoreRegister;
// Load and store TSO memory style
OpType RT_LoadMemTSO;
OpType RT_StoreMemTSO;
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
// Dynamic Dispatcher supporting operations
DEF_OP(LoadRegisterSRA);
DEF_OP(StoreRegisterSRA);
DEF_OP(ParanoidLoadMemTSO);
DEF_OP(ParanoidStoreMemTSO);
File diff suppressed because it is too large. Load diff
@@ -10,13 +10,14 @@ $end_info$
#endif
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "FEXCore/Debug/InternalThreadState.h"
#include <FEXCore/Core/SignalDelegator.h>
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(GuestOpcode) {
auto Op = IROp->C<IR::IROp_GuestOpcode>();
@@ -27,10 +28,16 @@ DEF_OP(GuestOpcode) {
DEF_OP(Fence) {
auto Op = IROp->C<IR::IROp_Fence>();
switch (Op->Fence) {
case IR::Fence_Load.Val: dmb(ARMEmitter::BarrierScope::LD); break;
case IR::Fence_LoadStore.Val: dmb(ARMEmitter::BarrierScope::SY); break;
case IR::Fence_Store.Val: dmb(ARMEmitter::BarrierScope::ST); break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
case IR::Fence_Load.Val:
dmb(FEXCore::ARMEmitter::BarrierScope::LD);
break;
case IR::Fence_LoadStore.Val:
dmb(FEXCore::ARMEmitter::BarrierScope::SY);
break;
case IR::Fence_Store.Val:
dmb(FEXCore::ARMEmitter::BarrierScope::ST);
break;
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
}
}
@@ -48,7 +55,7 @@ DEF_OP(Break) {
.err_code = Op->Reason.ErrorRegister,
};
uint64_t Constant {};
uint64_t Constant{};
memcpy(&Constant, &State, sizeof(State));
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant);
@@ -120,39 +127,6 @@ DEF_OP(SetRoundingMode) {
msr(ARMEmitter::SystemRegister::FPCR, TMP1);
}
DEF_OP(PushRoundingMode) {
auto Op = IROp->C<IR::IROp_PushRoundingMode>();
auto Dest = GetReg(Node);
// Save the old rounding mode
mrs(Dest, ARMEmitter::SystemRegister::FPCR);
// vixl simulator doesn't support anything beyond ties-to-even rounding
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
return;
}
// Insert the rounding flags, reversing the mode bits as above
if (Op->RoundMode == 3) {
orr(ARMEmitter::Size::i64Bit, TMP1, Dest, 3 << 22);
} else if (Op->RoundMode == 0) {
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(3 << 22));
} else {
LOGMAN_THROW_AA_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(Op->RoundMode << 22));
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, (Op->RoundMode == 2 ? 1 : 2) << 22);
}
// Now save the new FPCR
msr(ARMEmitter::SystemRegister::FPCR, TMP1);
}
DEF_OP(PopRoundingMode) {
auto Op = IROp->C<IR::IROp_PopRoundingMode>();
msr(ARMEmitter::SystemRegister::FPCR, GetReg(Op->FPCR.ID()));
}
DEF_OP(Print) {
auto Op = IROp->C<IR::IROp_Print>();
@@ -162,7 +136,8 @@ DEF_OP(Print) {
if (IsGPR(Op->Value.ID())) {
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue));
} else {
}
else {
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value.ID()), false);
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value.ID()), true);
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
@@ -171,10 +146,12 @@ DEF_OP(Print) {
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
if (IsGPR(Op->Value.ID())) {
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r3);
} else {
}
else {
GenerateIndirectRuntimeCall<void, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
}
} else {
}
else {
blr(ARMEmitter::Reg::r3);
}
@@ -254,7 +231,8 @@ DEF_OP(RDRAND) {
if (Op->GetReseeded) {
mrs(Dst.first, ARMEmitter::SystemRegister::RNDRRS);
} else {
}
else {
mrs(Dst.first, ARMEmitter::SystemRegister::RNDR);
}
@@ -267,4 +245,5 @@ DEF_OP(Yield) {
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
@@ -8,17 +8,15 @@ $end_info$
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
DEF_OP(ExtractElementPair) {
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
LOGMAN_THROW_AA_FMT(Op->Header.Size == 4 || Op->Header.Size == 8, "Invalid size");
const auto EmitSize = Op->Header.Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto Dst = GetReg(Node);
const auto Pair = GetRegPair(Op->Pair.ID());
const auto Src = Op->Element == 0 ? Pair.first : Pair.second;
if (Dst != Src) {
mov(ConvertSize48(IROp), Dst, Src);
}
const auto Src = GetRegPair(Op->Pair.ID());
const std::array<ARMEmitter::Register, 2> Regs = {Src.first, Src.second};
mov(EmitSize, GetReg(Node), Regs[Op->Element]);
}
DEF_OP(CreateElementPair) {
@@ -44,25 +42,6 @@ DEF_OP(CreateElementPair) {
}
}
DEF_OP(Copy) {
auto Op = IROp->C<IR::IROp_Copy>();
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Source.ID()));
}
DEF_OP(Swap1) {
auto Op = IROp->C<IR::IROp_Swap1>();
auto A = GetReg(Op->A.ID()), B = GetReg(Op->B.ID());
LOGMAN_THROW_AA_FMT(B == GetReg(Node), "Invariant");
mov(ARMEmitter::Size::i64Bit, TMP1, A);
mov(ARMEmitter::Size::i64Bit, A, B);
mov(ARMEmitter::Size::i64Bit, B, TMP1);
}
DEF_OP(Swap2) {
// Implemented above
}
#undef DEF_OP
} // namespace FEXCore::CPU
}
File diff suppressed because it is too large. Load diff
+7 -3
View File
@@ -1,7 +1,7 @@
// SPDX-License-Identifier: MIT
#pragma once
#include "Interface/Core/CPUBackend.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/fextl/memory.h>
namespace FEXCore::Context {
@@ -15,8 +15,12 @@ struct InternalThreadState;
namespace FEXCore::CPU {
class CPUBackend;
[[nodiscard]]
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::ContextImpl *ctx,
FEXCore::Core::InternalThreadState *Thread);
CPUBackendFeatures GetX86JITBackendFeatures();
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx,
FEXCore::Core::InternalThreadState *Thread);
CPUBackendFeatures GetArm64JITBackendFeatures();
} // namespace FEXCore::CPU
@@ -13,8 +13,8 @@ $end_info$
#include "Interface/Core/LookupCache.h"
namespace FEXCore {
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
: BlockLinks_mbr {fextl::pmr::get_default_resource()}
LookupCache::LookupCache(FEXCore::Context::ContextImpl *CTX)
: BlockLinks_mbr { fextl::pmr::get_default_resource() }
, ctx {CTX} {
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
@@ -78,4 +78,5 @@ void LookupCache::ClearCache() {
BlockList.clear();
}
} // namespace FEXCore
}
+35 -61
View File
@@ -13,9 +13,6 @@
#include <stddef.h>
#include <utility>
#include <mutex>
#ifdef _M_ARM_64EC
#include <winnt.h>
#endif
namespace FEXCore {
@@ -26,12 +23,12 @@ public:
uintptr_t GuestCode;
};
LookupCache(FEXCore::Context::ContextImpl* CTX);
LookupCache(FEXCore::Context::ContextImpl *CTX);
~LookupCache();
uintptr_t FindBlock(uint64_t Address) {
// Try L1, no lock needed
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
if (L1Entry.GuestCode == Address) {
return L1Entry.HostCode;
}
@@ -40,7 +37,7 @@ public:
std::lock_guard<std::recursive_mutex> lk(WriteLock);
// Try L2
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
const auto PageIndex = (Address & (VirtualMemSize -1)) >> 12;
const auto PageOffset = Address & (0x0FFF);
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
@@ -51,7 +48,8 @@ public:
// Find there pointer for the address in the blocks
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
if (BlockPointers[PageOffset].GuestCode == Address) {
if (BlockPointers[PageOffset].GuestCode == Address)
{
L1Entry.GuestCode = Address;
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
return L1Entry.HostCode;
@@ -70,24 +68,6 @@ public:
return 0;
}
#ifdef _M_ARM_64EC
bool CheckPageEC(uint64_t Address) {
if (!RtlIsEcCode(Address)) {
return false;
}
std::lock_guard<std::recursive_mutex> lk(WriteLock);
// Mark L2 entry for this page as EC by setting the LSB, this can then be
// checked by the dispatcher to see if it needs to perform a call/return to
// EC code.
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
Pointers[PageIndex] |= 1;
return true;
}
#endif
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
// Appends Block {Address} to CodePages [Start, Start + Length)
@@ -97,8 +77,8 @@ public:
bool rv = false;
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
auto& CodePage = CodePages[CurrentPage];
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length -1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
auto &CodePage = CodePages[CurrentPage];
rv |= CodePage.size() == 0;
CodePage.push_back(Address);
}
@@ -107,7 +87,7 @@ public:
}
// Adds to Guest -> Host code mapping
void AddBlockMapping(uint64_t Address, void* HostCode) {
void AddBlockMapping(uint64_t Address, void *HostCode) {
std::lock_guard<std::recursive_mutex> lk(WriteLock);
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
@@ -115,27 +95,27 @@ public:
// There is no need to update L1 or L2, they will get updated on first lookup
// However, adding to L1 here increases performance
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
L1Entry.GuestCode = Address;
L1Entry.HostCode = (uintptr_t)HostCode;
}
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
void Erase(uint64_t Address) {
std::lock_guard<std::recursive_mutex> lk(WriteLock);
// Sever any links to this block
auto lower = BlockLinks->lower_bound({Address, nullptr});
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
auto lower = BlockLinks->lower_bound({Address, 0});
auto upper = BlockLinks->upper_bound({Address, UINTPTR_MAX});
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
it->second(Frame, it->first.HostLink);
it->second();
}
// Remove from BlockList
BlockList.erase(Address);
// Do L1
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
if (L1Entry.GuestCode == Address) {
L1Entry.GuestCode = 0;
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
@@ -144,11 +124,11 @@ public:
}
// Do full map
Address = Address & (VirtualMemSize - 1);
Address = Address & (VirtualMemSize -1);
uint64_t PageOffset = Address & (0x0FFF);
Address >>= 12;
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uintptr_t *Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uint64_t LocalPagePointer = Pointers[Address];
if (!LocalPagePointer) {
// Page for this code didn't even exist, nothing to do
@@ -161,7 +141,8 @@ public:
BlockPointers[PageOffset].HostCode = 0;
}
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
void AddBlockLink(uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
std::lock_guard<std::recursive_mutex> lk(WriteLock);
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
@@ -170,20 +151,14 @@ public:
void ClearCache();
void ClearL2Cache();
uintptr_t GetL1Pointer() const {
return L1Pointer;
}
uintptr_t GetPagePointer() const {
return PagePointer;
}
uintptr_t GetVirtualMemorySize() const {
return VirtualMemSize;
}
uintptr_t GetL1Pointer() const { return L1Pointer; }
uintptr_t GetPagePointer() const { return PagePointer; }
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
// This needs to be taken before reads or writes to L2, L3, CodePages,
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
// may only happen during cross thread invalidation (::Erase).
// All other operations must be done from the owning thread.
@@ -195,17 +170,17 @@ public:
private:
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
// Do L1
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
L1Entry.GuestCode = Address;
L1Entry.HostCode = HostCode;
// Do ful map
auto FullAddress = Address;
Address = Address & (VirtualMemSize - 1);
Address = Address & (VirtualMemSize -1);
uint64_t PageOffset = Address & (0x0FFF);
Address >>= 12;
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uintptr_t *Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
uint64_t LocalPagePointer = Pointers[Address];
if (!LocalPagePointer) {
// We don't have a page pointer for this address
@@ -249,16 +224,15 @@ private:
struct BlockLinkTag {
uint64_t GuestDestination;
FEXCore::Context::ExitFunctionLinkData* HostLink;
uintptr_t HostLink;
bool operator<(const BlockLinkTag& other) const {
if (GuestDestination < other.GuestDestination) {
bool operator <(const BlockLinkTag& other) const {
if (GuestDestination < other.GuestDestination)
return true;
} else if (GuestDestination == other.GuestDestination) {
else if (GuestDestination == other.GuestDestination)
return HostLink < other.HostLink;
} else {
else
return false;
}
}
};
@@ -269,9 +243,9 @@ private:
//
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
using BlockLinksMapType = std::pmr::map<BlockLinkTag, std::function<void()>>;
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
BlockLinksMapType* BlockLinks;
BlockLinksMapType *BlockLinks;
fextl::robin_map<uint64_t, uint64_t> BlockList;
@@ -283,7 +257,7 @@ private:
size_t AllocateOffset {};
FEXCore::Context::ContextImpl* ctx;
uint64_t VirtualMemSize {};
FEXCore::Context::ContextImpl *ctx;
uint64_t VirtualMemSize{};
};
} // namespace FEXCore
}
@@ -6,81 +6,84 @@
#include <cstdint>
namespace FEXCore::CodeSerialize {
// If any of the config options mismatch on load then the cache won't be used
// Any of these will result in codegen changes
struct FEX_PACKED CodeObjectSerializationConfig {
// Cookie in the header of the file, isn't part of the config hash
uint64_t Cookie {};
// If any of the config options mismatch on load then the cache won't be used
// Any of these will result in codegen changes
struct
FEX_PACKED
CodeObjectSerializationConfig {
// Cookie in the header of the file, isn't part of the config hash
uint64_t Cookie{};
// Instructions per block configuration
int32_t MaxInstPerBlock {};
// Instructions per block configuration
int32_t MaxInstPerBlock{};
// Follows CPUID 4000_0001_EAX[3:0]
unsigned Arch : 4;
// Follows CPUID 4000_0001_EAX[3:0]
unsigned Arch : 4;
// Multiblock enabled
unsigned MultiBlock : 1;
// Multiblock enabled
unsigned MultiBlock : 1;
// Hardware TSO enabled
unsigned HardwareTSOEnabled : 1;
// Hardware TSO enabled
unsigned HardwareTSOEnabled : 1;
// TSO enabled
unsigned TSOEnabled : 1;
// TSO enabled
unsigned TSOEnabled : 1;
// ABI local flag unsafe optimization
unsigned ABILocalFlags : 1;
// ABI local flag unsafe optimization
unsigned ABILocalFlags : 1;
// Paranoid TSO mode enabled
unsigned ParanoidTSO : 1;
// Static register allocation enabled
unsigned SRA : 1;
// Guest code execution mode (We don't support live mode switch)
unsigned Is64BitMode : 1;
// Paranoid TSO mode enabled
unsigned ParanoidTSO : 1;
// SMC checks style
unsigned SMCChecks : 2;
// Guest code execution mode (We don't support live mode switch)
unsigned Is64BitMode : 1;
// x87 reduced precision
unsigned x87ReducedPrecision : 1;
// SMC checks style
unsigned SMCChecks : 2;
// Padding to remove uninitialized data warning from asan
// Shows remaining amount of bits available for config
unsigned _Pad : 19;
// x87 reduced precision
unsigned x87ReducedPrecision : 1;
bool operator==(const CodeObjectSerializationConfig& other) const {
return Cookie == other.Cookie && MaxInstPerBlock == other.MaxInstPerBlock && Arch == other.Arch && MultiBlock == other.MultiBlock &&
HardwareTSOEnabled == other.HardwareTSOEnabled && TSOEnabled == other.TSOEnabled && ABILocalFlags == other.ABILocalFlags &&
ParanoidTSO == other.ParanoidTSO && Is64BitMode == other.Is64BitMode && SMCChecks == other.SMCChecks &&
x87ReducedPrecision == other.x87ReducedPrecision;
}
static uint64_t GetHash(const CodeObjectSerializationConfig& other) {
// For < 64-bits of data just pack directly
// Skip the cookie
uint64_t Hash {};
Hash <<= 32;
Hash |= other.MaxInstPerBlock;
Hash <<= 1;
Hash |= other.Arch;
Hash <<= 1;
Hash |= other.MultiBlock;
Hash <<= 1;
Hash |= other.HardwareTSOEnabled;
Hash <<= 1;
Hash |= other.TSOEnabled;
Hash <<= 1;
Hash |= other.ABILocalFlags;
Hash <<= 1;
Hash |= other.ParanoidTSO;
Hash <<= 1;
Hash |= other.Is64BitMode;
Hash <<= 2;
Hash |= other.SMCChecks;
Hash <<= 1;
Hash |= other.x87ReducedPrecision;
return Hash;
}
};
// Padding to remove uninitialized data warning from asan
// Shows remaining amount of bits available for config
unsigned _Pad : 18;
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is "
"generated!");
} // namespace FEXCore::CodeSerialize
bool operator==(CodeObjectSerializationConfig const &other) const {
return Cookie == other.Cookie &&
MaxInstPerBlock == other.MaxInstPerBlock &&
Arch == other.Arch &&
MultiBlock == other.MultiBlock &&
HardwareTSOEnabled == other.HardwareTSOEnabled &&
TSOEnabled == other.TSOEnabled &&
ABILocalFlags == other.ABILocalFlags &&
SRA == other.SRA &&
ParanoidTSO == other.ParanoidTSO &&
Is64BitMode == other.Is64BitMode &&
SMCChecks == other.SMCChecks &&
x87ReducedPrecision == other.x87ReducedPrecision;
}
static uint64_t GetHash(CodeObjectSerializationConfig const &other) {
// For < 64-bits of data just pack directly
// Skip the cookie
uint64_t Hash{};
Hash <<= 32; Hash |= other.MaxInstPerBlock;
Hash <<= 1; Hash |= other.Arch;
Hash <<= 1; Hash |= other.MultiBlock;
Hash <<= 1; Hash |= other.HardwareTSOEnabled;
Hash <<= 1; Hash |= other.TSOEnabled;
Hash <<= 1; Hash |= other.ABILocalFlags;
Hash <<= 1; Hash |= other.SRA;
Hash <<= 1; Hash |= other.ParanoidTSO;
Hash <<= 1; Hash |= other.Is64BitMode;
Hash <<= 2; Hash |= other.SMCChecks;
Hash <<= 1; Hash |= other.x87ReducedPrecision;
return Hash;
}
};
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is generated!");
}
@@ -11,112 +11,120 @@
#include <xxhash.h>
namespace FEXCore::CodeSerialize {
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename) {
#ifndef _WIN32
// This function adds a named region *JOB* to our named region handler
// This needs to be as fast as possible to keep out of the way of the JIT
// This function adds a named region *JOB* to our named region handler
// This needs to be as fast as possible to keep out of the way of the JIT
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
if (!BaseFilename.empty()) {
// Create a new entry that once set up will be put in to our section object map
auto Entry = fextl::make_unique<CodeRegionEntry>(Base, Size, Offset, filename, NamedRegionHandler->DefaultCodeHeader(Base, Offset));
if (!BaseFilename.empty()) {
// Create a new entry that once set up will be put in to our section object map
auto Entry = fextl::make_unique<CodeRegionEntry>(
Base,
Size,
Offset,
filename,
NamedRegionHandler->DefaultCodeHeader(Base, Offset)
);
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
Entry->NamedJobRefCountMutex.lock();
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
Entry->NamedJobRefCountMutex.lock();
CodeRegionMapType::iterator EntryIterator;
CodeRegionMapType::iterator EntryIterator;
{
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
auto it = EntryMap.emplace(Base, std::move(Entry));
if (!it.second) {
// This happens when an application overwrites a previous region without unmapping what was there
// Lock this entry's Named job reference counter.
// Once this passes then we know that this section has been loaded.
it.first->second->NamedJobRefCountMutex.lock();
// Finalize anything the region needs to do first.
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
// munmap the file that was mapped
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
// Remove this entry from the unrelocated map as well
{
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
}
// Now overwrite the entry in the map
it = EntryMap.insert_or_assign(Base, std::move(Entry));
EntryIterator = it.first;
}
else {
// No overwrite, just insert
EntryIterator = it.first;
}
}
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
// do the loading for us.
//
// Create the async work queue job now so it can load
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
#endif
}
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
#ifndef _WIN32
// Removing a named region through the job system
// We need to find the entry that we are deleting first
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
{
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
auto it = EntryMap.find(Base);
if (it != EntryMap.end()) {
// Lock the job ref counter since we are erasing it
// Once this passes it will have been loaded
it->second->NamedJobRefCountMutex.lock();
auto it = EntryMap.emplace(Base, std::move(Entry));
if (!it.second) {
// This happens when an application overwrites a previous region without unmapping what was there
// Take the pointer from the map
EntryPointer = std::move(it->second);
// Lock this entry's Named job reference counter.
// Once this passes then we know that this section has been loaded.
it.first->second->NamedJobRefCountMutex.lock();
// We can now unmap the file data
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
// Finalize anything the region needs to do first.
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
// munmap the file that was mapped
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
// Remove this from the entry map
EntryMap.erase(it);
// Remove this entry from the unrelocated map as well
{
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
}
// Now overwrite the entry in the map
it = EntryMap.insert_or_assign(Base, std::move(Entry));
EntryIterator = it.first;
} else {
// No overwrite, just insert
EntryIterator = it.first;
}
}
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
// do the loading for us.
//
// Create the async work queue job now so it can load
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
#endif
}
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
#ifndef _WIN32
// Removing a named region through the job system
// We need to find the entry that we are deleting first
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
{
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
auto it = EntryMap.find(Base);
if (it != EntryMap.end()) {
// Lock the job ref counter since we are erasing it
// Once this passes it will have been loaded
it->second->NamedJobRefCountMutex.lock();
// Take the pointer from the map
EntryPointer = std::move(it->second);
// We can now unmap the file data
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
// Remove this from the entry map
EntryMap.erase(it);
// Remove this entry from the unrelocated map as well
{
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
else {
// Tried to remove something that wasn't in our code object tracking
return;
}
} else {
// Tried to remove something that wasn't in our code object tracking
return;
// Create the async work queue job now so it can finalize what it needs to do
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
// Create the async work queue job now so it can finalize what it needs to do
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
// Tell the async thread that it has work to do
CodeObjectCacheService->NotifyWork();
}
#endif
}
}
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
// XXX: Actually add serialization job
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
// XXX: Actually add serialization job
}
}
} // namespace FEXCore::CodeSerialize
@@ -7,67 +7,67 @@
#include <FEXCore/fextl/string.h>
namespace FEXCore::CodeSerialize {
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx) {
DefaultSerializationConfig.Cookie = CODE_COOKIE;
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl *ctx) {
DefaultSerializationConfig.Cookie = CODE_COOKIE;
// Initialize the Arch from CPUID
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
DefaultSerializationConfig.Arch = Arch;
// Initialize the Arch from CPUID
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
DefaultSerializationConfig.Arch = Arch;
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
}
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
DefaultSerializationConfig.SRA = ctx->Config.StaticRegisterAllocation;
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
}
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename,
const fextl::string& filename, bool Executable) {
// XXX: Add named region objects
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string &base_filename, const fextl::string &filename, bool Executable) {
// XXX: Add named region objects
// XXX: Until entry loading is complete just claim it is loaded
Entry->second->NamedJobRefCountMutex.unlock();
}
// XXX: Until entry loading is complete just claim it is loaded
Entry->second->NamedJobRefCountMutex.unlock();
}
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
// XXX: Remove named region objects
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
// XXX: Remove named region objects
// XXX: Until entry loading is complete just claim it is loaded
Entry->NamedJobRefCountMutex.unlock();
}
// XXX: Until entry loading is complete just claim it is loaded
Entry->NamedJobRefCountMutex.unlock();
}
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
// Walk through all of our jobs sequentially until the work queue is empty
while (NamedWorkQueueJobs.load()) {
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
// Walk through all of our jobs sequentially until the work queue is empty
while (NamedWorkQueueJobs.load()) {
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
{
// Lock the work queue mutex for a short moment and grab an item from the list
std::unique_lock lk {NamedWorkQueueMutex};
size_t WorkItems = WorkQueue.size();
if (WorkItems != 0) {
WorkItem = std::move(WorkQueue.front());
WorkQueue.pop();
{
// Lock the work queue mutex for a short moment and grab an item from the list
std::unique_lock lk {NamedWorkQueueMutex};
size_t WorkItems = WorkQueue.size();
if (WorkItems != 0) {
WorkItem = std::move(WorkQueue.front());
WorkQueue.pop();
}
// Atomically update the number of jobs
--NamedWorkQueueJobs;
}
// Atomically update the number of jobs
--NamedWorkQueueJobs;
}
if (WorkItem) {
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion *>(WorkItem.get());
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
}
if (WorkItem) {
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion*>(WorkItem.get());
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
}
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion*>(WorkItem.get());
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion *>(WorkItem.get());
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
}
}
}
}
}
} // namespace FEXCore::CodeSerialize
@@ -6,80 +6,80 @@
#include <FEXCore/Utils/Threads.h>
namespace {
static void* ThreadHandler(void* Arg) {
FEXCore::CodeSerialize::CodeObjectSerializeService* This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
This->ExecutionThread();
return nullptr;
static void* ThreadHandler(void *Arg) {
FEXCore::CodeSerialize::CodeObjectSerializeService *This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
This->ExecutionThread();
return nullptr;
}
}
} // namespace
namespace FEXCore::CodeSerialize {
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx)
: CTX {ctx}
, AsyncHandler {&NamedRegionHandler, this}
, NamedRegionHandler {ctx} {
Initialize();
}
void CodeObjectSerializeService::Shutdown() {
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
return;
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::ContextImpl *ctx)
: CTX {ctx}
, AsyncHandler { &NamedRegionHandler , this }
, NamedRegionHandler { ctx } {
Initialize();
}
WorkerThreadShuttingDown = true;
void CodeObjectSerializeService::Shutdown() {
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
return;
}
// Kick the working thread
WorkAvailable.NotifyAll();
WorkerThreadShuttingDown = true;
if (WorkerThread->joinable()) {
// Wait for worker thread to close down
WorkerThread->join(nullptr);
}
}
// Kick the working thread
WorkAvailable.NotifyAll();
void CodeObjectSerializeService::Initialize() {
// Add a canary so we don't crash on empty map iterator handling
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
FEXCore::Threads::SetSignalMask(OldMask);
}
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it) {
if (Base == ~0ULL) {
// Don't do closure on canary
return;
}
// XXX: Do code region closure
}
const CodeObjectFileSection* CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
// XXX: Actually fetch code objects from cache
return nullptr;
}
void CodeObjectSerializeService::ExecutionThread() {
// Set our thread name so we can see its relation
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
while (WorkerThreadShuttingDown.load() != true) {
// Wait for work
WorkAvailable.Wait();
// Handle named region async jobs first. Highest priority
NamedRegionHandler.HandleNamedRegionObjectJobs();
// XXX: Handle code serialization jobs second.
if (WorkerThread->joinable()) {
// Wait for worker thread to close down
WorkerThread->join(nullptr);
}
}
// Do final code region closures on thread shutdown
for (auto& it : AddressToEntryMap) {
DoCodeRegionClosure(it.first, it.second.get());
void CodeObjectSerializeService::Initialize() {
// Add a canary so we don't crash on empty map iterator handling
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
FEXCore::Threads::SetSignalMask(OldMask);
}
// Safely clear our maps now
AddressToEntryMap.clear();
UnrelocatedAddressToEntryMap.clear();
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it) {
if (Base == ~0ULL) {
// Don't do closure on canary
return;
}
// XXX: Do code region closure
}
CodeObjectFileSection const *CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
// XXX: Actually fetch code objects from cache
return nullptr;
}
void CodeObjectSerializeService::ExecutionThread() {
// Set our thread name so we can see its relation
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
while (WorkerThreadShuttingDown.load() != true) {
// Wait for work
WorkAvailable.Wait();
// Handle named region async jobs first. Highest priority
NamedRegionHandler.HandleNamedRegionObjectJobs();
// XXX: Handle code serialization jobs second.
}
// Do final code region closures on thread shutdown
for (auto &it : AddressToEntryMap) {
DoCodeRegionClosure(it.first, it.second.get());
}
// Safely clear our maps now
AddressToEntryMap.clear();
UnrelocatedAddressToEntryMap.clear();
}
}
} // namespace FEXCore::CodeSerialize
@@ -17,441 +17,445 @@
#include <shared_mutex>
namespace FEXCore::CodeSerialize {
// XXX: Does this need to be signal safe?
using CodeSerializationMutex = std::shared_mutex;
struct CodeSerializationData {};
// XXX: Does this need to be signal safe?
using CodeSerializationMutex = std::shared_mutex;
struct CodeSerializationData {
};
struct CodeObjectFileSection {
bool Serialized;
bool Invalid;
const CodeSerializationData* Data;
const char* HostCode;
uint64_t NumRelocations;
const char* Relocations;
};
/**
* @brief This is the file header that lives at the start of an object cache file
*
* This header is updated from multiple processes!
* Care must be taken to use OS locks when updating the file backing including this header
*/
struct CodeObjectSerializationHeader {
// The configuration that this file has
CodeObjectSerializationConfig Config;
// The original RIP that this object section was mapped at
uint64_t OriginalBase {};
// The original offset in to the file that this object section was loaded from
uint64_t OriginalOffset {};
// Total amount of code that should be in this file
uint64_t TotalCodeSize {};
// Used to reserve the TSL map
uint64_t NumCodeEntries {};
// The number of relocations that point to this section
uint64_t NumRelocationsTo {};
// Total relocations in this file
uint64_t TotalRelocationsCount {};
};
struct CodeRegionEntry {
/**
* @name Threaded initialization objects for the initial object creation
* @{ */
// Base address in memory where the code region is at
uint64_t Base {};
// Size of this code entry
uint64_t Size {};
// The offset inside the file that is mapped to Base
uint64_t Offset {};
// Filename of the object
fextl::string Filename {};
CodeObjectSerializationHeader EntryHeader {};
/** @} */
// The filename of the object cache for this entry
fextl::string ObjectEntrySourceFilename {};
// In the case of file corruption that we can detect, we can disable serialization early for an entry
// We should be resiliant to corruption but things happen
bool StillSerializing {true};
// Long lived FD for serialization if we have multiple jobs to serialize
// Bursts of code entries are common and this reduces file lock overhead
//
// Especially useful over network mounts where file locks are very slow
int CurrentSerializedFD {-1};
struct CodeObjectFileSection {
bool Serialized;
bool Invalid;
const CodeSerializationData *Data;
const char *HostCode;
uint64_t NumRelocations;
const char *Relocations;
};
/**
* @name Objects required to sync objects between threads
* @{ */
// Refcount for the number of outstanding code entries waiting to be written for this object section
CodeSerializationMutex ObjectJobRefCountMutex;
// Refcount for outstanding named object region entry loading itself
// Will block JIT code cache look up when this has a unique_lock held
CodeSerializationMutex NamedJobRefCountMutex;
/** @} */
/**
* @name Object Entry data management
* @{ */
/**
* @name This is the raw file data that we loaded from the code region entry file
* @{ */
char* CodeData {};
size_t FileSize {};
fextl::vector<CodeObjectFileSection> FileCodeSections;
/** @} */
// This per section map takes the most time to load and needs to be quick
// This is the map of all code segments for this entry
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap {};
/** @} */
// Default initialization
CodeRegionEntry() = default;
// Initializer specifically for threaded loading
CodeRegionEntry(uint64_t Base, uint64_t Size, uint64_t Offset, const fextl::string& Filename, const CodeObjectSerializationHeader& DefaultHeader)
: Base {Base}
, Size {Size}
, Offset {Offset}
, Filename {Filename}
, EntryHeader {DefaultHeader} {}
};
// Map type must use an interator that isn't invalidation on erase/insert
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
class NamedRegionObjectHandler;
class CodeObjectSerializeService;
class AsyncJobHandler final {
public:
/**
* @brief Structure containing all the data required to async serialize code objects
* @brief This is the file header that lives at the start of an object cache file
*
* This header is updated from multiple processes!
* Care must be taken to use OS locks when updating the file backing including this header
*/
struct SerializationJobData {
uint64_t GuestRIP; ///< The RIP for the guest
// XXX: Support multiblock
uint64_t GuestCodeLength; ///< The Guest's code length
uint64_t GuestCodeHash; ///< Hash of the guest code
struct CodeObjectSerializationHeader {
// The configuration that this file has
CodeObjectSerializationConfig Config;
// The original RIP that this object section was mapped at
uint64_t OriginalBase{};
// The original offset in to the file that this object section was loaded from
uint64_t OriginalOffset{};
// Total amount of code that should be in this file
uint64_t TotalCodeSize{};
// Used to reserve the TSL map
uint64_t NumCodeEntries{};
// The number of relocations that point to this section
uint64_t NumRelocationsTo{};
// Total relocations in this file
uint64_t TotalRelocationsCount{};
};
void* HostCodeBegin; ///< Host JIT code starting memory address
size_t HostCodeLength; ///< Host JIT code length
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
struct CodeRegionEntry {
/**
* @name Threaded initialization objects for the initial object creation
* @{ */
// Base address in memory where the code region is at
uint64_t Base{};
// This is the thread specific ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
// This way it will wait until the async job handler is complete with it.
CodeSerializationMutex* ThreadJobRefCount;
// Size of this code entry
uint64_t Size{};
// These are the reolocations for this serialization job
// Relatively small number of entries most of the time
fextl::vector<FEXCore::CPU::Relocation> Relocations;
// The offset inside the file that is mapped to Base
uint64_t Offset{};
// Filename of the object
fextl::string Filename{};
CodeObjectSerializationHeader EntryHeader{};
/** @} */
// The filename of the object cache for this entry
fextl::string ObjectEntrySourceFilename{};
// In the case of file corruption that we can detect, we can disable serialization early for an entry
// We should be resiliant to corruption but things happen
bool StillSerializing {true};
// Long lived FD for serialization if we have multiple jobs to serialize
// Bursts of code entries are common and this reduces file lock overhead
//
// Especially useful over network mounts where file locks are very slow
int CurrentSerializedFD {-1};
/**
* @name Objects filled in from the Code Object Serialization service when a job is added
* @name Objects required to sync objects between threads
* @{ */
// This is the code region's ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
CodeSerializationMutex* ObjectJobRefCountMutexPtr;
// Refcount for the number of outstanding code entries waiting to be written for this object section
CodeSerializationMutex ObjectJobRefCountMutex;
// This is the code region iterator to reduce the number of map lookups
// This will remain valid while jobs are outstanding for this region
CodeRegionMapType::iterator CodeRegionIterator;
// Refcount for outstanding named object region entry loading itself
// Will block JIT code cache look up when this has a unique_lock held
CodeSerializationMutex NamedJobRefCountMutex;
/** @} */
/**
* @name Object Entry data management
* @{ */
/**
* @name This is the raw file data that we loaded from the code region entry file
* @{ */
char *CodeData{};
size_t FileSize{};
fextl::vector<CodeObjectFileSection> FileCodeSections;
/** @} */
// This per section map takes the most time to load and needs to be quick
// This is the map of all code segments for this entry
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap{};
/** @} */
// Default initialization
CodeRegionEntry() = default;
// Initializer specifically for threaded loading
CodeRegionEntry(uint64_t Base,
uint64_t Size,
uint64_t Offset,
fextl::string const &Filename,
CodeObjectSerializationHeader const &DefaultHeader)
: Base {Base}
, Size {Size}
, Offset {Offset}
, Filename {Filename}
, EntryHeader {DefaultHeader} {
}
};
AsyncJobHandler(NamedRegionObjectHandler* NamedRegionHandler, CodeObjectSerializeService* CodeObjectCacheService)
: NamedRegionHandler {NamedRegionHandler}
, CodeObjectCacheService {CodeObjectCacheService} {}
// Map type must use an interator that isn't invalidation on erase/insert
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
protected:
friend class CodeObjectSerializeService;
friend class NamedRegionObjectHandler;
/**
* @name Async job submission functions
* @{ */
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename);
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
/** @} */
class NamedRegionObjectHandler;
class CodeObjectSerializeService;
/**
* @name Async named region handling
* @{ */
/**
* @brief The async named region jobs to handle.
*
* Only two, Code serialization goes in to a different queue.
*/
enum class NamedRegionJobType {
JOB_ADD_NAMED_REGION,
JOB_REMOVE_NAMED_REGION,
class AsyncJobHandler final {
public:
/**
* @brief Structure containing all the data required to async serialize code objects
*/
struct SerializationJobData {
uint64_t GuestRIP; ///< The RIP for the guest
// XXX: Support multiblock
uint64_t GuestCodeLength; ///< The Guest's code length
uint64_t GuestCodeHash; ///< Hash of the guest code
void *HostCodeBegin; ///< Host JIT code starting memory address
size_t HostCodeLength; ///< Host JIT code length
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
// This is the thread specific ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
// This way it will wait until the async job handler is complete with it.
CodeSerializationMutex *ThreadJobRefCount;
// These are the reolocations for this serialization job
// Relatively small number of entries most of the time
fextl::vector<FEXCore::CPU::Relocation> Relocations;
/**
* @name Objects filled in from the Code Object Serialization service when a job is added
* @{ */
// This is the code region's ref counter for outstanding jobs.
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
CodeSerializationMutex *ObjectJobRefCountMutexPtr;
// This is the code region iterator to reduce the number of map lookups
// This will remain valid while jobs are outstanding for this region
CodeRegionMapType::iterator CodeRegionIterator;
/** @} */
};
AsyncJobHandler(NamedRegionObjectHandler *NamedRegionHandler, CodeObjectSerializeService *CodeObjectCacheService)
: NamedRegionHandler {NamedRegionHandler}
, CodeObjectCacheService {CodeObjectCacheService} {}
protected:
friend class CodeObjectSerializeService;
friend class NamedRegionObjectHandler;
/**
* @name Async job submission functions
* @{ */
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename);
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
/** @} */
/**
* @name Async named region handling
* @{ */
/**
* @brief The async named region jobs to handle.
*
* Only two, Code serialization goes in to a different queue.
*/
enum class NamedRegionJobType {
JOB_ADD_NAMED_REGION,
JOB_REMOVE_NAMED_REGION,
};
class NamedRegionWorkItem {
public:
NamedRegionJobType GetType() const { return Type; }
protected:
friend class WorkItemAddNamedRegion;
NamedRegionWorkItem(NamedRegionJobType type)
: Type {type} {}
private:
NamedRegionJobType Type;
};
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
public:
WorkItemAddNamedRegion(const fextl::string &base, const fextl::string &filename, bool executable, CodeRegionMapType::iterator entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
, BaseFilename {base}
, Filename {filename}
, Executable {executable}
, Entry {entry}
{}
const fextl::string BaseFilename;
const fextl::string Filename;
bool Executable;
CodeRegionMapType::iterator Entry;
};
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
public:
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
, Base {base}
, Size {size}
, Entry {std::move(entry)} {}
uint64_t Base;
uint64_t Size;
fextl::unique_ptr<CodeRegionEntry> Entry;
};
/** @} */
private:
NamedRegionObjectHandler *NamedRegionHandler;
CodeObjectSerializeService *CodeObjectCacheService;
};
class NamedRegionWorkItem {
public:
NamedRegionJobType GetType() const {
return Type;
}
class NamedRegionObjectHandler final {
public:
NamedRegionObjectHandler(FEXCore::Context::ContextImpl *ctx);
protected:
friend class WorkItemAddNamedRegion;
NamedRegionWorkItem(NamedRegionJobType type)
: Type {type} {}
void HandleNamedRegionObjectJobs();
private:
NamedRegionJobType Type;
CodeObjectSerializationConfig const &GetDefaultSerializationConfig() const {
return DefaultSerializationConfig;
}
protected:
friend class AsyncJobHandler;
// Return a default code header based off the default serialization config
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
return CodeObjectSerializationHeader {
.Config = DefaultSerializationConfig,
.OriginalBase = Base,
.OriginalOffset = Offset,
.NumCodeEntries = 0,
.NumRelocationsTo = 0,
.TotalRelocationsCount = 0,
};
}
/**
* @brief Adds an asynchronous add named region work item to the object queue
*
* This adds the job that will do the loading of file resources and data tracking.
*/
void AsyncAddNamedRegionWorkItem(const fextl::string &base, const fextl::string &filename, bool executable, CodeRegionMapType::iterator entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion> (
base,
filename,
executable,
entry
));
++NamedWorkQueueJobs;
}
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion> (
Base,
Size,
std::move(Entry)
));
++NamedWorkQueueJobs;
}
private:
// Code version. If the code emission changes then this needs to increment
constexpr static uint32_t CODE_VERSION = 0x0;
// Default cookie header for the file header
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
// Code serialization config for our current process configuration
CodeObjectSerializationConfig DefaultSerializationConfig;
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
std::atomic<uint64_t> NamedWorkQueueJobs{};
// Mutex for ading new jobs to the work queue
std::mutex NamedWorkQueueMutex{};
// The job queue itself
// Jobs get consumed as a FIFO
// Jobs always get appended to the end
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue{};
/**
* @name Named Region object handling
* @{ */
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string &base_filename, const fextl::string &filename, bool Executable);
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
/** @} */
};
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
public:
WorkItemAddNamedRegion(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
, BaseFilename {base}
, Filename {filename}
, Executable {executable}
, Entry {entry} {}
const fextl::string BaseFilename;
const fextl::string Filename;
bool Executable;
CodeRegionMapType::iterator Entry;
/**
* @brief Context specific code object serialization class
*
* Contains everything required for FEXCore to serialize code objects
*/
class CodeObjectSerializeService final {
public:
CodeObjectSerializeService(FEXCore::Context::ContextImpl *ctx);
/**
* @brief Initialize the internal interface
*
* Is a public interface to allow the service to reinitialize after forking
*/
void Initialize();
/**
* @brief Safely shut down the Code Object serialization service.
*
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
*/
void Shutdown();
/**
* @name Async interface
* @{ */
/**
* @brief Loads a named region in to the code serialization service. As async as possible.
*
* @param Base - Virtual address that this named region is loaded
* @param Size - The size of the region
* @param Offset - The offset from the file
* @param filename - The filename itself
*/
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename) {
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
}
/**
* @brief Unloads a named region from the code serialization service. As async as possible.
*
* @param Base - Virtual address of the named region
* @param Size - The size of the region
*/
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
}
/**
* @brief Adds a code object serialization job. As async as possible.
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
*
* @param Data - A fully filled out struct containing all the code serialization
*/
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
}
/** @} */
/**
* @name Synchronous interface
* @{ */
/**
* @brief Synchronously waits for this thread's job queue to become empty.
*
* This is necessary for when a thread is shutting down
*
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
*/
static void WaitForEmptyJobQueue(CodeSerializationMutex *ThreadJobRefCount) {
// Once the shared mutex is empty this unique lock will be gained
std::unique_lock lk {*ThreadJobRefCount};
}
/**
* @brief Fetches object code from the Code Object Cache for JIT.
*
* @param GuestRIP - Which GuestRIP to search the cache for
*
* @return Data required for the JIT to relocate the Object code.
*/
CodeObjectFileSection const *FetchCodeObjectFromCache(uint64_t GuestRIP);
/** @} */
// Public for threading
void ExecutionThread();
protected:
friend class AsyncJobHandler;
/**
* @brief Safely closes out code object regions from the map
*
* @param it - iterator to do a closure on
*/
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it);
CodeSerializationMutex &GetEntryMapMutex() { return EntryMapMutex; }
CodeSerializationMutex &GetUnrelocatedEntryMapMutex() { return EntryMapMutex; }
CodeRegionMapType &GetEntryMap() { return AddressToEntryMap; }
CodeRegionPtrMapType &GetUnrelocatedEntryMap() { return UnrelocatedAddressToEntryMap; }
/**
* @brief Notify the async thread that it has work to do
*/
void NotifyWork() { WorkAvailable.NotifyOne(); }
private:
FEXCore::Context::ContextImpl *CTX;
Event WorkAvailable{};
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
std::atomic_bool WorkerThreadShuttingDown {false};
AsyncJobHandler AsyncHandler;
NamedRegionObjectHandler NamedRegionHandler;
// Mutex to hold when modifying the entry maps
CodeSerializationMutex EntryMapMutex;
CodeSerializationMutex UnrelocatedEntryMapMutex;
// Entry maps
CodeRegionMapType AddressToEntryMap;
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
};
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
public:
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
, Base {base}
, Size {size}
, Entry {std::move(entry)} {}
uint64_t Base;
uint64_t Size;
fextl::unique_ptr<CodeRegionEntry> Entry;
};
/** @} */
private:
NamedRegionObjectHandler* NamedRegionHandler;
CodeObjectSerializeService* CodeObjectCacheService;
};
class NamedRegionObjectHandler final {
public:
NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx);
void HandleNamedRegionObjectJobs();
const CodeObjectSerializationConfig& GetDefaultSerializationConfig() const {
return DefaultSerializationConfig;
}
protected:
friend class AsyncJobHandler;
// Return a default code header based off the default serialization config
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
return CodeObjectSerializationHeader {
.Config = DefaultSerializationConfig,
.OriginalBase = Base,
.OriginalOffset = Offset,
.NumCodeEntries = 0,
.NumRelocationsTo = 0,
.TotalRelocationsCount = 0,
};
}
/**
* @brief Adds an asynchronous add named region work item to the object queue
*
* This adds the job that will do the loading of file resources and data tracking.
*/
void AsyncAddNamedRegionWorkItem(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion>(base, filename, executable, entry));
++NamedWorkQueueJobs;
}
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
std::unique_lock lk {NamedWorkQueueMutex};
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion>(Base, Size, std::move(Entry)));
++NamedWorkQueueJobs;
}
private:
// Code version. If the code emission changes then this needs to increment
constexpr static uint32_t CODE_VERSION = 0x0;
// Default cookie header for the file header
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
// Code serialization config for our current process configuration
CodeObjectSerializationConfig DefaultSerializationConfig;
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
std::atomic<uint64_t> NamedWorkQueueJobs {};
// Mutex for ading new jobs to the work queue
std::mutex NamedWorkQueueMutex {};
// The job queue itself
// Jobs get consumed as a FIFO
// Jobs always get appended to the end
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue {};
/**
* @name Named Region object handling
* @{ */
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename, const fextl::string& filename, bool Executable);
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
/** @} */
};
/**
* @brief Context specific code object serialization class
*
* Contains everything required for FEXCore to serialize code objects
*/
class CodeObjectSerializeService final {
public:
CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx);
/**
* @brief Initialize the internal interface
*
* Is a public interface to allow the service to reinitialize after forking
*/
void Initialize();
/**
* @brief Safely shut down the Code Object serialization service.
*
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
*/
void Shutdown();
/**
* @name Async interface
* @{ */
/**
* @brief Loads a named region in to the code serialization service. As async as possible.
*
* @param Base - Virtual address that this named region is loaded
* @param Size - The size of the region
* @param Offset - The offset from the file
* @param filename - The filename itself
*/
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
}
/**
* @brief Unloads a named region from the code serialization service. As async as possible.
*
* @param Base - Virtual address of the named region
* @param Size - The size of the region
*/
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
}
/**
* @brief Adds a code object serialization job. As async as possible.
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
*
* @param Data - A fully filled out struct containing all the code serialization
*/
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
}
/** @} */
/**
* @name Synchronous interface
* @{ */
/**
* @brief Synchronously waits for this thread's job queue to become empty.
*
* This is necessary for when a thread is shutting down
*
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
*/
static void WaitForEmptyJobQueue(CodeSerializationMutex* ThreadJobRefCount) {
// Once the shared mutex is empty this unique lock will be gained
std::unique_lock lk {*ThreadJobRefCount};
}
/**
* @brief Fetches object code from the Code Object Cache for JIT.
*
* @param GuestRIP - Which GuestRIP to search the cache for
*
* @return Data required for the JIT to relocate the Object code.
*/
const CodeObjectFileSection* FetchCodeObjectFromCache(uint64_t GuestRIP);
/** @} */
// Public for threading
void ExecutionThread();
protected:
friend class AsyncJobHandler;
/**
* @brief Safely closes out code object regions from the map
*
* @param it - iterator to do a closure on
*/
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it);
CodeSerializationMutex& GetEntryMapMutex() {
return EntryMapMutex;
}
CodeSerializationMutex& GetUnrelocatedEntryMapMutex() {
return EntryMapMutex;
}
CodeRegionMapType& GetEntryMap() {
return AddressToEntryMap;
}
CodeRegionPtrMapType& GetUnrelocatedEntryMap() {
return UnrelocatedAddressToEntryMap;
}
/**
* @brief Notify the async thread that it has work to do
*/
void NotifyWork() {
WorkAvailable.NotifyOne();
}
private:
FEXCore::Context::ContextImpl* CTX;
Event WorkAvailable {};
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
std::atomic_bool WorkerThreadShuttingDown {false};
AsyncJobHandler AsyncHandler;
NamedRegionObjectHandler NamedRegionHandler;
// Mutex to hold when modifying the entry maps
CodeSerializationMutex EntryMapMutex;
CodeSerializationMutex UnrelocatedEntryMapMutex;
// Entry maps
CodeRegionMapType AddressToEntryMap;
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
};
} // namespace FEXCore::CodeSerialize
}
@@ -3,77 +3,77 @@
#include <FEXCore/IR/IR.h>
namespace FEXCore::CPU {
enum class RelocationTypes : uint8_t {
// 8 byte literal in memory for symbol
// Aligned to struct RelocNamedSymbolLiteral
RELOC_NAMED_SYMBOL_LITERAL,
enum class RelocationTypes : uint8_t {
// 8 byte literal in memory for symbol
// Aligned to struct RelocNamedSymbolLiteral
RELOC_NAMED_SYMBOL_LITERAL,
// Fixed size named thunk move
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocNamedThunkMove
RELOC_NAMED_THUNK_MOVE,
// Fixed size named thunk move
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocNamedThunkMove
RELOC_NAMED_THUNK_MOVE,
// Fixed size guest RIP move
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocGuestRIPMove
RELOC_GUEST_RIP_MOVE,
};
struct RelocationTypeHeader final {
RelocationTypes Type;
};
struct RelocNamedSymbolLiteral final {
enum class NamedSymbol : uint8_t {
///< Thread specific relocations
// JIT Literal pointers
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
// Fixed size guest RIP move
// 4 instruction constant generation on AArch64
// 64-bit mov on x86-64
// Aligned to struct RelocGuestRIPMove
RELOC_GUEST_RIP_MOVE,
};
RelocationTypeHeader Header {};
struct RelocationTypeHeader final {
RelocationTypes Type;
};
NamedSymbol Symbol;
struct RelocNamedSymbolLiteral final {
enum class NamedSymbol : uint8_t {
///< Thread specific relocations
// JIT Literal pointers
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
};
// Offset in to the code section to begin the relocation
uint64_t Offset {};
};
RelocationTypeHeader Header{};
struct RelocNamedThunkMove final {
RelocationTypeHeader Header {};
NamedSymbol Symbol;
// GPR index the constant is being moved to
uint8_t RegisterIndex;
// Offset in to the code section to begin the relocation
uint64_t Offset{};
};
// The thunk SHA256 hash
IR::SHA256Sum Symbol;
struct RelocNamedThunkMove final {
RelocationTypeHeader Header{};
// Offset in to the code section to begin the relocation
uint64_t Offset {};
};
// GPR index the constant is being moved to
uint8_t RegisterIndex;
struct RelocGuestRIPMove final {
RelocationTypeHeader Header {};
// The thunk SHA256 hash
IR::SHA256Sum Symbol;
// GPR index the constant is being moved to
uint8_t RegisterIndex;
// Offset in to the code section to begin the relocation
uint64_t Offset{};
};
// Offset in to the code section to begin the relocation
uint64_t Offset {};
struct RelocGuestRIPMove final {
RelocationTypeHeader Header{};
// The unrelocated RIP that is being moved
uint64_t GuestRIP;
};
// GPR index the constant is being moved to
uint8_t RegisterIndex;
union Relocation {
RelocationTypeHeader Header {};
// Offset in to the code section to begin the relocation
uint64_t Offset{};
RelocNamedSymbolLiteral NamedSymbolLiteral;
// This makes our union of relocations at least 48 bytes
// It might be more efficient to not use a union
RelocNamedThunkMove NamedThunkMove;
// The unrelocated RIP that is being moved
uint64_t GuestRIP;
};
RelocGuestRIPMove GuestRIPMove;
};
} // namespace FEXCore::CPU
union Relocation {
RelocationTypeHeader Header{};
RelocNamedSymbolLiteral NamedSymbolLiteral;
// This makes our union of relocations at least 48 bytes
// It might be more efficient to not use a union
RelocNamedThunkMove NamedThunkMove;
RelocGuestRIPMove GuestRIPMove;
};
}
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -8,6 +8,7 @@ $end_info$
#include "Interface/Core/X86Tables/X86Tables.h"
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Core/OpcodeDispatcher.h"
@@ -22,10 +23,10 @@ class OrderedNode;
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref RotatedNode {};
OrderedNode *RotatedNode{};
if (CTX->HostFeatures.SupportsSHA) {
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
@@ -34,7 +35,8 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
} else {
}
else {
// SHA1 extension missing, manually rotate.
// Emulate rotate.
auto ShiftLeft = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Dest, 30);
@@ -47,25 +49,25 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
}
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref NewVec = _VExtr(16, 8, Dest, Src, 1);
OrderedNode *NewVec = _VExtr(16, 8, Dest, Src, 1);
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
Ref Result = _VXor(16, 1, Dest, NewVec);
OrderedNode *Result = _VXor(16, 1, Dest, NewVec);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
// Do all the work without it.
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(OpSize::i32Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
// Shift the incoming source left by a 32-bit element, inserting Zeros.
// This could be slightly improved to use a VInsGPR with the zero register.
@@ -90,45 +92,48 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
}
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
using FnType = Ref (*)(OpDispatchBuilder&, Ref, Ref, Ref);
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(),
"Src1 needs to be literal here to indicate function and constants");
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
using FnType = OrderedNode* (*)(OpDispatchBuilder&, OrderedNode*, OrderedNode*, OrderedNode*);
const auto f0 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._Andn(OpSize::i32Bit, D, B));
};
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
const auto f1 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
};
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
return Self.BitwiseAtLeastTwo(B, C, D);
const auto f2 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._And(OpSize::i32Bit, B, D)), Self._And(OpSize::i32Bit, C, D));
};
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
const auto f3 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
};
constexpr std::array<uint32_t, 4> k_array {
constexpr std::array<uint32_t, 4> k_array{
0x5A827999U,
0x6ED9EBA1U,
0x8F1BBCDCU,
0xCA62C1D6U,
};
constexpr std::array<FnType, 4> fn_array {
f0,
f1,
f2,
f3,
constexpr std::array<FnType, 4> fn_array{
f0, f1, f2, f3,
};
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
const uint64_t Imm8 = Op->Src[1].Data.Literal.Value & 0b11;
const FnType Fn = fn_array[Imm8];
auto K = _Constant(32, k_array[Imm8]);
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto W0E = _VExtractToGPR(16, 4, Src, 3);
auto W1 = _VExtractToGPR(16, 4, Src, 2);
auto W2 = _VExtractToGPR(16, 4, Src, 1);
auto W3 = _VExtractToGPR(16, 4, Src, 0);
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
const auto Round0 = [&]() -> RoundResult {
auto A = _VExtractToGPR(16, 4, Dest, 3);
@@ -136,8 +141,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
auto C = _VExtractToGPR(16, 4, Dest, 1);
auto D = _VExtractToGPR(16, 4, Dest, 0);
auto A1 =
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W0E), K);
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W0E), K);
auto B1 = A;
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
auto D1 = C;
@@ -145,13 +149,9 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
return {A1, B1, C1, D1, E1};
};
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
// Kill W and E at the beginning
auto W = _VExtractToGPR(16, 4, Src, W_idx);
auto Q = _Add(OpSize::i32Bit, W, E);
auto ANext =
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), Q), K);
const auto Round1To3 = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C,
OrderedNode *D, OrderedNode *E, OrderedNode *W) -> RoundResult {
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W), E), K);
auto BNext = A;
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
auto DNext = C;
@@ -161,11 +161,11 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
};
auto [A1, B1, C1, D1, E1] = Round0();
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, W1);
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, W2);
auto Final = Round1To3(A3, B3, C3, D3, E3, W3);
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
auto Dest2 = _VInsGPR(16, 4, 2, Dest3, std::get<1>(Final));
auto Dest1 = _VInsGPR(16, 4, 1, Dest2, std::get<2>(Final));
auto Dest0 = _VInsGPR(16, 4, 0, Dest1, std::get<3>(Final));
@@ -174,17 +174,17 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
}
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result {};
OrderedNode *Result{};
if (CTX->HostFeatures.SupportsSHA) {
Result = _VSha256U0(Dest, Src);
} else {
const auto Sigma0 = [this](Ref W) -> Ref {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))),
_Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
}
else {
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
};
auto W4 = _VExtractToGPR(16, 4, Src, 0);
@@ -208,13 +208,12 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
}
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
const auto Sigma1 = [this](Ref W) -> Ref {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))),
_Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
const auto Sigma1 = [this](OrderedNode* W) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))), _Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
};
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
auto W14 = _VExtractToGPR(16, 4, Src, 2);
auto W15 = _VExtractToGPR(16, 4, Src, 3);
@@ -231,86 +230,76 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
StoreResult(FPRClass, Op, D0, -1);
}
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
// Returns whether at least 2/3 of A/B/C is true.
// Expressed as (A & (B | C)) | (B & C)
//
// Equivalent to expression in SHA calculations: (A & B) ^ (A & C) ^ (B & C)
auto And = _And(OpSize::i32Bit, B, C);
auto Or = _Or(OpSize::i32Bit, B, C);
return _Or(OpSize::i32Bit, _And(OpSize::i32Bit, A, Or), And);
}
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
const auto Ch = [this](OrderedNode *E, OrderedNode *F, OrderedNode *G) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
};
const auto Sigma0 = [this](Ref A) -> Ref {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), A, ShiftType::ROR, 13), A,
ShiftType::ROR, 22);
const auto Major = [this](OrderedNode *A, OrderedNode *B, OrderedNode *C) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, A, B), _And(OpSize::i32Bit, A, C)), _And(OpSize::i32Bit, B, C));
};
const auto Sigma1 = [this](Ref E) -> Ref {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), E, ShiftType::ROR, 11), E,
ShiftType::ROR, 25);
const auto Sigma0 = [this](OrderedNode *A) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), _Ror(OpSize::i32Bit, A, _Constant(32, 13))), _Ror(OpSize::i32Bit, A, _Constant(32, 22)));
};
const auto Sigma1 = [this](OrderedNode *E) -> OrderedNode* {
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), _Ror(OpSize::i32Bit, E, _Constant(32, 11))), _Ror(OpSize::i32Bit, E, _Constant(32, 25)));
};
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
// Hardcoded to XMM0
auto XMM0 = LoadXMMRegister(0);
auto E0 = _VExtractToGPR(16, 4, Src, 1);
auto F0 = _VExtractToGPR(16, 4, Src, 0);
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
Q0 = _Add(OpSize::i32Bit, Q0, H0);
auto A0 = _VExtractToGPR(16, 4, Src, 3);
auto B0 = _VExtractToGPR(16, 4, Src, 2);
auto C0 = _VExtractToGPR(16, 4, Dest, 3);
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
auto D0 = _VExtractToGPR(16, 4, Dest, 2);
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
auto E0 = _VExtractToGPR(16, 4, Src, 1);
auto F0 = _VExtractToGPR(16, 4, Src, 0);
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
auto WK1 = _VExtractToGPR(16, 4, XMM0, 1);
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
G0 = _VExtractToGPR(16, 4, Dest, 1);
Q1 = _Add(OpSize::i32Bit, Q1, G0);
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*,
OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
const auto Round = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C, OrderedNode *D,
OrderedNode *E, OrderedNode *F, OrderedNode *G, OrderedNode *H,
OrderedNode* WK) -> RoundResult {
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Ch(E, F, G), Sigma1(E)), WK), H), Major(A, B, C)), Sigma0(A));
auto BNext = A;
auto CNext = B;
auto DNext = C;
auto ENext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Ch(E, F, G), Sigma1(E)), WK), H), D);
auto FNext = E;
auto GNext = F;
auto HNext = G;
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
return {ANext, BNext, CNext, DNext, ENext, FNext, GNext, HNext};
};
// Rematerialize C0. As with G0.
C0 = _VExtractToGPR(16, 4, Dest, 3);
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
auto Res3 = _VInsGPR(16, 4, 3, Dest, A2);
auto Res2 = _VInsGPR(16, 4, 2, Res3, A1);
auto Res1 = _VInsGPR(16, 4, 1, Res2, E2);
auto Res0 = _VInsGPR(16, 4, 0, Res1, E1);
auto [A1, B1, C1, D1, E1, F1, G1, H1] = Round(A0, B0, C0, D0, E0, F0, G0, H0, WK0);
auto Final = Round(A1, B1, C1, D1, E1, F1, G1, H1, WK1);
auto Res3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
auto Res2 = _VInsGPR(16, 4, 2, Res3, std::get<1>(Final));
auto Res1 = _VInsGPR(16, 4, 1, Res2, std::get<4>(Final));
auto Res0 = _VInsGPR(16, 4, 0, Res1, std::get<5>(Final));
StoreResult(FPRClass, Op, Res0, -1);
}
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESImc(Src);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Result = _VAESImc(Src);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEnc(16, Dest, Src, LoadZeroVector(16));
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode *Result = _VAESEnc(16, Dest, Src, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -321,17 +310,19 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESENC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode *Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEncLast(16, Dest, Src, LoadZeroVector(16));
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode *Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -342,17 +333,19 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESENCLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode *Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDec(16, Dest, Src, LoadZeroVector(16));
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode *Result = _VAESDec(16, Dest, Src, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -363,17 +356,19 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESDEC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode *Result = _VAESDec(DstSize, State, Key, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDecLast(16, Dest, Src, LoadZeroVector(16));
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode *Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -384,44 +379,51 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
// TODO: Handle 256-bit VAESDECLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
OrderedNode *Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
StoreResult(FPRClass, Op, Result, -1);
}
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const uint64_t RCON = Op->Src[1].Literal();
OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
const uint64_t RCON = Op->Src[1].Data.Literal.Value;
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(16, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
return _VAESKeyGenAssist(Src, KeyGenSwizzle, LoadZeroVector(16), RCON);
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
return _VAESKeyGenAssist(Src, KeyGenSwizzle, ZeroRegister, RCON);
}
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
Ref Result = AESKeyGenAssistImpl(Op);
OrderedNode *Result = AESKeyGenAssistImpl(Op);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Selector needs to be literal here");
auto Res = _PCLMUL(16, Dest, Src, Selector & 0b1'0001);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[1].Data.Literal.Value);
auto Res = _PCLMUL(16, Dest, Src, Selector);
StoreResult(FPRClass, Op, Res, -1);
}
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Selector needs to be literal here");
const auto DstSize = GetDstSize(Op);
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
OrderedNode *Res = _PCLMUL(DstSize, Src1, Src2, Selector);
StoreResult(FPRClass, Op, Res, -1);
}
} // namespace FEXCore::IR
}
Loaded 100 of 953 files, more files were not shown because too many files have changed in this diff. Show more