mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 11:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7c79e5dea1 | ||
|
|
faa494c288 | ||
|
|
b33e0e3839 | ||
|
|
a15ed4c9da | ||
|
|
4b76eb0b3f | ||
|
|
86315027c3 | ||
|
|
a114850c2a | ||
|
|
997d1bf04b | ||
|
|
c294782a60 | ||
|
|
9781b957d0 | ||
|
|
6228226c08 | ||
|
|
31341bb7c2 | ||
|
|
3fda47e870 | ||
|
|
9c6f749ebe | ||
|
|
711ae84175 | ||
|
|
bfb06b2d55 | ||
|
|
640e911af4 | ||
|
|
76b5ca4bcc | ||
|
|
28fa88ff39 | ||
|
|
1069cabad0 | ||
|
|
fe70ec7277 | ||
|
|
4a4fa64254 | ||
|
|
6463054fa3 | ||
|
|
81a4206805 | ||
|
|
a0bf6a4255 | ||
|
|
7b88b0f271 | ||
|
|
308488c419 | ||
|
|
81c219c212 | ||
|
|
68e543c81d | ||
|
|
2372c9458b | ||
|
|
1a11343f34 | ||
|
|
cf75cdd16d | ||
|
|
84d5b3ee59 | ||
|
|
376936c808 | ||
|
|
20bd86473b | ||
|
|
932b8f38f4 | ||
|
|
a7f4e99278 | ||
|
|
7391456e48 | ||
|
|
a6d061b711 | ||
|
|
d92580bccf | ||
|
|
c8704a7f71 | ||
|
|
5cb11aed3d | ||
|
|
625d8ac177 | ||
|
|
352dcdb478 | ||
|
|
02ebb6e320 | ||
|
|
53bed6de6a | ||
|
|
00b81c8cbc | ||
|
|
c126b209f1 | ||
|
|
97386a7260 | ||
|
|
a054b998c5 | ||
|
|
f9097c0a92 | ||
|
|
55c054bd8e | ||
|
|
8799a48a0f | ||
|
|
e1efde9605 | ||
|
|
07e58f8db7 | ||
|
|
905aa935f5 | ||
|
|
7e97db8d98 | ||
|
|
340630a937 | ||
|
|
7614ac9f14 | ||
|
|
1d806ac218 | ||
|
|
2b4ec88dae | ||
|
|
028c220041 | ||
|
|
6524716404 | ||
|
|
20559853ee | ||
|
|
a9b7ad841c | ||
|
|
ae5e388e1a | ||
|
|
271700e9f6 | ||
|
|
1ba678f631 | ||
|
|
ef48700f3e | ||
|
|
a0f2cae1cb | ||
|
|
66cbb66732 | ||
|
|
7ff4cd9108 | ||
|
|
5ed593b59f | ||
|
|
f1f0c47f16 | ||
|
|
4cb2432b5c | ||
|
|
27ba66a181 | ||
|
|
65b5281d7c | ||
|
|
1a8b61b9fc | ||
|
|
f0dad86332 | ||
|
|
eedb120fd0 | ||
|
|
98841fe07a | ||
|
|
b0aeb501f4 | ||
|
|
1616b4e77c | ||
|
|
b26bf2eaf6 | ||
|
|
e91e1d5533 | ||
|
|
574fdcef32 | ||
|
|
0e93fd0f3e | ||
|
|
063954c9b3 | ||
|
|
7b3e031678 | ||
|
|
a775e474d5 | ||
|
|
0e99019586 | ||
|
|
d0ff3e64d9 | ||
|
|
d9493e5d9b | ||
|
|
eb83c9e7f2 | ||
|
|
bb24e1419c | ||
|
|
cbfa426b59 | ||
|
|
526e3e654f | ||
|
|
e03434ebde | ||
|
|
243bb45a68 | ||
|
|
bd5b817c3a | ||
|
|
b54d4931fb | ||
|
|
95589f6172 | ||
|
|
098859caf7 | ||
|
|
c632543451 | ||
|
|
801cf72f95 | ||
|
|
650cd2c46e | ||
|
|
5b48ce2228 | ||
|
|
2173c26fd8 | ||
|
|
982391ba9d | ||
|
|
a99c48b7a3 | ||
|
|
90d5bd3aec | ||
|
|
59c3f96a23 | ||
|
|
3661da1bc6 | ||
|
|
859df5e0b2 | ||
|
|
e8127b92e8 | ||
|
|
7786c23405 | ||
|
|
904646e93b | ||
|
|
c43af8e975 | ||
|
|
a787daae41 | ||
|
|
a05cc06ab4 | ||
|
|
031e756a78 | ||
|
|
2a9f1ce8cb | ||
|
|
8c53a9f051 | ||
|
|
cf26ec7898 | ||
|
|
582c3dae6e | ||
|
|
2abac03ab0 | ||
|
|
8cc684fa12 | ||
|
|
d92de1d947 | ||
|
|
202a60b77a | ||
|
|
aa8d04c341 | ||
|
|
d6425d05f3 | ||
|
|
e07c81a5e7 | ||
|
|
fa76961873 | ||
|
|
b92c206db9 | ||
|
|
efff942724 | ||
|
|
8d32113521 | ||
|
|
37f2b417e4 | ||
|
|
bd0b5eceb8 | ||
|
|
b632f7215c | ||
|
|
e8abc88702 | ||
|
|
29c6281e11 | ||
|
|
4214d9bda0 | ||
|
|
0a7b3efb41 | ||
|
|
067a5444dc | ||
|
|
c7f159972d | ||
|
|
a70d0a5dd4 | ||
|
|
cd9ffd2045 | ||
|
|
5c590b9a50 | ||
|
|
eb4bb5875e | ||
|
|
3b052e826f | ||
|
|
65ec191dc1 | ||
|
|
b1ddd8cd3b | ||
|
|
f2d001e721 | ||
|
|
7852909cc4 | ||
|
|
fad243d3f6 | ||
|
|
f8b68d8b5a | ||
|
|
e2a095372e | ||
|
|
5c29c9d464 | ||
|
|
3bed305660 | ||
|
|
f6639c3594 | ||
|
|
96087a69fa | ||
|
|
ca1ec232c9 | ||
|
|
ad0dd34412 | ||
|
|
7b1bb159fa | ||
|
|
5c7f2934de | ||
|
|
5d79d4eb50 | ||
|
|
3f66173bc7 | ||
|
|
b64a594b16 | ||
|
|
4452f0acba | ||
|
|
784cdd7b6b | ||
|
|
1a1545da0f | ||
|
|
a70ea30c02 | ||
|
|
2aa1fd7fa3 | ||
|
|
15b86e4c5a | ||
|
|
67baff8a57 | ||
|
|
fedc24be1e | ||
|
|
2a625a467b | ||
|
|
6f5e4fd34b | ||
|
|
bdda99e44f | ||
|
|
9bca052146 | ||
|
|
706065b0e2 | ||
|
|
9fd32f07cb | ||
|
|
deba6a1b76 | ||
|
|
6866c3d0ac | ||
|
|
d25ace43aa | ||
|
|
d3b2ddf641 | ||
|
|
9010b3c117 | ||
|
|
b0e001b660 | ||
|
|
eadacbd67b | ||
|
|
7de29749be | ||
|
|
1f3843ccad | ||
|
|
bc76df9901 | ||
|
|
6e92cc454d | ||
|
|
ed3af580c5 | ||
|
|
2ad170b7d7 | ||
|
|
8564290f76 | ||
|
|
1d96631af7 | ||
|
|
c513b9685d | ||
|
|
d11a36eaea | ||
|
|
f46e88ebdb | ||
|
|
20eb338644 | ||
|
|
aa26b6288e | ||
|
|
3d31291c3d | ||
|
|
624bc3fce5 | ||
|
|
d1722ab119 | ||
|
|
61758ea47d | ||
|
|
7b74ca1931 | ||
|
|
24fd28ed9e | ||
|
|
970d5d5b13 | ||
|
|
79454ed8a6 | ||
|
|
7f90ca53f7 | ||
|
|
542f454630 | ||
|
|
4ea6305940 | ||
|
|
4d8ffa2abb | ||
|
|
ade0c46845 | ||
|
|
1450c92b60 | ||
|
|
53f02ee869 | ||
|
|
6f29e75f67 | ||
|
|
c1c797bcba | ||
|
|
bbc232741b | ||
|
|
dfe0bdd7f2 | ||
|
|
e26481e3cc | ||
|
|
86b5a2f352 | ||
|
|
2bf880c43a | ||
|
|
583d4f8f94 | ||
|
|
3ca2c4377f | ||
|
|
32ec4a3c81 | ||
|
|
76983476b9 | ||
|
|
150af80f3f | ||
|
|
a8b59c16d6 | ||
|
|
949717a95f | ||
|
|
693d86dd67 | ||
|
|
ea4fce7a43 | ||
|
|
3034edb0aa | ||
|
|
c025039651 | ||
|
|
64f47d1ec2 | ||
|
|
ea31363221 | ||
|
|
70befc216f | ||
|
|
50f62663ac | ||
|
|
60755acef0 | ||
|
|
002ca360f8 | ||
|
|
4952b2e16c | ||
|
|
cccf263080 | ||
|
|
5a35e119fe | ||
|
|
9ab930cb26 | ||
|
|
824f122680 | ||
|
|
8852d94416 | ||
|
|
0c24aea27e | ||
|
|
82ba16c6ed | ||
|
|
45ea0cd782 | ||
|
|
6ce366ef35 | ||
|
|
862d63adf2 | ||
|
|
167896dc9d | ||
|
|
12fb26f9c0 | ||
|
|
d490cb1b79 | ||
|
|
94fecb9dad | ||
|
|
7dcacfe990 | ||
|
|
29b05f6b90 | ||
|
|
8d4d8fe3e5 | ||
|
|
ab8ee64352 | ||
|
|
2a9fcc6a66 | ||
|
|
20da1e4244 | ||
|
|
fd391b1b18 | ||
|
|
ba3029b1f6 | ||
|
|
6757a80365 | ||
|
|
f79991a9d8 | ||
|
|
ca6b2e43e6 | ||
|
|
8a3d08e1d8 | ||
|
|
cd2a6ce820 | ||
|
|
4e269d8b80 | ||
|
|
552e76c001 | ||
|
|
caff3cb799 | ||
|
|
0d33dacc37 | ||
|
|
ba7b69eea2 | ||
|
|
8056bee82b | ||
|
|
cc635a54f8 | ||
|
|
217d9d8c50 | ||
|
|
a047ac1699 | ||
|
|
bb0b114fc8 | ||
|
|
ccd6c15316 | ||
|
|
0a1fe1c8c2 | ||
|
|
f43fe5fd63 | ||
|
|
dce9f651fd | ||
|
|
0d71f169d0 | ||
|
|
430ac0f70a | ||
|
|
063b81da1d | ||
|
|
7629007cfa | ||
|
|
03c6abdad4 | ||
|
|
c99cbe6d0a | ||
|
|
e3ee65e491 | ||
|
|
aee00f524c | ||
|
|
a76321c6c1 | ||
|
|
f7586f4459 | ||
|
|
e33a76a2ad | ||
|
|
c37a12e806 | ||
|
|
ed59f73a65 | ||
|
|
92f31648b9 | ||
|
|
85f8ad3842 | ||
|
|
64f71d87bb | ||
|
|
ff0c7637c9 | ||
|
|
54403e2146 | ||
|
|
d8202335e0 | ||
|
|
9ec20c4bef | ||
|
|
ce7acd9b71 | ||
|
|
8a607135fd | ||
|
|
26a66790ab | ||
|
|
6d94d79409 | ||
|
|
2359a9899c | ||
|
|
aeb41e9ae2 | ||
|
|
b892da72f3 | ||
|
|
a86f2d3e2c | ||
|
|
6edba49784 |
No files matched your search
+109
@@ -0,0 +1,109 @@
|
||||
Language: Cpp
|
||||
BasedOnStyle: WebKit
|
||||
AccessModifierOffset: -2
|
||||
AlignAfterOpenBracket: Align
|
||||
AlignArrayOfStructures: None
|
||||
AlignConsecutiveAssignments: None
|
||||
AlignConsecutiveBitFields: Consecutive
|
||||
AlignConsecutiveDeclarations: None
|
||||
AlignConsecutiveMacros: None
|
||||
AlignEscapedNewlines: DontAlign
|
||||
AlignOperands: Align
|
||||
AlignTrailingComments: true
|
||||
AllowAllParametersOfDeclarationOnNextLine: false
|
||||
AllowShortCaseLabelsOnASingleLine: true
|
||||
AllowShortEnumsOnASingleLine: true
|
||||
AllowShortFunctionsOnASingleLine: Empty
|
||||
AllowShortIfStatementsOnASingleLine: WithoutElse
|
||||
AllowShortLambdasOnASingleLine: Inline
|
||||
AlwaysBreakAfterDefinitionReturnType: None
|
||||
AlwaysBreakAfterReturnType: None
|
||||
AlwaysBreakBeforeMultilineStrings: false
|
||||
AlwaysBreakTemplateDeclarations: true
|
||||
AttributeMacros:
|
||||
- JEMALLOC_NOTHROW
|
||||
- FEX_ALIGNED
|
||||
- FEX_ANNOTATE
|
||||
- FEX_DEFAULT_VISIBILITY
|
||||
- FEX_NAKED
|
||||
- FEX_PACKED
|
||||
- FEXCORE_PRESERVE_ALL_ATTR
|
||||
- GLIBC_ALIAS_FUNCTION
|
||||
BinPackArguments: true
|
||||
BinPackParameters: true
|
||||
BitFieldColonSpacing: Both
|
||||
BreakAfterAttributes: Always # clang 16 required
|
||||
BreakBeforeBraces: Attach
|
||||
BreakBeforeBinaryOperators: None
|
||||
BreakBeforeInlineASMColon: OnlyMultiline # clang 16 required
|
||||
BreakBeforeTernaryOperators: false
|
||||
BreakConstructorInitializers: BeforeComma
|
||||
BreakInheritanceList: BeforeColon
|
||||
ColumnLimit: 140
|
||||
CompactNamespaces: false
|
||||
ConstructorInitializerIndentWidth: 2
|
||||
ContinuationIndentWidth: 2
|
||||
Cpp11BracedListStyle: true
|
||||
DerivePointerAlignment: false
|
||||
EmptyLineAfterAccessModifier: Leave
|
||||
EmptyLineBeforeAccessModifier: Leave
|
||||
ExperimentalAutoDetectBinPacking: false
|
||||
FixNamespaceComments: true
|
||||
IncludeBlocks: Preserve
|
||||
IndentAccessModifiers: false
|
||||
IndentCaseBlocks: false
|
||||
IndentCaseLabels: false
|
||||
IndentExternBlock: AfterExternBlock
|
||||
IndentGotoLabels: false
|
||||
IndentPPDirectives: None
|
||||
IndentRequires: false
|
||||
IndentWidth: 2
|
||||
InsertBraces: true
|
||||
KeepEmptyLinesAtTheStartOfBlocks: true
|
||||
LambdaBodyIndentation: OuterScope
|
||||
LineEnding: LF # clang 16 required
|
||||
MaxEmptyLinesToKeep: 2
|
||||
NamespaceIndentation: Inner
|
||||
QualifierAlignment: Left
|
||||
PackConstructorInitializers: Never
|
||||
PenaltyBreakAssignment: 2
|
||||
PenaltyBreakBeforeFirstCallParameter: 2
|
||||
PenaltyBreakOpenParenthesis: 2
|
||||
PenaltyBreakString: 10
|
||||
PenaltyBreakTemplateDeclaration: 8
|
||||
PenaltyExcessCharacter: 2
|
||||
PenaltyReturnTypeOnItsOwnLine: 16
|
||||
PointerAlignment: Left
|
||||
RemoveBracesLLVM: false
|
||||
ReferenceAlignment: Left
|
||||
ReflowComments: true
|
||||
RequiresClausePosition: WithPreceding
|
||||
SeparateDefinitionBlocks: Leave
|
||||
SortIncludes: Never
|
||||
SpaceAfterCStyleCast: false
|
||||
SpaceAfterLogicalNot: false
|
||||
SpaceAfterTemplateKeyword: false
|
||||
SpaceAroundPointerQualifiers: Default
|
||||
SpaceBeforeAssignmentOperators: true
|
||||
SpaceBeforeCaseColon: false
|
||||
SpaceBeforeCpp11BracedList: true
|
||||
SpaceBeforeInheritanceColon: true
|
||||
SpaceBeforeParens: Custom
|
||||
SpaceBeforeParensOptions:
|
||||
AfterControlStatements: true
|
||||
AfterFunctionDeclarationName: false
|
||||
AfterFunctionDefinitionName: false
|
||||
AfterOverloadedOperator: false
|
||||
AfterRequiresInClause: true
|
||||
BeforeNonEmptyParentheses: false
|
||||
SpaceBeforeRangeBasedForLoopColon: true
|
||||
SpaceBeforeSquareBrackets: false
|
||||
SpaceInEmptyBlock: false
|
||||
SpaceInEmptyParentheses: false
|
||||
SpacesBeforeTrailingComments: 1
|
||||
SpacesInAngles: Leave
|
||||
SpacesInCStyleCastParentheses: false
|
||||
SpacesInConditionalStatement: false
|
||||
SpacesInParentheses: false
|
||||
Standard: c++20
|
||||
UseTab: Never
|
||||
@@ -0,0 +1,14 @@
|
||||
# This file is used to ignore files and directories from clang-format
|
||||
|
||||
# Ignore all files in the External directory
|
||||
External/*
|
||||
|
||||
# SoftFloat-3e code doesn't belong to us
|
||||
FEXCore/Source/Common/SoftFloat-3e/*
|
||||
Source/Common/cpp-optparse/*
|
||||
|
||||
# Files with human-indented tables for readability - don't mess with these
|
||||
FEXCore/Source/Interface/Core/X86Tables/X87Tables.cpp
|
||||
FEXCore/Source/Interface/Core/X86Tables/XOPTables.cpp
|
||||
FEXCore/Source/Interface/Core/X86Tables/*
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
# Since version 2.23 (released in August 2019), git-blame has a feature
|
||||
# to ignore or bypass certain commits.
|
||||
#
|
||||
# This file contains a list of commits that are not likely what you
|
||||
# are looking for in a blame, such as mass reformatting or renaming.
|
||||
# You can set this file as a default ignore file for blame by running
|
||||
# the following command.
|
||||
#
|
||||
# $ git config blame.ignoreRevsFile .git-blame-ignore-revs
|
||||
|
||||
# Whole tree reformat PR#3571
|
||||
2b4ec88daebd35fefb5bf5c73d7fc2b4155771ed
|
||||
|
||||
# Second reformat to find fixed point PR#3577
|
||||
905aa935f5ce344a48ef4d5edab3c31efa8d793e
|
||||
@@ -64,7 +64,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -237,7 +237,7 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
|
||||
@@ -71,7 +71,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -171,7 +171,7 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
|
||||
@@ -64,7 +64,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -90,7 +90,7 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
|
||||
@@ -74,7 +74,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -121,7 +121,7 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
|
||||
@@ -74,7 +74,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
# Inspired by LLVM's pr-code-format.yml at
|
||||
# https://github.com/llvm/llvm-project/blob/main/.github/workflows/pr-code-format.yml
|
||||
|
||||
name: "Check code formatting"
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
jobs:
|
||||
code_formatter:
|
||||
runs-on: [self-hosted, X64]
|
||||
if: github.repository == 'FEX-Emu/FEX'
|
||||
|
||||
steps:
|
||||
- name: Fetch FEX sources
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ github.event.pull_request.head.sha }}
|
||||
|
||||
- name: Checkout through merge base
|
||||
uses: rmacklin/fetch-through-merge-base@v0
|
||||
with:
|
||||
base_ref: ${{ github.event.pull_request.base.ref }}
|
||||
head_ref: ${{ github.event.pull_request.head.sha }}
|
||||
deepen_length: 500
|
||||
|
||||
- name: Get changed files
|
||||
id: changed-files
|
||||
uses: tj-actions/changed-files@v39
|
||||
with:
|
||||
separator: ","
|
||||
skip_initial_fetch: true
|
||||
|
||||
- name: "Listed files"
|
||||
env:
|
||||
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
|
||||
run: |
|
||||
echo "Formatting files:"
|
||||
echo "$CHANGED_FILES"
|
||||
|
||||
- name: Check for correct clang-format version
|
||||
run: clang-format --version | grep -qF '16.0.6'
|
||||
|
||||
- name: Check git-clang-format-16 exists
|
||||
run: which git-clang-format-16
|
||||
|
||||
- name: Setup Python env
|
||||
uses: actions/setup-python@v4
|
||||
with:
|
||||
python-version: '3.11'
|
||||
cache: 'pip'
|
||||
cache-dependency-path: './External/code-format-helper/requirements_formatting.txt'
|
||||
|
||||
- name: Install python dependencies
|
||||
run: pip install -r ./External/code-format-helper/requirements_formatting.txt
|
||||
|
||||
- name: Run code formatter
|
||||
env:
|
||||
CLANG_FORMAT_PATH: 'git-clang-format-16'
|
||||
GITHUB_PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
START_REV: ${{ github.event.pull_request.base.sha }}
|
||||
END_REV: ${{ github.event.pull_request.head.sha }}
|
||||
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
|
||||
# TODO(pmatos): Once we adopt v18, we should be able
|
||||
# to take advantage of the new --diff_from_common_commit option
|
||||
# explicitly in code-format-helper.py and not have to diff starting at
|
||||
# the merge base.
|
||||
run: |
|
||||
python ./External/code-format-helper/code-format-helper.py \
|
||||
--repo "FEX-emu/FEX" \
|
||||
--issue-number $GITHUB_PR_NUMBER \
|
||||
--start-rev $(git merge-base $START_REV $END_REV) \
|
||||
--end-rev $END_REV \
|
||||
--changed-files "$CHANGED_FILES"
|
||||
@@ -65,7 +65,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -106,7 +106,7 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
run: truncate --size="<20M" ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
|
||||
+3
-65
@@ -9,7 +9,6 @@ option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
@@ -26,7 +25,6 @@ option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
|
||||
@@ -125,6 +123,9 @@ endif()
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^arm64ec")
|
||||
set(_M_ARM_64EC 1)
|
||||
add_definitions(-D_M_ARM_64EC=1)
|
||||
|
||||
# Required as FEX is not allowed to lock the CRT heap lock during compilation or callbacks
|
||||
set(ENABLE_JEMALLOC TRUE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_CCACHE)
|
||||
@@ -163,18 +164,6 @@ if (NOT ENABLE_OFFLINE_TELEMETRY)
|
||||
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
|
||||
endif()
|
||||
|
||||
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
add_definitions(-DTERMUX_BUILD=1)
|
||||
set(TERMUX_BUILD 1)
|
||||
|
||||
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
|
||||
# Termux builds can't rely on X11 packages
|
||||
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
|
||||
set(BUILD_FEXCONFIG FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
add_definitions(-DENABLE_ASAN=1)
|
||||
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
|
||||
@@ -357,57 +346,6 @@ if (ENABLE_IWYU)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_CLANG_FORMAT)
|
||||
find_program(CLANG_TIDY_EXE "clang-tidy")
|
||||
if (NOT CLANG_TIDY_EXE)
|
||||
message(FATAL_ERROR "Couldn't find clang-tidy")
|
||||
endif()
|
||||
|
||||
set(CLANG_TIDY_FLAGS
|
||||
"-checks=*"
|
||||
"-fuchsia*"
|
||||
"-bugprone-macro-parentheses"
|
||||
"-clang-analyzer-core.*"
|
||||
"-cppcoreguidelines-pro-type-*"
|
||||
"-cppcoreguidelines-pro-bounds-array-to-pointer-decay"
|
||||
"-cppcoreguidelines-pro-bounds-pointer-arithmetic"
|
||||
"-cppcoreguidelines-avoid-c-arrays"
|
||||
"-cppcoreguidelines-avoid-magic-numbers"
|
||||
"-cppcoreguidelines-pro-bounds-constant-array-index"
|
||||
"-cppcoreguidelines-no-malloc"
|
||||
"-cppcoreguidelines-special-member-functions"
|
||||
"-cppcoreguidelines-owning-memory"
|
||||
"-cppcoreguidelines-macro-usage"
|
||||
"-cppcoreguidelines-avoid-goto"
|
||||
"-google-readability-function-size"
|
||||
"-google-readability-namespace-comments"
|
||||
"-google-readability-braces-around-statements"
|
||||
"-google-build-using-namespace"
|
||||
"-hicpp-*"
|
||||
"-llvm-namespace-comment"
|
||||
"-llvm-include-order" # Messes up with case sensitivity
|
||||
"-llvmlibc-*"
|
||||
"-misc-unused-parameters"
|
||||
"-modernize-loop-convert"
|
||||
"-modernize-use-auto"
|
||||
"-modernize-avoid-c-arrays"
|
||||
"-modernize-use-nodiscard"
|
||||
"readability-*"
|
||||
"-readability-function-size"
|
||||
"-readability-implicit-bool-conversion"
|
||||
"-readability-braces-around-statements"
|
||||
"-readability-else-after-return"
|
||||
"-readability-magic-numbers"
|
||||
"-readability-named-parameter"
|
||||
"-readability-uppercase-literal-suffix"
|
||||
"-cert-err34-c"
|
||||
"-cert-err58-cpp"
|
||||
"-bugprone-exception-escape"
|
||||
)
|
||||
string(REPLACE ";" "," CLANG_TIDY_FLAGS "${CLANG_TIDY_FLAGS}")
|
||||
set(CMAKE_CXX_CLANG_TIDY ${CLANG_TIDY_EXE} "${CLANG_TIDY_FLAGS}")
|
||||
endif()
|
||||
|
||||
add_compile_options(-Wall)
|
||||
|
||||
configure_file(
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"Comment": "Bypasses libGL's glX and instead sends GLX requests directly via xcb",
|
||||
"ThunksDB": {
|
||||
"GL": 0
|
||||
}
|
||||
}
|
||||
@@ -2,9 +2,6 @@
|
||||
"DB": {
|
||||
"GL": {
|
||||
"Library" : "libGL-guest.so",
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libGL.so",
|
||||
"@PREFIX_LIB@/libGL.so.1",
|
||||
@@ -33,16 +30,10 @@
|
||||
},
|
||||
"Vulkan": {
|
||||
"Library": "libvulkan-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libvulkan.so",
|
||||
"@PREFIX_LIB@/libvulkan.so.1",
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
],
|
||||
"Comment": [
|
||||
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
|
||||
Vendored
+1
-1
Submodule External/Catch2 updated: d4b0b34561...8ac8190e49.
+394
@@ -0,0 +1,394 @@
|
||||
#!/usr/bin/env python3
|
||||
#
|
||||
# ====- code-format-helper, runs code formatters from the ci or in a hook --*- python -*--==#
|
||||
#
|
||||
# Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
# See https://llvm.org/LICENSE.txt for license information.
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
#
|
||||
# ==--------------------------------------------------------------------------------------==#
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import List, Optional
|
||||
|
||||
"""
|
||||
This script is run by GitHub actions to ensure that the code in PR's conform to
|
||||
the coding style of LLVM. It can also be installed as a pre-commit git hook to
|
||||
check the coding style before submitting it. The canonical source of this script
|
||||
is in the LLVM source tree under llvm/utils/git.
|
||||
|
||||
For C/C++ code it uses clang-format and for Python code it uses darker (which
|
||||
in turn invokes black).
|
||||
|
||||
You can learn more about the LLVM coding style on llvm.org:
|
||||
https://llvm.org/docs/CodingStandards.html
|
||||
|
||||
You can install this script as a git hook by symlinking it to the .git/hooks
|
||||
directory:
|
||||
|
||||
ln -s $(pwd)/llvm/utils/git/code-format-helper.py .git/hooks/pre-commit
|
||||
|
||||
You can control the exact path to clang-format or darker with the following
|
||||
environment variables: $CLANG_FORMAT_PATH and $DARKER_FORMAT_PATH.
|
||||
"""
|
||||
|
||||
|
||||
class FormatArgs:
|
||||
start_rev: str = None
|
||||
end_rev: str = None
|
||||
repo: str = None
|
||||
changed_files: List[str] = []
|
||||
token: str = None
|
||||
verbose: bool = True
|
||||
issue_number: int = 0
|
||||
write_comment_to_file: str = None
|
||||
|
||||
def __init__(self, args: argparse.Namespace = None) -> None:
|
||||
if not args is None:
|
||||
self.start_rev = args.start_rev
|
||||
self.end_rev = args.end_rev
|
||||
self.repo = args.repo
|
||||
self.token = args.token
|
||||
self.changed_files = args.changed_files
|
||||
self.issue_number = args.issue_number
|
||||
self.write_comment_to_file = args.write_comment_to_file
|
||||
|
||||
|
||||
class FormatHelper:
|
||||
COMMENT_TAG = "<!--CODE FORMAT COMMENT: {fmt}-->"
|
||||
name: str
|
||||
friendly_name: str
|
||||
comment: dict = None
|
||||
|
||||
@property
|
||||
def comment_tag(self) -> str:
|
||||
return self.COMMENT_TAG.replace("fmt", self.name)
|
||||
|
||||
@property
|
||||
def instructions(self) -> str:
|
||||
raise NotImplementedError()
|
||||
|
||||
def has_tool(self) -> bool:
|
||||
raise NotImplementedError()
|
||||
|
||||
def format_run(self, changed_files: List[str], args: FormatArgs) -> Optional[str]:
|
||||
raise NotImplementedError()
|
||||
|
||||
def pr_comment_text_for_diff(self, diff: str) -> str:
|
||||
return f"""
|
||||
:warning: {self.friendly_name}, {self.name} found issues in your code. :warning:
|
||||
|
||||
<details>
|
||||
<summary>
|
||||
You can test this locally with the following command:
|
||||
</summary>
|
||||
|
||||
``````````bash
|
||||
{self.instructions}
|
||||
``````````
|
||||
|
||||
</details>
|
||||
|
||||
<details>
|
||||
<summary>
|
||||
View the diff from {self.name} here.
|
||||
</summary>
|
||||
|
||||
``````````diff
|
||||
{diff}
|
||||
``````````
|
||||
|
||||
</details>
|
||||
"""
|
||||
|
||||
# TODO: any type should be replaced with the correct github type, but it requires refactoring to
|
||||
# not require the github module to be installed everywhere.
|
||||
def find_comment(self, pr: any) -> any:
|
||||
for comment in pr.as_issue().get_comments():
|
||||
if self.comment_tag in comment.body:
|
||||
return comment
|
||||
return None
|
||||
|
||||
def update_pr(self, comment_text: str, args: FormatArgs, create_new: bool) -> None:
|
||||
import github
|
||||
from github import IssueComment, PullRequest
|
||||
|
||||
repo = github.Github(args.token).get_repo(args.repo)
|
||||
pr = repo.get_issue(args.issue_number).as_pull_request()
|
||||
|
||||
comment_text = self.comment_tag + "\n\n" + comment_text
|
||||
|
||||
existing_comment = self.find_comment(pr)
|
||||
|
||||
if args.write_comment_to_file:
|
||||
if create_new or existing_comment:
|
||||
self.comment = {"body": comment_text}
|
||||
if existing_comment:
|
||||
self.comment["id"] = existing_comment.id
|
||||
return
|
||||
|
||||
if existing_comment:
|
||||
existing_comment.edit(comment_text)
|
||||
elif create_new:
|
||||
pr.as_issue().create_comment(comment_text)
|
||||
|
||||
def run(self, changed_files: List[str], args: FormatArgs) -> bool:
|
||||
changed_files = [arg for arg in changed_files if "third-party" not in arg]
|
||||
diff = self.format_run(changed_files, args)
|
||||
should_update_gh = args.token is not None and args.repo is not None
|
||||
|
||||
if diff is None:
|
||||
if should_update_gh:
|
||||
comment_text = (
|
||||
":white_check_mark: With the latest revision "
|
||||
f"this PR passed the {self.friendly_name}."
|
||||
)
|
||||
self.update_pr(comment_text, args, create_new=False)
|
||||
return True
|
||||
elif len(diff) > 0:
|
||||
if should_update_gh:
|
||||
comment_text = self.pr_comment_text_for_diff(diff)
|
||||
self.update_pr(comment_text, args, create_new=True)
|
||||
else:
|
||||
print(
|
||||
f"Warning: {self.friendly_name}, {self.name} detected "
|
||||
"some issues with your code formatting..."
|
||||
)
|
||||
return False
|
||||
else:
|
||||
# The formatter failed but didn't output a diff (e.g. some sort of
|
||||
# infrastructure failure).
|
||||
comment_text = (
|
||||
f":warning: The {self.friendly_name} failed without printing "
|
||||
"a diff. Check the logs for stderr output. :warning:"
|
||||
)
|
||||
self.update_pr(comment_text, args, create_new=False)
|
||||
return False
|
||||
|
||||
|
||||
class ClangFormatHelper(FormatHelper):
|
||||
name = "clang-format"
|
||||
friendly_name = "C/C++ code formatter"
|
||||
|
||||
@property
|
||||
def instructions(self) -> str:
|
||||
return " ".join(self.cf_cmd)
|
||||
|
||||
def should_include_extensionless_file(self, path: str) -> bool:
|
||||
return path.startswith("libcxx/include")
|
||||
|
||||
def filter_changed_files(self, changed_files: List[str]) -> List[str]:
|
||||
filtered_files = []
|
||||
for path in changed_files:
|
||||
_, ext = os.path.splitext(path)
|
||||
if ext in (".cpp", ".c", ".h", ".hpp", ".hxx", ".cxx", ".inc", ".cppm"):
|
||||
filtered_files.append(path)
|
||||
elif ext == "" and self.should_include_extensionless_file(path):
|
||||
filtered_files.append(path)
|
||||
return filtered_files
|
||||
|
||||
@property
|
||||
def clang_fmt_path(self) -> str:
|
||||
if "CLANG_FORMAT_PATH" in os.environ:
|
||||
return os.environ["CLANG_FORMAT_PATH"]
|
||||
return "git-clang-format"
|
||||
|
||||
def has_tool(self) -> bool:
|
||||
cmd = [self.clang_fmt_path, "-h"]
|
||||
proc = None
|
||||
try:
|
||||
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||
except:
|
||||
return False
|
||||
return proc.returncode == 0
|
||||
|
||||
def format_run(self, changed_files: List[str], args: FormatArgs) -> Optional[str]:
|
||||
cpp_files = self.filter_changed_files(changed_files)
|
||||
if not cpp_files:
|
||||
return None
|
||||
|
||||
cf_cmd = [self.clang_fmt_path, "--diff"]
|
||||
|
||||
if args.start_rev and args.end_rev:
|
||||
cf_cmd.append(args.start_rev)
|
||||
cf_cmd.append(args.end_rev)
|
||||
|
||||
cf_cmd.append("--")
|
||||
cf_cmd += cpp_files
|
||||
|
||||
if args.verbose:
|
||||
print(f"Running: {' '.join(cf_cmd)}")
|
||||
self.cf_cmd = cf_cmd
|
||||
proc = subprocess.run(cf_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||
sys.stdout.write(proc.stderr.decode("utf-8"))
|
||||
|
||||
if proc.returncode != 0:
|
||||
# formatting needed, or the command otherwise failed
|
||||
if args.verbose:
|
||||
print(f"error: {self.name} exited with code {proc.returncode}")
|
||||
# Print the diff in the log so that it is viewable there
|
||||
print(proc.stdout.decode("utf-8"))
|
||||
return proc.stdout.decode("utf-8")
|
||||
else:
|
||||
return None
|
||||
|
||||
|
||||
class DarkerFormatHelper(FormatHelper):
|
||||
name = "darker"
|
||||
friendly_name = "Python code formatter"
|
||||
|
||||
@property
|
||||
def instructions(self) -> str:
|
||||
return " ".join(self.darker_cmd)
|
||||
|
||||
def filter_changed_files(self, changed_files: List[str]) -> List[str]:
|
||||
filtered_files = []
|
||||
for path in changed_files:
|
||||
name, ext = os.path.splitext(path)
|
||||
if ext == ".py":
|
||||
filtered_files.append(path)
|
||||
|
||||
return filtered_files
|
||||
|
||||
@property
|
||||
def darker_fmt_path(self) -> str:
|
||||
if "DARKER_FORMAT_PATH" in os.environ:
|
||||
return os.environ["DARKER_FORMAT_PATH"]
|
||||
return "darker"
|
||||
|
||||
def has_tool(self) -> bool:
|
||||
cmd = [self.darker_fmt_path, "--version"]
|
||||
proc = None
|
||||
try:
|
||||
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||
except:
|
||||
return False
|
||||
return proc.returncode == 0
|
||||
|
||||
def format_run(self, changed_files: List[str], args: FormatArgs) -> Optional[str]:
|
||||
py_files = self.filter_changed_files(changed_files)
|
||||
if not py_files:
|
||||
return None
|
||||
darker_cmd = [
|
||||
self.darker_fmt_path,
|
||||
"--check",
|
||||
"--diff",
|
||||
]
|
||||
if args.start_rev and args.end_rev:
|
||||
darker_cmd += ["-r", f"{args.start_rev}...{args.end_rev}"]
|
||||
darker_cmd += py_files
|
||||
if args.verbose:
|
||||
print(f"Running: {' '.join(darker_cmd)}")
|
||||
self.darker_cmd = darker_cmd
|
||||
proc = subprocess.run(
|
||||
darker_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE
|
||||
)
|
||||
if args.verbose:
|
||||
sys.stdout.write(proc.stderr.decode("utf-8"))
|
||||
|
||||
if proc.returncode != 0:
|
||||
# formatting needed, or the command otherwise failed
|
||||
if args.verbose:
|
||||
print(f"error: {self.name} exited with code {proc.returncode}")
|
||||
# Print the diff in the log so that it is viewable there
|
||||
print(proc.stdout.decode("utf-8"))
|
||||
return proc.stdout.decode("utf-8")
|
||||
else:
|
||||
sys.stdout.write(proc.stdout.decode("utf-8"))
|
||||
return None
|
||||
|
||||
|
||||
ALL_FORMATTERS = (DarkerFormatHelper(), ClangFormatHelper())
|
||||
|
||||
|
||||
def hook_main():
|
||||
# fill out args
|
||||
args = FormatArgs()
|
||||
args.verbose = False
|
||||
|
||||
# find the changed files
|
||||
cmd = ["git", "diff", "--cached", "--name-only", "--diff-filter=d"]
|
||||
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||
output = proc.stdout.decode("utf-8")
|
||||
for line in output.splitlines():
|
||||
args.changed_files.append(line)
|
||||
|
||||
failed_fmts = []
|
||||
for fmt in ALL_FORMATTERS:
|
||||
if fmt.has_tool():
|
||||
if not fmt.run(args.changed_files, args):
|
||||
failed_fmts.append(fmt.name)
|
||||
if fmt.comment:
|
||||
comments.append(fmt.comment)
|
||||
else:
|
||||
print(f"Couldn't find {fmt.name}, can't check " + fmt.friendly_name.lower())
|
||||
|
||||
if len(failed_fmts) > 0:
|
||||
sys.exit(1)
|
||||
|
||||
sys.exit(0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
script_path = os.path.abspath(__file__)
|
||||
if ".git/hooks" in script_path:
|
||||
hook_main()
|
||||
sys.exit(0)
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument(
|
||||
"--token", type=str, required=False, help="GitHub authentication token"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--repo",
|
||||
type=str,
|
||||
default=os.getenv("GITHUB_REPOSITORY", "llvm/llvm-project"),
|
||||
help="The GitHub repository that we are working with in the form of <owner>/<repo> (e.g. llvm/llvm-project)",
|
||||
)
|
||||
parser.add_argument("--issue-number", type=int, required=True)
|
||||
parser.add_argument(
|
||||
"--start-rev",
|
||||
type=str,
|
||||
required=True,
|
||||
help="Compute changes from this revision.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--end-rev", type=str, required=True, help="Compute changes to this revision"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--changed-files",
|
||||
type=str,
|
||||
help="Comma separated list of files that has been changed",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--write-comment-to-file",
|
||||
type=str,
|
||||
help="Don't post comments on the PR, instead write the comments and metadata a file",
|
||||
)
|
||||
|
||||
args = FormatArgs(parser.parse_args())
|
||||
|
||||
changed_files = []
|
||||
if args.changed_files:
|
||||
changed_files = args.changed_files.split(",")
|
||||
|
||||
failed_formatters = []
|
||||
comments = []
|
||||
for fmt in ALL_FORMATTERS:
|
||||
if not fmt.run(changed_files, args):
|
||||
failed_formatters.append(fmt.name)
|
||||
if fmt.comment:
|
||||
comments.append(fmt.comment)
|
||||
|
||||
if len(comments):
|
||||
with open(args.write_comment_to_file, "w") as f:
|
||||
import json
|
||||
|
||||
json.dump(comments, f)
|
||||
|
||||
if len(failed_formatters) > 0:
|
||||
print(f"error: some formatters failed: {' '.join(failed_formatters)}")
|
||||
sys.exit(1)
|
||||
@@ -0,0 +1,52 @@
|
||||
#
|
||||
# This file is autogenerated by pip-compile with Python 3.11
|
||||
# by the following command:
|
||||
#
|
||||
# pip-compile --output-file=llvm/utils/git/requirements_formatting.txt llvm/utils/git/requirements_formatting.txt.in
|
||||
#
|
||||
black==23.9.1
|
||||
# via
|
||||
# -r llvm/utils/git/requirements_formatting.txt.in
|
||||
# darker
|
||||
certifi==2023.7.22
|
||||
# via requests
|
||||
cffi==1.15.1
|
||||
# via
|
||||
# cryptography
|
||||
# pynacl
|
||||
charset-normalizer==3.2.0
|
||||
# via requests
|
||||
click==8.1.7
|
||||
# via black
|
||||
cryptography==41.0.3
|
||||
# via pyjwt
|
||||
darker==1.7.2
|
||||
# via -r llvm/utils/git/requirements_formatting.txt.in
|
||||
deprecated==1.2.14
|
||||
# via pygithub
|
||||
idna==3.4
|
||||
# via requests
|
||||
mypy-extensions==1.0.0
|
||||
# via black
|
||||
packaging==23.1
|
||||
# via black
|
||||
pathspec==0.11.2
|
||||
# via black
|
||||
platformdirs==3.10.0
|
||||
# via black
|
||||
pycparser==2.21
|
||||
# via cffi
|
||||
pygithub==1.59.1
|
||||
# via -r llvm/utils/git/requirements_formatting.txt.in
|
||||
pyjwt[crypto]==2.8.0
|
||||
# via pygithub
|
||||
pynacl==1.5.0
|
||||
# via pygithub
|
||||
requests==2.31.0
|
||||
# via pygithub
|
||||
toml==0.10.2
|
||||
# via darker
|
||||
urllib3==2.0.4
|
||||
# via requests
|
||||
wrapt==1.15.0
|
||||
# via deprecated
|
||||
Vendored
+1
-1
Submodule External/drm-headers updated: 07099adb70...34a20394f7.
Vendored
+1
-1
Submodule External/jemalloc updated: 16f8061955...5695452413.
@@ -13,8 +13,6 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(_M_ARM_64 1)
|
||||
endif()
|
||||
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
|
||||
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
|
||||
cmake_policy(SET CMP0083 NEW) # Follow new PIE policy
|
||||
include(CheckPIESupported)
|
||||
|
||||
@@ -200,6 +200,9 @@ if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
else()
|
||||
list (APPEND LIBS synchronization)
|
||||
if (_M_ARM_64EC)
|
||||
list (APPEND LIBS kernelbase)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_JEMALLOC)
|
||||
|
||||
@@ -18,7 +18,7 @@ struct BitSet final {
|
||||
constexpr static size_t MinimumSize = sizeof(ElementType);
|
||||
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
|
||||
|
||||
ElementType *Memory;
|
||||
ElementType* Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
@@ -62,11 +62,10 @@ struct BitSetView final {
|
||||
constexpr static size_t MinimumSize = sizeof(ElementType);
|
||||
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
|
||||
|
||||
ElementType *Memory;
|
||||
ElementType* Memory;
|
||||
|
||||
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0,
|
||||
"Bitset view offset needs to be aligned to size of backing element");
|
||||
void GetView(BitSet<T>& Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
|
||||
@@ -7,133 +7,142 @@
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
JITSymbols::JITSymbols() {
|
||||
}
|
||||
JITSymbols::JITSymbols() {}
|
||||
|
||||
JITSymbols::~JITSymbols() {
|
||||
if (fd != -1) {
|
||||
close(fd);
|
||||
}
|
||||
JITSymbols::~JITSymbols() {
|
||||
if (fd != -1) {
|
||||
close(fd);
|
||||
}
|
||||
}
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
void JITSymbols::InitFile() {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
#ifdef __ANDROID__
|
||||
// Android simpleperf looks in /data/local/tmp instead of /tmp
|
||||
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
|
||||
// Android simpleperf looks in /data/local/tmp instead of /tmp
|
||||
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
|
||||
#else
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
#endif
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (fd == -1) return;
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
}
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
void JITSymbols::RegisterJITSpace(const void* HostAddr, uint32_t CodeSize) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
|
||||
if (fd == -1) return;
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
}
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto Buffer = fextl::fmt::format("{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
auto Result = write(fd, Buffer.c_str(), Buffer.size());
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
// Buffered JIT symbols.
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Buffered JIT symbols.
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (fd == -1) return;
|
||||
// Calculate remaining sizes.
|
||||
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
|
||||
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
|
||||
|
||||
// Calculate remaining sizes.
|
||||
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
|
||||
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
|
||||
// Couldn't fit, need to force a write.
|
||||
WriteBuffer(Buffer, true);
|
||||
// Rerun
|
||||
Register(Buffer, HostAddr, GuestAddr, CodeSize);
|
||||
return;
|
||||
}
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
|
||||
// Couldn't fit, need to force a write.
|
||||
WriteBuffer(Buffer, true);
|
||||
// Rerun
|
||||
Register(Buffer, HostAddr, GuestAddr, CodeSize);
|
||||
Buffer->Offset += FMTResult.size;
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Calculate remaining sizes.
|
||||
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
|
||||
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto FMTResult =
|
||||
fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
|
||||
// Couldn't fit, need to force a write.
|
||||
WriteBuffer(Buffer, true);
|
||||
// Rerun
|
||||
Register(Buffer, HostAddr, CodeSize, Name, Offset);
|
||||
return;
|
||||
}
|
||||
|
||||
Buffer->Offset += FMTResult.size;
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (fd == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Calculate remaining sizes.
|
||||
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
|
||||
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
|
||||
// Couldn't fit, need to force a write.
|
||||
WriteBuffer(Buffer, true);
|
||||
// Rerun
|
||||
RegisterNamedRegion(Buffer, HostAddr, CodeSize, Name);
|
||||
return;
|
||||
}
|
||||
|
||||
Buffer->Offset += FMTResult.size;
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite) {
|
||||
auto Now = std::chrono::steady_clock::now();
|
||||
if (!ForceWrite) {
|
||||
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) && Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
|
||||
// Still buffering, no need to write.
|
||||
return;
|
||||
}
|
||||
|
||||
Buffer->Offset += FMTResult.size;
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset) {
|
||||
if (fd == -1) return;
|
||||
|
||||
// Calculate remaining sizes.
|
||||
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
|
||||
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}+0x{:x} ({})\n", HostAddr, CodeSize, Name, Offset, HostAddr);
|
||||
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
|
||||
// Couldn't fit, need to force a write.
|
||||
WriteBuffer(Buffer, true);
|
||||
// Rerun
|
||||
Register(Buffer, HostAddr, CodeSize, Name, Offset);
|
||||
return;
|
||||
}
|
||||
|
||||
Buffer->Offset += FMTResult.size;
|
||||
WriteBuffer(Buffer);
|
||||
Buffer->LastWrite = Now;
|
||||
auto Result = write(fd, Buffer->Buffer, Buffer->Offset);
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
if (fd == -1) return;
|
||||
|
||||
// Calculate remaining sizes.
|
||||
const auto RemainingSize = Buffer->BUFFER_SIZE - Buffer->Offset;
|
||||
const auto CurrentBufferOffset = &Buffer->Buffer[Buffer->Offset];
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
const auto FMTResult = fmt::format_to_n(CurrentBufferOffset, RemainingSize, "{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
if (FMTResult.out >= &Buffer->Buffer[Buffer->BUFFER_SIZE]) {
|
||||
// Couldn't fit, need to force a write.
|
||||
WriteBuffer(Buffer, true);
|
||||
// Rerun
|
||||
RegisterNamedRegion(Buffer, HostAddr, CodeSize, Name);
|
||||
return;
|
||||
}
|
||||
|
||||
Buffer->Offset += FMTResult.size;
|
||||
WriteBuffer(Buffer);
|
||||
}
|
||||
|
||||
void JITSymbols::WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite) {
|
||||
auto Now = std::chrono::steady_clock::now();
|
||||
if (!ForceWrite) {
|
||||
if (((Buffer->LastWrite - Now) < Buffer->MAXIMUM_THRESHOLD) &&
|
||||
Buffer->Offset < Buffer->NEEDS_WRITE_DISTANCE) {
|
||||
// Still buffering, no need to write.
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
Buffer->LastWrite = Now;
|
||||
auto Result = write(fd, Buffer->Buffer, Buffer->Offset);
|
||||
if (Result == -1 && errno == EBADF) {
|
||||
fd = -1;
|
||||
}
|
||||
|
||||
Buffer->Offset = 0;
|
||||
}
|
||||
Buffer->Offset = 0;
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -17,20 +17,20 @@ public:
|
||||
~JITSymbols();
|
||||
|
||||
void InitFile();
|
||||
void RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterJITSpace(const void *HostAddr, uint32_t CodeSize);
|
||||
void RegisterNamedRegion(const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterJITSpace(const void* HostAddr, uint32_t CodeSize);
|
||||
|
||||
// Allocate JIT buffer.
|
||||
static fextl::unique_ptr<Core::JITSymbolBuffer> AllocateBuffer() {
|
||||
return fextl::make_unique<Core::JITSymbolBuffer>();
|
||||
}
|
||||
|
||||
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(Core::JITSymbolBuffer *Buffer, const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
void RegisterNamedRegion(Core::JITSymbolBuffer* Buffer, const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
|
||||
private:
|
||||
int fd{-1};
|
||||
void WriteBuffer(Core::JITSymbolBuffer *Buffer, bool ForceWrite = false);
|
||||
int fd {-1};
|
||||
void WriteBuffer(Core::JITSymbolBuffer* Buffer, bool ForceWrite = false);
|
||||
};
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -45,13 +45,13 @@ struct FEX_PACKED X80SoftFloat {
|
||||
uint16_t Exponent : 15;
|
||||
uint16_t Sign : 1;
|
||||
|
||||
X80SoftFloat() { memset(this, 0, sizeof(*this)); }
|
||||
X80SoftFloat() {
|
||||
memset(this, 0, sizeof(*this));
|
||||
}
|
||||
X80SoftFloat(uint16_t _Sign, uint16_t _Exponent, uint64_t _Significand)
|
||||
: Significand {_Significand}
|
||||
, Exponent {_Exponent}
|
||||
, Sign {_Sign}
|
||||
{
|
||||
}
|
||||
, Sign {_Sign} {}
|
||||
|
||||
fextl::string str() const {
|
||||
fextl::ostringstream string;
|
||||
@@ -63,21 +63,19 @@ struct FEX_PACKED X80SoftFloat {
|
||||
}
|
||||
|
||||
// Ops
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FADD(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
faddp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs), [rhs] "m"(rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -85,21 +83,19 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSUB(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fsubp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs), [rhs] "m"(rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -107,21 +103,19 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FMUL(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fmulp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs), [rhs] "m"(rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -129,21 +123,19 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FDIV(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fdivp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs), [rhs] "m"(rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -151,11 +143,10 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
@@ -163,10 +154,9 @@ struct FEX_PACKED X80SoftFloat {
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs), [rhs] "m"(rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -174,11 +164,10 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FREM1(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
@@ -186,10 +175,9 @@ struct FEX_PACKED X80SoftFloat {
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs), [rhs] "m"(rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -197,30 +185,27 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(const X80SoftFloat& lhs) {
|
||||
return extF80_roundToInt(lhs, softfloat_roundingMode, false);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs, uint_fast8_t RoundMode) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FRNDINT(const X80SoftFloat& lhs, uint_fast8_t RoundMode) {
|
||||
return extF80_roundToInt(lhs, RoundMode, false);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FXTRACT_SIG(const X80SoftFloat& lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -231,20 +216,19 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FXTRACT_EXP(const X80SoftFloat& lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -253,19 +237,17 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static void FCMP(const X80SoftFloat& lhs, const X80SoftFloat& rhs, bool* eq, bool* lt, bool* nan) {
|
||||
*eq = extF80_eq(lhs, rhs);
|
||||
*lt = extF80_lt(lhs, rhs);
|
||||
*nan = IsNan(lhs) || IsNan(rhs);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSCALE(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
@@ -273,10 +255,9 @@ struct FEX_PACKED X80SoftFloat {
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs), [rhs] "m"(rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -289,20 +270,19 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat F2XM1(const X80SoftFloat& lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
f2xm1; # st0 = 2^st(0) - 1
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -313,22 +293,20 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FYL2X(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st(1)
|
||||
fldt %[lhs]; # st(0)
|
||||
fyl2x; # st(1) * log2l(st(0))
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs), [rhs] "m"(rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -339,22 +317,20 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FATAN(const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[lhs];
|
||||
fldt %[rhs];
|
||||
fpatan;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs), [rhs] "m"(rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -365,21 +341,20 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FTAN(const X80SoftFloat& lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fptan;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -389,20 +364,19 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSIN(const X80SoftFloat& lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsin;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -412,20 +386,19 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FCOS(const X80SoftFloat& lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fcos;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -435,19 +408,18 @@ struct FEX_PACKED X80SoftFloat {
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSQRT(const X80SoftFloat& lhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
asm(R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsqrt;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
: [result] "=m"(Result)
|
||||
: [lhs] "m"(lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
@@ -471,7 +443,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
const float128_t Result = extF80_to_f128(*this);
|
||||
return FEXCore::BitCast<BIGFLOAT>(Result);
|
||||
#else
|
||||
BIGFLOAT result{};
|
||||
BIGFLOAT result {};
|
||||
memcpy(&result, this, sizeof(result));
|
||||
return result;
|
||||
#endif
|
||||
@@ -570,19 +542,17 @@ struct FEX_PACKED X80SoftFloat {
|
||||
}
|
||||
|
||||
operator extFloat80_t() const {
|
||||
extFloat80_t Result{};
|
||||
extFloat80_t Result {};
|
||||
Result.signif = Significand;
|
||||
Result.signExp = Exponent | (Sign << 15);
|
||||
return Result;
|
||||
}
|
||||
|
||||
static bool IsNan(X80SoftFloat const &lhs) {
|
||||
return (lhs.Exponent == 0x7FFF) &&
|
||||
(lhs.Significand & IntegerBit) &&
|
||||
(lhs.Significand & Bottom62Significand);
|
||||
static bool IsNan(const X80SoftFloat& lhs) {
|
||||
return (lhs.Exponent == 0x7FFF) && (lhs.Significand & IntegerBit) && (lhs.Significand & Bottom62Significand);
|
||||
}
|
||||
|
||||
static bool SignBit(X80SoftFloat const &lhs) {
|
||||
static bool SignBit(const X80SoftFloat& lhs) {
|
||||
return lhs.Sign;
|
||||
}
|
||||
|
||||
|
||||
@@ -7,44 +7,51 @@
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::StrConv {
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, bool *Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint8_t *Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint16_t *Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint32_t *Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, int32_t *Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, uint64_t *Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
template <typename T,
|
||||
typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, T *Result) {
|
||||
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]] static bool Conv(std::string_view Value, fextl::string *Result) {
|
||||
*Result = Value;
|
||||
return true;
|
||||
}
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, bool* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int32_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, fextl::string* Result) {
|
||||
*Result = Value;
|
||||
return true;
|
||||
}
|
||||
} // namespace FEXCore::StrConv
|
||||
@@ -29,7 +29,7 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class Context;
|
||||
class Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Config {
|
||||
@@ -40,472 +40,464 @@ namespace DefaultValues {
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
} // namespace DefaultValues
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
PATH_CONFIG_DIR_GLOBAL,
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
PATH_CONFIG_FILE_GLOBAL,
|
||||
PATH_CONFIG_TELEMETRY_FOLDER,
|
||||
PATH_LAST,
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
|
||||
void SetDataDirectory(const std::string_view Path) {
|
||||
Paths[PATH_DATA_DIR] = Path;
|
||||
}
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
PATH_CONFIG_DIR_GLOBAL,
|
||||
PATH_CONFIG_FILE_LOCAL,
|
||||
PATH_CONFIG_FILE_GLOBAL,
|
||||
PATH_LAST,
|
||||
};
|
||||
static std::array<fextl::string, Paths::PATH_LAST> Paths;
|
||||
void SetConfigDirectory(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_DIR_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
void SetDataDirectory(const std::string_view Path) {
|
||||
Paths[PATH_DATA_DIR] = Path;
|
||||
}
|
||||
void SetConfigFileLocation(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
void SetConfigDirectory(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_DIR_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
void SetConfigFileLocation(const std::string_view Path, bool Global) {
|
||||
Paths[PATH_CONFIG_FILE_LOCAL + Global] = Path;
|
||||
}
|
||||
|
||||
fextl::string const& GetDataDirectory() {
|
||||
return Paths[PATH_DATA_DIR];
|
||||
}
|
||||
|
||||
fextl::string const& GetConfigDirectory(bool Global) {
|
||||
return Paths[PATH_CONFIG_DIR_LOCAL + Global];
|
||||
}
|
||||
|
||||
fextl::string const& GetConfigFileLocation(bool Global) {
|
||||
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
|
||||
}
|
||||
|
||||
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
|
||||
fextl::string ConfigFile = GetConfigDirectory(Global);
|
||||
|
||||
if (!Global &&
|
||||
!FHU::Filesystem::Exists(ConfigFile) &&
|
||||
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
|
||||
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
|
||||
// Let's go local in this case
|
||||
return fextl::fmt::format("./{}.json", Program);
|
||||
const fextl::string& GetTelemetryDirectory() {
|
||||
auto& Path = Paths[PATH_CONFIG_TELEMETRY_FOLDER];
|
||||
if (Path.empty()) {
|
||||
FEX_CONFIG_OPT(TelemetryDirectory, TELEMETRYDIRECTORY);
|
||||
if (!TelemetryDirectory().empty()) {
|
||||
Path = TelemetryDirectory;
|
||||
Path += "/";
|
||||
} else {
|
||||
Path = Config::GetDataDirectory() + "Telemetry/";
|
||||
}
|
||||
}
|
||||
|
||||
ConfigFile += "AppConfig/";
|
||||
return Path;
|
||||
}
|
||||
|
||||
// Attempt to create the local folder if it doesn't exist
|
||||
if (!Global &&
|
||||
!FHU::Filesystem::Exists(ConfigFile) &&
|
||||
!FHU::Filesystem::CreateDirectories(ConfigFile)) {
|
||||
// Let's go local in this case
|
||||
return fextl::fmt::format("./{}.json", Program);
|
||||
const fextl::string& GetDataDirectory() {
|
||||
return Paths[PATH_DATA_DIR];
|
||||
}
|
||||
|
||||
const fextl::string& GetConfigDirectory(bool Global) {
|
||||
return Paths[PATH_CONFIG_DIR_LOCAL + Global];
|
||||
}
|
||||
|
||||
const fextl::string& GetConfigFileLocation(bool Global) {
|
||||
return Paths[PATH_CONFIG_FILE_LOCAL + Global];
|
||||
}
|
||||
|
||||
fextl::string GetApplicationConfig(const std::string_view Program, bool Global) {
|
||||
fextl::string ConfigFile = GetConfigDirectory(Global);
|
||||
|
||||
if (!Global && !FHU::Filesystem::Exists(ConfigFile) && !FHU::Filesystem::CreateDirectories(ConfigFile)) {
|
||||
LogMan::Msg::DFmt("Couldn't create config directory: '{}'", ConfigFile);
|
||||
// Let's go local in this case
|
||||
return fextl::fmt::format("./{}.json", Program);
|
||||
}
|
||||
|
||||
ConfigFile += "AppConfig/";
|
||||
|
||||
// Attempt to create the local folder if it doesn't exist
|
||||
if (!Global && !FHU::Filesystem::Exists(ConfigFile) && !FHU::Filesystem::CreateDirectories(ConfigFile)) {
|
||||
// Let's go local in this case
|
||||
return fextl::fmt::format("./{}.json", Program);
|
||||
}
|
||||
|
||||
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, uint64_t Config) {}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, const fextl::string& Config) {}
|
||||
|
||||
uint64_t GetConfig(FEXCore::Context::Context* CTX, ConfigOption Option) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer* Meta {};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 10> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN, FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP, FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP, FEXCore::Config::LayerType::LAYER_LOCAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_ARGUMENTS, FEXCore::Config::LayerType::LAYER_USER_OVERRIDE,
|
||||
FEXCore::Config::LayerType::LAYER_ENVIRONMENT, FEXCore::Config::LayerType::LAYER_TOP};
|
||||
|
||||
Layer::Layer(const LayerType _Type)
|
||||
: Type {_Type} {}
|
||||
|
||||
Layer::~Layer() {}
|
||||
|
||||
class MetaLayer final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
MetaLayer(const LayerType _Type)
|
||||
: FEXCore::Config::Layer(_Type) {}
|
||||
~MetaLayer() {}
|
||||
void Load();
|
||||
|
||||
private:
|
||||
void MergeConfigMap(const LayerOptions& Options);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value);
|
||||
};
|
||||
|
||||
void MetaLayer::Load() {
|
||||
OptionMap.clear();
|
||||
|
||||
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
|
||||
auto it = ConfigLayers.find(*CurrentLayer);
|
||||
if (it != ConfigLayers.end() && *CurrentLayer != Type) {
|
||||
// Merge this layer's options to this layer
|
||||
MergeConfigMap(it->second->GetOptionMap());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
|
||||
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value) {
|
||||
// Environment variables need a bit of additional work
|
||||
// We want to merge the arrays rather than overwrite entirely
|
||||
auto MetaEnvironment = OptionMap.find(Option);
|
||||
if (MetaEnvironment == OptionMap.end()) {
|
||||
// Doesn't exist, just insert
|
||||
OptionMap.insert_or_assign(Option, Value);
|
||||
return;
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, uint64_t Config) {
|
||||
}
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](const FEXCore::Config::LayerValue& Value) {
|
||||
for (const auto& EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == fextl::string::npos) {
|
||||
// Broken environment variable
|
||||
// Skip
|
||||
continue;
|
||||
}
|
||||
auto Key = fextl::string(EnvVar.begin(), EnvVar.begin() + ItEq);
|
||||
auto Value = fextl::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
|
||||
|
||||
void SetConfig(FEXCore::Context::Context *CTX, ConfigOption Option, fextl::string const &Config) {
|
||||
}
|
||||
|
||||
uint64_t GetConfig(FEXCore::Context::Context *CTX, ConfigOption Option) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 9> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_ARGUMENTS,
|
||||
FEXCore::Config::LayerType::LAYER_ENVIRONMENT,
|
||||
FEXCore::Config::LayerType::LAYER_TOP
|
||||
// Add the key to the map, overwriting whatever previous value was there
|
||||
LookupMap.insert_or_assign(std::move(Key), std::move(Value));
|
||||
}
|
||||
};
|
||||
|
||||
Layer::Layer(const LayerType _Type)
|
||||
: Type {_Type} {
|
||||
AddToMap(MetaEnvironment->second);
|
||||
AddToMap(Value);
|
||||
|
||||
// Now with the two layers merged in the map
|
||||
// Add all the values to the option
|
||||
Erase(Option);
|
||||
for (auto& Val : LookupMap) {
|
||||
// Set will emplace multiple options in to its list
|
||||
Set(Option, Val.first + "=" + Val.second);
|
||||
}
|
||||
}
|
||||
|
||||
Layer::~Layer() {
|
||||
}
|
||||
|
||||
class MetaLayer final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
MetaLayer(const LayerType _Type)
|
||||
: FEXCore::Config::Layer (_Type) {
|
||||
}
|
||||
~MetaLayer() {
|
||||
}
|
||||
void Load();
|
||||
|
||||
private:
|
||||
void MergeConfigMap(const LayerOptions &Options);
|
||||
void MergeEnvironmentVariables(ConfigOption const &Option, LayerValue const &Value);
|
||||
};
|
||||
|
||||
void MetaLayer::Load() {
|
||||
OptionMap.clear();
|
||||
|
||||
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
|
||||
auto it = ConfigLayers.find(*CurrentLayer);
|
||||
if (it != ConfigLayers.end() && *CurrentLayer != Type) {
|
||||
// Merge this layer's options to this layer
|
||||
MergeConfigMap(it->second->GetOptionMap());
|
||||
}
|
||||
void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto& it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
MergeEnvironmentVariables(it.first, it.second);
|
||||
} else {
|
||||
OptionMap.insert_or_assign(it.first, it.second);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Initialize() {
|
||||
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
|
||||
Meta = ConfigLayers.begin()->second.get();
|
||||
}
|
||||
|
||||
void MetaLayer::MergeEnvironmentVariables(ConfigOption const &Option, LayerValue const &Value) {
|
||||
// Environment variables need a bit of additional work
|
||||
// We want to merge the arrays rather than overwrite entirely
|
||||
auto MetaEnvironment = OptionMap.find(Option);
|
||||
if (MetaEnvironment == OptionMap.end()) {
|
||||
// Doesn't exist, just insert
|
||||
OptionMap.insert_or_assign(Option, Value);
|
||||
return;
|
||||
}
|
||||
void Shutdown() {
|
||||
ConfigLayers.clear();
|
||||
Meta = nullptr;
|
||||
}
|
||||
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](FEXCore::Config::LayerValue const &Value) {
|
||||
for (const auto &EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == fextl::string::npos) {
|
||||
// Broken environment variable
|
||||
// Skip
|
||||
continue;
|
||||
}
|
||||
auto Key = fextl::string(EnvVar.begin(), EnvVar.begin() + ItEq);
|
||||
auto Value = fextl::string(EnvVar.begin() + ItEq + 1, EnvVar.end());
|
||||
|
||||
// Add the key to the map, overwriting whatever previous value was there
|
||||
LookupMap.insert_or_assign(std::move(Key), std::move(Value));
|
||||
}
|
||||
};
|
||||
|
||||
AddToMap(MetaEnvironment->second);
|
||||
AddToMap(Value);
|
||||
|
||||
// Now with the two layers merged in the map
|
||||
// Add all the values to the option
|
||||
Erase(Option);
|
||||
for (auto &Val : LookupMap) {
|
||||
// Set will emplace multiple options in to its list
|
||||
Set(Option, Val.first + "=" + Val.second);
|
||||
void Load() {
|
||||
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
|
||||
auto it = ConfigLayers.find(*CurrentLayer);
|
||||
if (it != ConfigLayers.end()) {
|
||||
it->second->Load();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void MetaLayer::MergeConfigMap(const LayerOptions &Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto &it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV ||
|
||||
it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
MergeEnvironmentVariables(it.first, it.second);
|
||||
}
|
||||
else {
|
||||
OptionMap.insert_or_assign(it.first, it.second);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Initialize() {
|
||||
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
|
||||
Meta = ConfigLayers.begin()->second.get();
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
ConfigLayers.clear();
|
||||
Meta = nullptr;
|
||||
}
|
||||
|
||||
void Load() {
|
||||
for (auto CurrentLayer = LoadOrder.begin(); CurrentLayer != LoadOrder.end(); ++CurrentLayer) {
|
||||
auto it = ConfigLayers.find(*CurrentLayer);
|
||||
if (it != ConfigLayers.end()) {
|
||||
it->second->Load();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::string ExpandPath(fextl::string const &ContainerPrefix, fextl::string PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
|
||||
// Expand home if it exists
|
||||
if (FHU::Filesystem::IsRelative(PathName)) {
|
||||
fextl::string Home = getenv("HOME") ?: "";
|
||||
// Home expansion only works if it is the first character
|
||||
// This matches bash behaviour
|
||||
if (PathName.at(0) == '~') {
|
||||
PathName.replace(0, 1, Home);
|
||||
return PathName;
|
||||
}
|
||||
|
||||
// Expand relative path to absolute
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
char *RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
|
||||
if (RealPath) {
|
||||
PathName = RealPath;
|
||||
}
|
||||
|
||||
// Only return if it exists
|
||||
if (FHU::Filesystem::Exists(PathName)) {
|
||||
return PathName;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// If the containerprefix and pathname isn't empty
|
||||
// Then we check if the pathname exists in our current namespace
|
||||
// If the path DOESN'T exist but DOES exist with the prefix applied
|
||||
// then redirect to the prefix
|
||||
//
|
||||
// This might not be expected behaviour for some edge cases but since
|
||||
// all paths aren't mounted inside the container, then it'll be fine
|
||||
//
|
||||
// Main catch case for this is the default thunk install folders
|
||||
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
|
||||
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
|
||||
if (!ContainerPrefix.empty() && !PathName.empty()) {
|
||||
if (!FHU::Filesystem::Exists(PathName)) {
|
||||
auto ContainerPath = ContainerPrefix + PathName;
|
||||
if (FHU::Filesystem::Exists(ContainerPath)) {
|
||||
return ContainerPath;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
fextl::string ExpandPath(const fextl::string& ContainerPrefix, fextl::string PathName) {
|
||||
if (PathName.empty()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
constexpr char ContainerManager[] = "/run/host/container-manager";
|
||||
|
||||
fextl::string FindContainer() {
|
||||
// We only support pressure-vessel at the moment
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
return ManagerStr;
|
||||
}
|
||||
// Expand home if it exists
|
||||
if (FHU::Filesystem::IsRelative(PathName)) {
|
||||
fextl::string Home = getenv("HOME") ?: "";
|
||||
// Home expansion only works if it is the first character
|
||||
// This matches bash behaviour
|
||||
if (PathName.at(0) == '~') {
|
||||
PathName.replace(0, 1, Home);
|
||||
return PathName;
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
fextl::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
|
||||
// We are running inside of pressure vessel
|
||||
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
|
||||
return "/run/host/";
|
||||
// Expand relative path to absolute
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
char* RealPath = FHU::Filesystem::Absolute(PathName.c_str(), ExistsTempPath);
|
||||
if (RealPath) {
|
||||
PathName = RealPath;
|
||||
}
|
||||
|
||||
// Only return if it exists
|
||||
if (FHU::Filesystem::Exists(PathName)) {
|
||||
return PathName;
|
||||
}
|
||||
} else {
|
||||
// If the containerprefix and pathname isn't empty
|
||||
// Then we check if the pathname exists in our current namespace
|
||||
// If the path DOESN'T exist but DOES exist with the prefix applied
|
||||
// then redirect to the prefix
|
||||
//
|
||||
// This might not be expected behaviour for some edge cases but since
|
||||
// all paths aren't mounted inside the container, then it'll be fine
|
||||
//
|
||||
// Main catch case for this is the default thunk install folders
|
||||
// HostThunks: $CMAKE_INSTALL_PREFIX/lib/fex-emu/HostThunks/
|
||||
// GuestThunks: $CMAKE_INSTALL_PREFIX/share/fex-emu/GuestThunks/
|
||||
if (!ContainerPrefix.empty() && !PathName.empty()) {
|
||||
if (!FHU::Filesystem::Exists(PathName)) {
|
||||
auto ContainerPath = ContainerPrefix + PathName;
|
||||
if (FHU::Filesystem::Exists(ContainerPath)) {
|
||||
return ContainerPath;
|
||||
}
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
void ReloadMetaLayer() {
|
||||
Meta->Load();
|
||||
constexpr char ContainerManager[] = "/run/host/container-manager";
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
fextl::string FindContainer() {
|
||||
// We only support pressure-vessel at the moment
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::vector<char> Manager {};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
return ManagerStr;
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
fextl::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
if (FHU::Filesystem::Exists(ContainerManager)) {
|
||||
fextl::vector<char> Manager {};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
fextl::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
if (strncmp(ManagerStr.data(), "pressure-vessel", Manager.size()) == 0) {
|
||||
// We are running inside of pressure vessel
|
||||
// Our $CMAKE_INSTALL_PREFIX paths are now inside of /run/host/$CMAKE_INSTALL_PREFIX
|
||||
return "/run/host/";
|
||||
}
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
void ReloadMetaLayer() {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
#if (_M_X86_64)
|
||||
constexpr uint32_t MaxCoreNumber = 1;
|
||||
constexpr uint32_t MaxCoreNumber = 1;
|
||||
#else
|
||||
constexpr uint32_t MaxCoreNumber = 0;
|
||||
constexpr uint32_t MaxCoreNumber = 0;
|
||||
#endif
|
||||
if (Core > MaxCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
|
||||
if (Core > MaxCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast<uint32_t>(FEXCore::Config::CONFIG_IRJIT)));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
}
|
||||
|
||||
fextl::string ContainerPrefix {FindContainerPrefix()};
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
}
|
||||
};
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
|
||||
FEX_CONFIG_OPT(PathName, ROOTFS);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
} else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CACHEOBJECTCODECOMPILATION)) {
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
}
|
||||
|
||||
fextl::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, fextl::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
}
|
||||
};
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_ROOTFS)) {
|
||||
FEX_CONFIG_OPT(PathName, ROOTFS);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix,PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
}
|
||||
else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedRootFS = GetDataDirectory() + "RootFS/" + PathName();
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKHOSTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKHOSTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, PathName());
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKGUESTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKGUESTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, PathName());
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
} else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKHOSTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKHOSTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKHOSTLIBS, PathName());
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKGUESTLIBS)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKGUESTLIBS);
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_THUNKGUESTLIBS, PathName());
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THUNKCONFIG)) {
|
||||
FEX_CONFIG_OPT(PathName, THUNKCONFIG);
|
||||
auto ExpandedString = ExpandPath(ContainerPrefix, PathName());
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
}
|
||||
else if (!PathName().empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedConfig = GetDataDirectory() + "ThunkConfigs/" + PathName();
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
|
||||
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
|
||||
if (PathName() != "stdout" && PathName() != "stderr" && PathName() != "server") {
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, PathName());
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_DUMPIR) &&
|
||||
!FEXCore::Config::Exists(FEXCore::Config::CONFIG_PASSMANAGERDUMPIR)) {
|
||||
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
|
||||
FEX_CONFIG_OPT(PathName, DUMPIR);
|
||||
if (PathName() != "no") {
|
||||
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR, fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
|
||||
// Single stepping also enforces single instruction size blocks
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_OUTPUTLOG)) {
|
||||
FEX_CONFIG_OPT(PathName, OUTPUTLOG);
|
||||
if (PathName() != "stdout" && PathName() != "stderr" && PathName() != "server") {
|
||||
ExpandPathIfExists(FEXCore::Config::CONFIG_OUTPUTLOG, PathName());
|
||||
}
|
||||
}
|
||||
|
||||
void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer) {
|
||||
ConfigLayers.emplace(_Layer->GetLayerType(), std::move(_Layer));
|
||||
}
|
||||
|
||||
bool Exists(ConfigOption Option) {
|
||||
return Meta->OptionExists(Option);
|
||||
}
|
||||
|
||||
std::optional<LayerValue*> All(ConfigOption Option) {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
return Meta->Get(Option);
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
|
||||
void Erase(ConfigOption Option) {
|
||||
Meta->Erase(Option);
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string_view Data) {
|
||||
Meta->EraseSet(Option, Data);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::Get(FEXCore::Config::ConfigOption Option) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::GetIfExists(FEXCore::Config::ConfigOption Option, T Default) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (Value && FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
return Result;
|
||||
}
|
||||
else {
|
||||
return Default;
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_DUMPIR) && !FEXCore::Config::Exists(FEXCore::Config::CONFIG_PASSMANAGERDUMPIR)) {
|
||||
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
|
||||
FEX_CONFIG_OPT(PathName, DUMPIR);
|
||||
if (PathName() != "no") {
|
||||
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
}
|
||||
}
|
||||
|
||||
template<>
|
||||
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, fextl::string Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
}
|
||||
else {
|
||||
return Default;
|
||||
}
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
|
||||
// Single stepping also enforces single instruction size blocks
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
|
||||
}
|
||||
|
||||
template<>
|
||||
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
}
|
||||
else {
|
||||
return fextl::string(Default);
|
||||
}
|
||||
}
|
||||
|
||||
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
|
||||
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
|
||||
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
|
||||
template int16_t Value<int16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int16_t Default);
|
||||
template uint16_t Value<uint16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint16_t Default);
|
||||
template int32_t Value<int32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int32_t Default);
|
||||
template uint32_t Value<uint32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint32_t Default);
|
||||
template int64_t Value<int64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int64_t Default);
|
||||
template uint64_t Value<uint64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint64_t Default);
|
||||
|
||||
// Constructor
|
||||
template Value<fextl::string>::Value(FEXCore::Config::ConfigOption _Option, fextl::string Default);
|
||||
template Value<bool>::Value(FEXCore::Config::ConfigOption _Option, bool Default);
|
||||
template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t Default);
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
}
|
||||
|
||||
void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer) {
|
||||
ConfigLayers.emplace(_Layer->GetLayerType(), std::move(_Layer));
|
||||
}
|
||||
|
||||
bool Exists(ConfigOption Option) {
|
||||
return Meta->OptionExists(Option);
|
||||
}
|
||||
|
||||
std::optional<LayerValue*> All(ConfigOption Option) {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
return Meta->Get(Option);
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
|
||||
void Erase(ConfigOption Option) {
|
||||
Meta->Erase(Option);
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string_view Data) {
|
||||
Meta->EraseSet(Option, Data);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::Get(FEXCore::Config::ConfigOption Option) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::GetIfExists(FEXCore::Config::ConfigOption Option, T Default) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (Value && FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
return Result;
|
||||
} else {
|
||||
return Default;
|
||||
}
|
||||
}
|
||||
|
||||
template<>
|
||||
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, fextl::string Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
} else {
|
||||
return Default;
|
||||
}
|
||||
}
|
||||
|
||||
template<>
|
||||
fextl::string Value<fextl::string>::GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
if (Value) {
|
||||
return **Value;
|
||||
} else {
|
||||
return fextl::string(Default);
|
||||
}
|
||||
}
|
||||
|
||||
template bool Value<bool>::GetIfExists(FEXCore::Config::ConfigOption Option, bool Default);
|
||||
template int8_t Value<int8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int8_t Default);
|
||||
template uint8_t Value<uint8_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint8_t Default);
|
||||
template int16_t Value<int16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int16_t Default);
|
||||
template uint16_t Value<uint16_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint16_t Default);
|
||||
template int32_t Value<int32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int32_t Default);
|
||||
template uint32_t Value<uint32_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint32_t Default);
|
||||
template int64_t Value<int64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, int64_t Default);
|
||||
template uint64_t Value<uint64_t>::GetIfExists(FEXCore::Config::ConfigOption Option, uint64_t Default);
|
||||
|
||||
// Constructor
|
||||
template Value<fextl::string>::Value(FEXCore::Config::ConfigOption _Option, fextl::string Default);
|
||||
template Value<bool>::Value(FEXCore::Config::ConfigOption _Option, bool Default);
|
||||
template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t Default);
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List);
|
||||
} // namespace FEXCore::Config
|
||||
@@ -368,6 +368,14 @@
|
||||
"File to write FEX output to.",
|
||||
"[stdout, stderr, server, <Filename>]"
|
||||
]
|
||||
},
|
||||
"TelemetryDirectory": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Redirects the telemetry folder that FEX usually writes to.",
|
||||
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
@@ -379,9 +387,8 @@
|
||||
"Desc": [
|
||||
"Checks code for modification before execution.",
|
||||
"\tnone: No checks",
|
||||
"\tmtrack: Page tracking based invalidation",
|
||||
"\tfull: Validate code before every run (slow)",
|
||||
"\tmman: Invalidate on mmap, mprotect, munmap (deprecated, use mtrack)"
|
||||
"\tmtrack: Page tracking based invalidation (default)",
|
||||
"\tfull: Validate code before every run (slow)"
|
||||
]
|
||||
},
|
||||
"TSOEnabled": {
|
||||
@@ -392,6 +399,29 @@
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
]
|
||||
},
|
||||
"VectorTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
|
||||
]
|
||||
},
|
||||
"MemcpySetTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
|
||||
"Only affects REP MOVS and REP STOS instructions"
|
||||
]
|
||||
},
|
||||
"HalfBarrierTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if unaligned loads and stores should be backpatched to half-barrier atomics.",
|
||||
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
@@ -439,6 +469,14 @@
|
||||
"Hides the hypervisor CPUID bit when set.",
|
||||
"Should only be used for applications that have issues with this set."
|
||||
]
|
||||
},
|
||||
"StartupSleep": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"Desc": [
|
||||
"Sleeps the process at startup for a duration of seconds.",
|
||||
"Useful if an application crashes too quickly to attach a debugger."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
|
||||
@@ -13,57 +13,61 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void InitializeStaticTables(OperatingMode Mode) {
|
||||
X86Tables::InitializeInfoTables(Mode);
|
||||
IR::InstallOpcodeHandlers(Mode);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext() {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
CustomExitHandler = std::move(handler);
|
||||
}
|
||||
|
||||
ExitHandler FEXCore::Context::ContextImpl::GetExitHandler() const {
|
||||
return CustomExitHandler;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
|
||||
return HostFeatures;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator *_SignalDelegation) {
|
||||
SignalDelegation = _SignalDelegation;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) {
|
||||
SyscallHandler = Handler;
|
||||
SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
|
||||
return CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
|
||||
return CPUID.RunXCRFunction(Function);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
}
|
||||
void InitializeStaticTables(OperatingMode Mode) {
|
||||
X86Tables::InitializeInfoTables(Mode);
|
||||
IR::InstallOpcodeHandlers(Mode);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext() {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
CustomExitHandler = std::move(handler);
|
||||
}
|
||||
|
||||
ExitHandler FEXCore::Context::ContextImpl::GetExitHandler() const {
|
||||
return CustomExitHandler;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
|
||||
return HostFeatures;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetSignalDelegator(FEXCore::SignalDelegator* _SignalDelegation) {
|
||||
SignalDelegation = _SignalDelegation;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) {
|
||||
SyscallHandler = Handler;
|
||||
SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunction(uint32_t Function, uint32_t Leaf) {
|
||||
return CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
|
||||
return CPUID.RunXCRFunction(Function);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const {
|
||||
return Thread->CPUBackend->IsAddressInCodeBuffer(Address);
|
||||
}
|
||||
} // namespace FEXCore::Context
|
||||
@@ -13,6 +13,7 @@
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
@@ -45,359 +46,376 @@ namespace CodeSerialize {
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class Dispatcher;
|
||||
}
|
||||
} // namespace CPU
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
struct SourcecodeMap;
|
||||
}
|
||||
}
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
struct SourcecodeMap;
|
||||
} // namespace HLE
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
namespace Validation {
|
||||
class IRValidation;
|
||||
}
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::Context {
|
||||
enum CoreRunningMode {
|
||||
MODE_RUN = 0,
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
enum CoreRunningMode {
|
||||
MODE_RUN = 0,
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
|
||||
struct ExitFunctionLinkData {
|
||||
uint64_t HostBranch;
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
struct ExitFunctionLinkData {
|
||||
uint64_t HostBranch;
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
|
||||
using BlockDelinkerFunc = void(*)(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record);
|
||||
constexpr uint32_t TSC_SCALE = 128;
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE = 128;
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
|
||||
void SetExitHandler(ExitHandler handler) override;
|
||||
ExitHandler GetExitHandler() const override;
|
||||
void SetExitHandler(ExitHandler handler) override;
|
||||
ExitHandler GetExitHandler() const override;
|
||||
|
||||
ExitReason RunUntilExit(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
ExitReason RunUntilExit(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void ExecuteThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState* Thread, uint64_t RIP) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) override;
|
||||
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) override;
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - CTX->RunUntilExit(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
* - ThreadHandler calls `CTX->ExecutionThread(Thread)`
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - CTX->RunUntilExit(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
* - ThreadHandler calls `CTX->ExecutionThread(Thread)`
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
|
||||
FEXCore::Core::InternalThreadState*
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread, bool NeedsTLSUninstall) override;
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState* Thread, bool NeedsTLSUninstall) override;
|
||||
|
||||
#ifndef _WIN32
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override;
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState* Thread, bool Child) override;
|
||||
#endif
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator* SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) override;
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) override;
|
||||
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
|
||||
|
||||
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
|
||||
}
|
||||
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
|
||||
}
|
||||
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
|
||||
}
|
||||
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
|
||||
}
|
||||
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
|
||||
}
|
||||
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache() override {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
void FinalizeAOTIRCache() override {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
|
||||
FEXCore::ForkableSharedMutex &GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
|
||||
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
|
||||
void MarkMemoryShared(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr);
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
|
||||
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
|
||||
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
friend class FEXCore::CPU::Arm64JITCore;
|
||||
#endif
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator = nullptr, void* Data = nullptr);
|
||||
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
void AppendThunkDefinitions(const fextl::vector<FEXCore::IR::ThunkDefinition>& Definitions) override;
|
||||
|
||||
struct {
|
||||
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
|
||||
uint64_t VirtualMemSize{1ULL << 36};
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
friend class FEXCore::CPU::Arm64JITCore;
|
||||
#endif
|
||||
|
||||
// this is for internal use
|
||||
bool ValidateIRarser { false };
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
// Used if the JIT needs to have its interrupt fault code emitted.
|
||||
bool NeedsPendingInterruptFaultCheck { false };
|
||||
struct {
|
||||
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
|
||||
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
|
||||
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
|
||||
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
|
||||
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
} Config;
|
||||
// this is for internal use
|
||||
bool ValidateIRarser {false};
|
||||
|
||||
// Used if the JIT needs to have its interrupt fault code emitted.
|
||||
bool NeedsPendingInterruptFaultCheck {false};
|
||||
|
||||
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
|
||||
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
|
||||
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath, THUNKHOSTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkHostLibsPath32, THUNKHOSTLIBS32);
|
||||
FEX_CONFIG_OPT(ThunkConfigFile, THUNKCONFIG);
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
} Config;
|
||||
|
||||
|
||||
std::atomic_bool CoreShuttingDown{false};
|
||||
std::atomic_bool CoreShuttingDown {false};
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
// CPUID depends on HostFeatures so needs to be initialized after that.
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
|
||||
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
// CPUID depends on HostFeatures so needs to be initialized after that.
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler {};
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
fextl::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
fextl::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
#endif
|
||||
|
||||
SignalDelegator *SignalDelegation{};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
ContextImpl();
|
||||
~ContextImpl();
|
||||
ContextImpl();
|
||||
~ContextImpl();
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData *HostLink, const BlockDelinkerFunc &delinker);
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestDestination,
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink, const BlockDelinkerFunc& delinker);
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, ExitFunctionLinkData *Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, ExitFunctionLinkData* Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
return Fn(Frame, Record);
|
||||
}
|
||||
return Fn(Frame, Record);
|
||||
}
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
LOGMAN_THROW_A_FMT(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}",
|
||||
Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
FEXCore::IR::IRListView* IRData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
bool GeneratedIR;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Initializes TID, PID and TLS data for a thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
void GetVDSOSigReturn(VDSOSigReturn *VDSOPointers) override {
|
||||
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
|
||||
}
|
||||
|
||||
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const { return AtomicTSOEmulationEnabled; }
|
||||
|
||||
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
|
||||
SupportsHardwareTSO = HardwareTSOSupported;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
// Returns if Software TSO emulation is required.
|
||||
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
|
||||
// This will still return true if on a single thread and TSO is currently disabled.
|
||||
//
|
||||
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
|
||||
// we return consistent results.
|
||||
//
|
||||
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
|
||||
bool SoftwareTSORequired() const {
|
||||
if (SupportsHardwareTSO) return false;
|
||||
|
||||
return Config.TSOEnabled;
|
||||
}
|
||||
|
||||
void EnableExitOnHLT() override { ExitOnHLT = true; }
|
||||
|
||||
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
|
||||
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
|
||||
protected:
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
}
|
||||
else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool ExitOnHLT = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::atomic<bool> HasCustomIRHandlers{};
|
||||
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void *, void *>> CustomIRHandlers;
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
}
|
||||
[[nodiscard]]
|
||||
GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
FEXCore::IR::IRListView* IRData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
bool GeneratedIR;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]]
|
||||
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Initializes TID, PID and TLS data for a thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState* ParentThread, FEXCore::Core::InternalThreadState* ChildThread);
|
||||
|
||||
uint8_t GetGPRSize() const {
|
||||
return Config.Is64BitMode ? 8 : 4;
|
||||
}
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
void GetVDSOSigReturn(VDSOSigReturn* VDSOPointers) override {
|
||||
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
|
||||
}
|
||||
|
||||
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const {
|
||||
return AtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
|
||||
SupportsHardwareTSO = HardwareTSOSupported;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
// Returns if Software TSO emulation is required.
|
||||
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
|
||||
// This will still return true if on a single thread and TSO is currently disabled.
|
||||
//
|
||||
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
|
||||
// we return consistent results.
|
||||
//
|
||||
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
|
||||
bool SoftwareTSORequired() const {
|
||||
if (SupportsHardwareTSO) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return Config.TSOEnabled;
|
||||
}
|
||||
|
||||
void EnableExitOnHLT() override {
|
||||
ExitOnHLT = true;
|
||||
}
|
||||
|
||||
bool ExitOnHLTEnabled() const {
|
||||
return ExitOnHLT;
|
||||
}
|
||||
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
|
||||
protected:
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
} else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
* @param State The internal FEX thread state object
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* Ptr);
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool ExitOnHLT = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::atomic<bool> HasCustomIRHandlers {};
|
||||
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void*, void*>> CustomIRHandlers;
|
||||
};
|
||||
} // namespace FEXCore::Context
|
||||
@@ -32,24 +32,31 @@ namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29,
|
||||
FEXCore::ARMEmitter::Reg::r4,
|
||||
FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6,
|
||||
FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8,
|
||||
FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10,
|
||||
FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12,
|
||||
FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14,
|
||||
FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16,
|
||||
FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
FEXCore::ARMEmitter::Reg::r29,
|
||||
// PF/AF must be last.
|
||||
REG_PF, REG_AF,
|
||||
REG_PF,
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
|
||||
@@ -60,48 +67,47 @@ namespace x64 {
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
#else
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r0,
|
||||
FEXCore::ARMEmitter::Reg::r1, FEXCore::ARMEmitter::Reg::r27,
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r8,
|
||||
FEXCore::ARMEmitter::Reg::r0,
|
||||
FEXCore::ARMEmitter::Reg::r1,
|
||||
FEXCore::ARMEmitter::Reg::r27,
|
||||
// SP's register location isn't specified by the ARM64EC ABI, we choose to use r23
|
||||
FEXCore::ARMEmitter::Reg::r23, FEXCore::ARMEmitter::Reg::r29,
|
||||
FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r26,
|
||||
FEXCore::ARMEmitter::Reg::r2, FEXCore::ARMEmitter::Reg::r3,
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r20,
|
||||
FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22,
|
||||
REG_PF, REG_AF,
|
||||
FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r29,
|
||||
FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26,
|
||||
FEXCore::ARMEmitter::Reg::r2,
|
||||
FEXCore::ARMEmitter::Reg::r3,
|
||||
FEXCore::ARMEmitter::Reg::r4,
|
||||
FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
FEXCore::ARMEmitter::Reg::r20,
|
||||
FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22,
|
||||
REG_PF,
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r14,FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
|
||||
@@ -111,142 +117,131 @@ namespace x64 {
|
||||
}};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1, FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9, FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13, FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19, FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23, FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27, FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
|
||||
#endif
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAMask = {
|
||||
[]() -> uint32_t {
|
||||
uint32_t Mask{};
|
||||
for (auto Reg : PreserveAll_SRA) {
|
||||
switch (Reg.Idx()) {
|
||||
case 0:
|
||||
case 1:
|
||||
case 2:
|
||||
case 3:
|
||||
case 4:
|
||||
case 5:
|
||||
case 6:
|
||||
case 7:
|
||||
case 8:
|
||||
case 16:
|
||||
case 17:
|
||||
Mask |= (1U << Reg.Idx());
|
||||
break;
|
||||
default: break;
|
||||
}
|
||||
constexpr uint32_t PreserveAll_SRAMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRA) {
|
||||
switch (Reg.Idx()) {
|
||||
case 0:
|
||||
case 1:
|
||||
case 2:
|
||||
case 3:
|
||||
case 4:
|
||||
case 5:
|
||||
case 6:
|
||||
case 7:
|
||||
case 8:
|
||||
case 16:
|
||||
case 17: Mask |= (1U << Reg.Idx()); break;
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
|
||||
return Mask;
|
||||
}()
|
||||
};
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 1> PreserveAll_Dynamic = {
|
||||
// Only LR needs to get saved.
|
||||
FEXCore::ARMEmitter::Reg::r30
|
||||
};
|
||||
FEXCore::ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {
|
||||
[]() -> uint32_t {
|
||||
uint32_t Mask{};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()
|
||||
};
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
// v0 ~ v1 are temps
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4,
|
||||
FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
|
||||
// This is /all/ of the SRA registers
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> PreserveAll_SRAFPRSVE = SRAFPR;
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {
|
||||
[]() -> uint32_t {
|
||||
uint32_t Mask{};
|
||||
for (auto Reg : PreserveAll_SRAFPRSVE) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()
|
||||
};
|
||||
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPRSVE) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic FPRs when the host supports SVE-256bit.
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> PreserveAll_DynamicFPRSVE = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
}
|
||||
} // namespace x64
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 10> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r4,
|
||||
FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6,
|
||||
FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8,
|
||||
FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10,
|
||||
FEXCore::ARMEmitter::Reg::r11,
|
||||
// PF/AF must be last.
|
||||
REG_PF, REG_AF,
|
||||
REG_PF,
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 15> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r20,
|
||||
FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22,
|
||||
FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24,
|
||||
FEXCore::ARMEmitter::Reg::r25,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30,
|
||||
FEXCore::ARMEmitter::Reg::r12,
|
||||
FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14,
|
||||
FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16,
|
||||
FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r29,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
};
|
||||
@@ -264,10 +259,8 @@ namespace x32 {
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
@@ -275,118 +268,94 @@ namespace x32 {
|
||||
// v0 ~ v1 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 5> PreserveAll_SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8,
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6,
|
||||
FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r8,
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAMask = {
|
||||
[]() -> uint32_t {
|
||||
uint32_t Mask{};
|
||||
for (auto Reg : PreserveAll_SRA) {
|
||||
switch (Reg.Idx()) {
|
||||
case 0:
|
||||
case 1:
|
||||
case 2:
|
||||
case 3:
|
||||
case 4:
|
||||
case 5:
|
||||
case 6:
|
||||
case 7:
|
||||
case 8:
|
||||
case 16:
|
||||
case 17:
|
||||
Mask |= (1U << Reg.Idx());
|
||||
break;
|
||||
default: break;
|
||||
}
|
||||
constexpr uint32_t PreserveAll_SRAMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRA) {
|
||||
switch (Reg.Idx()) {
|
||||
case 0:
|
||||
case 1:
|
||||
case 2:
|
||||
case 3:
|
||||
case 4:
|
||||
case 5:
|
||||
case 6:
|
||||
case 7:
|
||||
case 8:
|
||||
case 16:
|
||||
case 17: Mask |= (1U << Reg.Idx()); break;
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
|
||||
return Mask;
|
||||
}()
|
||||
};
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 3> PreserveAll_Dynamic = {
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r30
|
||||
};
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {
|
||||
[]() -> uint32_t {
|
||||
uint32_t Mask{};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()
|
||||
};
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
// v0 ~ v1 are temps
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4,
|
||||
FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
|
||||
// This is /all/ of the SRA registers
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> PreserveAll_SRAFPRSVE = SRAFPR;
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {
|
||||
[]() -> uint32_t {
|
||||
uint32_t Mask{};
|
||||
for (auto Reg : PreserveAll_SRAFPRSVE) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()
|
||||
};
|
||||
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPRSVE) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic FPRs when the host supports SVE-256bit.
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 22> PreserveAll_DynamicFPRSVE = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
}
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
|
||||
} // namespace x32
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr, size_t size)
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr, size_t size)
|
||||
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
|
||||
, EmitterCTX {ctx}
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -422,8 +391,7 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr
|
||||
#ifdef _M_ARM_64EC
|
||||
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
|
||||
#endif
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
@@ -439,11 +407,13 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
|
||||
if (Is64Bit && ((~Constant)>> 16) == 0) {
|
||||
if (Is64Bit && ((~Constant) >> 16) == 0) {
|
||||
movn(s, Reg, (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
nop();
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -459,12 +429,14 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
movn(s, Reg.W(), (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
nop();
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
int RequiredMoveSegments{};
|
||||
int RequiredMoveSegments {};
|
||||
|
||||
// Count the number of move segments
|
||||
// We only want to use ADRP+ADD if we have more than 1 segment
|
||||
@@ -483,7 +455,9 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
if (IsImm) {
|
||||
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
nop();
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -507,23 +481,20 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
// If this is 4k page aligned then we only need ADRP
|
||||
if ((AlignedOffset & 0xFFF) == 0) {
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
// If the constant is within 1MB of PC then we can still use ADR to load in a single instruction
|
||||
// 21-bit signed integer here
|
||||
int64_t SmallOffset = static_cast<int64_t>(Constant) - static_cast<int64_t>(PC);
|
||||
if (vixl::IsInt21(SmallOffset)) {
|
||||
adr(Reg, SmallOffset);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
// Need to use ADRP + ADD
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
add(s, Reg, Reg, Constant & 0xFFF);
|
||||
NumMoves = 2;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
int CurrentSegment = 0;
|
||||
for (; CurrentSegment < Segments; ++CurrentSegment) {
|
||||
uint16_t Part = (Constant >> (CurrentSegment * 16)) & 0xFFFF;
|
||||
@@ -569,18 +540,14 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
|
||||
}};
|
||||
|
||||
for (auto &RegPair : CalleeSaved) {
|
||||
for (auto& RegPair : CalleeSaved) {
|
||||
stp<ARMEmitter::IndexType::PRE>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, -16);
|
||||
}
|
||||
|
||||
// Additionally we need to store the lower 64bits of v8-v15
|
||||
// Here's a fun thing, we can use two ST4 instructions to store everything
|
||||
// We just need a single sub to sp before that
|
||||
const std::array<
|
||||
std::tuple<ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister>, 2> FPRs = {{
|
||||
const std::array< std::tuple<ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister>, 2> FPRs = {{
|
||||
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
}};
|
||||
@@ -591,37 +558,21 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We just saved x19 so it is safe
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r19, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
for (auto &RegQuad : FPRs) {
|
||||
st4(ARMEmitter::SubRegSize::i64Bit,
|
||||
std::get<0>(RegQuad),
|
||||
std::get<1>(RegQuad),
|
||||
std::get<2>(RegQuad),
|
||||
std::get<3>(RegQuad),
|
||||
0,
|
||||
ARMEmitter::Reg::r19,
|
||||
32);
|
||||
for (auto& RegQuad : FPRs) {
|
||||
st4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
|
||||
ARMEmitter::Reg::r19, 32);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
const std::array<
|
||||
std::tuple<ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister,
|
||||
ARMEmitter::DRegister>, 2> FPRs = {{
|
||||
const std::array< std::tuple<ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister>, 2> FPRs = {{
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
|
||||
}};
|
||||
|
||||
for (auto &RegQuad : FPRs) {
|
||||
ld4(ARMEmitter::SubRegSize::i64Bit,
|
||||
std::get<0>(RegQuad),
|
||||
std::get<1>(RegQuad),
|
||||
std::get<2>(RegQuad),
|
||||
std::get<3>(RegQuad),
|
||||
0,
|
||||
ARMEmitter::Reg::rsp,
|
||||
32);
|
||||
for (auto& RegQuad : FPRs) {
|
||||
ld4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
|
||||
ARMEmitter::Reg::rsp, 32);
|
||||
}
|
||||
|
||||
const fextl::vector<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>> CalleeSaved = {{
|
||||
@@ -633,7 +584,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
|
||||
}};
|
||||
|
||||
for (auto &RegPair : CalleeSaved) {
|
||||
for (auto& RegPair : CalleeSaved) {
|
||||
ldp<ARMEmitter::IndexType::POST>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
}
|
||||
@@ -652,8 +603,8 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
@@ -671,18 +622,15 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
|
||||
GPRSpillMask &= ~PFAFSpillMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
auto Reg2 = StaticRegisters[i + 1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) && ((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
}
|
||||
else if (((1U << Reg1.Idx()) & GPRSpillMask)) {
|
||||
} else if (((1U << Reg1.Idx()) & GPRSpillMask)) {
|
||||
str(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
}
|
||||
else if (((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
str(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
|
||||
} else if (((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
str(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i + 1]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -716,21 +664,17 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) && ((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
}
|
||||
else if (((1U << Reg1.Idx()) & FPRSpillMask)) {
|
||||
} else if (((1U << Reg1.Idx()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
}
|
||||
else if (((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0]));
|
||||
} else if (((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i + 1][0]));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -741,7 +685,7 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
[[maybe_unused]] bool FoundRegister{};
|
||||
[[maybe_unused]] bool FoundRegister {};
|
||||
for (auto Reg : StaticRegisters) {
|
||||
if (((1U << Reg.Idx()) & GPRFillMask)) {
|
||||
TmpReg = Reg;
|
||||
@@ -765,8 +709,8 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
@@ -811,21 +755,17 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) && ((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
}
|
||||
else if (((1U << Reg1.Idx()) & FPRFillMask)) {
|
||||
} else if (((1U << Reg1.Idx()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
}
|
||||
else if (((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0]));
|
||||
} else if (((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i + 1][0]));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -837,18 +777,15 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
|
||||
GPRFillMask &= ~PFAFMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
auto Reg2 = StaticRegisters[i + 1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) && ((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
}
|
||||
else if ((1U << Reg1.Idx()) & GPRFillMask) {
|
||||
} else if ((1U << Reg1.Idx()) & GPRFillMask) {
|
||||
ldr(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
}
|
||||
else if ((1U << Reg2.Idx()) & GPRFillMask) {
|
||||
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
|
||||
} else if ((1U << Reg2.Idx()) & GPRFillMask) {
|
||||
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i + 1]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -880,8 +817,7 @@ void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, boo
|
||||
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
size_t i = 0;
|
||||
for (; i < (VRegs.size() % 4); i += 2) {
|
||||
const auto Reg1 = VRegs[i];
|
||||
@@ -965,8 +901,7 @@ void Arm64Emitter::PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Regi
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
@@ -1004,13 +939,12 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
|
||||
void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs{};
|
||||
uint32_t PreserveSRAMask{};
|
||||
uint32_t PreserveSRAFPRMask{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs {};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs {};
|
||||
uint32_t PreserveSRAMask {};
|
||||
uint32_t PreserveSRAFPRMask {};
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
DynamicGPRs = x64::PreserveAll_Dynamic;
|
||||
DynamicFPRs = x64::PreserveAll_DynamicFPR;
|
||||
@@ -1021,8 +955,7 @@ void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpR
|
||||
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
|
||||
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
DynamicGPRs = x32::PreserveAll_Dynamic;
|
||||
DynamicFPRs = x32::PreserveAll_DynamicFPR;
|
||||
PreserveSRAMask = x32::PreserveAll_SRAMask;
|
||||
@@ -1056,10 +989,10 @@ void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpR
|
||||
void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
|
||||
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs{};
|
||||
uint32_t PreserveSRAMask{};
|
||||
uint32_t PreserveSRAFPRMask{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs {};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs {};
|
||||
uint32_t PreserveSRAMask {};
|
||||
uint32_t PreserveSRAFPRMask {};
|
||||
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
DynamicGPRs = x64::PreserveAll_Dynamic;
|
||||
@@ -1071,8 +1004,7 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
|
||||
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
DynamicGPRs = x32::PreserveAll_Dynamic;
|
||||
DynamicFPRs = x32::PreserveAll_DynamicFPR;
|
||||
PreserveSRAMask = x32::PreserveAll_SRAMask;
|
||||
@@ -1101,4 +1033,4 @@ void Arm64Emitter::Align16B() {
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -80,17 +80,17 @@ constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PRe
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters{};
|
||||
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters {};
|
||||
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters {};
|
||||
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters {};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters {};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
@@ -152,8 +152,7 @@ protected:
|
||||
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
|
||||
if (SupportsPreserveAllABI) {
|
||||
SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
SpillStaticRegs(TmpReg, FPRs);
|
||||
PushDynamicRegsAndLR(TmpReg);
|
||||
}
|
||||
@@ -162,8 +161,7 @@ protected:
|
||||
void FillForABICall(bool SupportsPreserveAllABI, bool FPRs = true) {
|
||||
if (SupportsPreserveAllABI) {
|
||||
FillForPreserveAllABICall(FPRs);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
PopDynamicRegsAndLR();
|
||||
FillStaticRegs(FPRs);
|
||||
}
|
||||
@@ -185,8 +183,7 @@ protected:
|
||||
|
||||
template<typename R, typename... P>
|
||||
void GenerateRuntimeCall(R (*Function)(P...)) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
|
||||
uintptr_t FunctionAddress = reinterpret_cast<uintptr_t>(Function);
|
||||
|
||||
@@ -204,8 +201,7 @@ protected:
|
||||
|
||||
template<typename R, typename... P>
|
||||
void GenerateIndirectRuntimeCall(ARMEmitter::Register Reg) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(&(vixl::aarch64::Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
|
||||
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
|
||||
|
||||
@@ -221,8 +217,8 @@ protected:
|
||||
|
||||
template<>
|
||||
void GenerateIndirectRuntimeCall<float, __uint128_t>(ARMEmitter::Register Reg) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(vixl::aarch64::Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
|
||||
uintptr_t SimulatorWrapperAddress =
|
||||
reinterpret_cast<uintptr_t>(&(vixl::aarch64::Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
|
||||
|
||||
hlt(vixl::aarch64::kIndirectRuntimeCallOpcode);
|
||||
|
||||
@@ -262,4 +258,4 @@ protected:
|
||||
#endif
|
||||
};
|
||||
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -5,102 +5,102 @@
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::ARMEmitter {
|
||||
class Buffer {
|
||||
public:
|
||||
Buffer() {
|
||||
SetBuffer(nullptr, 0);
|
||||
}
|
||||
class Buffer {
|
||||
public:
|
||||
Buffer() {
|
||||
SetBuffer(nullptr, 0);
|
||||
}
|
||||
|
||||
Buffer(uint8_t* Base, uint64_t BaseSize) {
|
||||
SetBuffer(Base, BaseSize);
|
||||
}
|
||||
Buffer(uint8_t* Base, uint64_t BaseSize) {
|
||||
SetBuffer(Base, BaseSize);
|
||||
}
|
||||
|
||||
void SetBuffer(uint8_t* Base, uint64_t BaseSize) {
|
||||
BufferBase = Base;
|
||||
CurrentOffset = BufferBase;
|
||||
Size = BaseSize;
|
||||
}
|
||||
void SetBuffer(uint8_t* Base, uint64_t BaseSize) {
|
||||
BufferBase = Base;
|
||||
CurrentOffset = BufferBase;
|
||||
Size = BaseSize;
|
||||
}
|
||||
|
||||
void dc8(uint8_t Data) {
|
||||
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void dc8(uint8_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
|
||||
void dc16(uint16_t Data) {
|
||||
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void dc16(uint16_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
|
||||
void dc32(uint32_t Data) {
|
||||
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void dc32(uint32_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
|
||||
void dc64(uint64_t Data) {
|
||||
decltype(Data) *Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void EmitString(const char *String) {
|
||||
const auto StringLength = strlen(String);
|
||||
memcpy(CurrentOffset, String, StringLength);
|
||||
CurrentOffset += StringLength;
|
||||
}
|
||||
void dc64(uint64_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void EmitString(const char* String) {
|
||||
const auto StringLength = strlen(String);
|
||||
memcpy(CurrentOffset, String, StringLength);
|
||||
CurrentOffset += StringLength;
|
||||
}
|
||||
|
||||
void Align() {
|
||||
// Align the buffer to instruction size
|
||||
auto CurrentAlignment = reinterpret_cast<uint64_t>(CurrentOffset) & 0b11;
|
||||
if (!CurrentAlignment) {
|
||||
return;
|
||||
}
|
||||
CurrentOffset += 4 - CurrentAlignment;
|
||||
}
|
||||
void Align() {
|
||||
// Align the buffer to instruction size
|
||||
auto CurrentAlignment = reinterpret_cast<uint64_t>(CurrentOffset) & 0b11;
|
||||
if (!CurrentAlignment) {
|
||||
return;
|
||||
}
|
||||
CurrentOffset += 4 - CurrentAlignment;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T GetCursorAddress() const {
|
||||
return reinterpret_cast<T>(CurrentOffset);
|
||||
}
|
||||
template<typename T>
|
||||
T GetCursorAddress() const {
|
||||
return reinterpret_cast<T>(CurrentOffset);
|
||||
}
|
||||
|
||||
static void ClearICache(void* Begin, std::size_t Length) {
|
||||
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
|
||||
}
|
||||
static void ClearICache(void* Begin, std::size_t Length) {
|
||||
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
|
||||
}
|
||||
|
||||
size_t GetCursorOffset() const {
|
||||
return static_cast<size_t>(CurrentOffset - BufferBase);
|
||||
}
|
||||
size_t GetCursorOffset() const {
|
||||
return static_cast<size_t>(CurrentOffset - BufferBase);
|
||||
}
|
||||
|
||||
uint8_t *GetBufferBase() const {
|
||||
return BufferBase;
|
||||
}
|
||||
uint8_t* GetBufferBase() const {
|
||||
return BufferBase;
|
||||
}
|
||||
|
||||
void CursorIncrement(size_t Size) {
|
||||
CurrentOffset += Size;
|
||||
}
|
||||
void CursorIncrement(size_t Size) {
|
||||
CurrentOffset += Size;
|
||||
}
|
||||
|
||||
void SetCursorOffset(size_t Offset) {
|
||||
CurrentOffset = BufferBase + Offset;
|
||||
}
|
||||
void SetCursorOffset(size_t Offset) {
|
||||
CurrentOffset = BufferBase + Offset;
|
||||
}
|
||||
|
||||
uint64_t GetBufferSize() const {
|
||||
return Size;
|
||||
}
|
||||
uint64_t GetBufferSize() const {
|
||||
return Size;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
size_t GetCursorOffsetFromAddress(const T* Address) const {
|
||||
return static_cast<size_t>(reinterpret_cast<const uint8_t*>(Address) - BufferBase);
|
||||
}
|
||||
template<typename T>
|
||||
size_t GetCursorOffsetFromAddress(const T* Address) const {
|
||||
return static_cast<size_t>(reinterpret_cast<const uint8_t*>(Address) - BufferBase);
|
||||
}
|
||||
|
||||
protected:
|
||||
protected:
|
||||
|
||||
void ResetBuffer() {
|
||||
CurrentOffset = BufferBase;
|
||||
}
|
||||
void ResetBuffer() {
|
||||
CurrentOffset = BufferBase;
|
||||
}
|
||||
|
||||
uint8_t* BufferBase;
|
||||
uint8_t* CurrentOffset;
|
||||
uint64_t Size;
|
||||
};
|
||||
}
|
||||
uint8_t* BufferBase;
|
||||
uint8_t* CurrentOffset;
|
||||
uint64_t Size;
|
||||
};
|
||||
} // namespace FEXCore::ARMEmitter
|
||||
File diff suppressed because it is too large.
Load diff
@@ -3762,7 +3762,12 @@ public:
|
||||
}
|
||||
else {
|
||||
if (MemSrc.MetaType.ImmType.Index == ARMEmitter::IndexType::OFFSET) {
|
||||
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
|
||||
if ((MemSrc.MetaType.ImmType.Imm & 0b111) || MemSrc.MetaType.ImmType.Imm < 0) {
|
||||
prfum<IndexType::OFFSET>(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
|
||||
}
|
||||
else {
|
||||
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ImmType.Imm);
|
||||
}
|
||||
}
|
||||
else {
|
||||
LOGMAN_MSG_A_FMT("Unexpected loadstore index type");
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -6,48 +6,46 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore {
|
||||
void BlockSamplingData::DumpBlockData() {
|
||||
std::fstream Output;
|
||||
Output.open("output.csv", std::fstream::out | std::fstream::binary);
|
||||
void BlockSamplingData::DumpBlockData() {
|
||||
std::fstream Output;
|
||||
Output.open("output.csv", std::fstream::out | std::fstream::binary);
|
||||
|
||||
if (!Output.is_open())
|
||||
return;
|
||||
|
||||
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
|
||||
|
||||
for (auto it : SamplingMap) {
|
||||
if (!it.second->TotalCalls)
|
||||
continue;
|
||||
|
||||
Output << "0x" << std::hex << it.first
|
||||
<< ", " << std::dec << it.second->Min
|
||||
<< ", " << std::dec << it.second->Max
|
||||
<< ", " << std::dec << it.second->TotalTime
|
||||
<< ", " << std::dec << it.second->TotalCalls
|
||||
<< ", " << std::dec << ((double)it.second->TotalTime / (double)it.second->TotalCalls)
|
||||
<< std::endl;
|
||||
}
|
||||
Output.close();
|
||||
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
|
||||
if (!Output.is_open()) {
|
||||
return;
|
||||
}
|
||||
|
||||
BlockSamplingData::BlockData *BlockSamplingData::GetBlockData(uint64_t RIP) {
|
||||
auto it = SamplingMap.find(RIP);
|
||||
if (it != SamplingMap.end()) {
|
||||
return it->second;
|
||||
}
|
||||
BlockData *NewData = new BlockData{};
|
||||
memset(NewData, 0, sizeof(BlockData));
|
||||
NewData->Min = ~0ULL;
|
||||
SamplingMap[RIP] = NewData;
|
||||
return NewData;
|
||||
}
|
||||
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
|
||||
|
||||
BlockSamplingData::~BlockSamplingData() {
|
||||
DumpBlockData();
|
||||
for (auto it : SamplingMap) {
|
||||
delete it.second;
|
||||
for (auto it : SamplingMap) {
|
||||
if (!it.second->TotalCalls) {
|
||||
continue;
|
||||
}
|
||||
SamplingMap.clear();
|
||||
|
||||
Output << "0x" << std::hex << it.first << ", " << std::dec << it.second->Min << ", " << std::dec << it.second->Max << ", " << std::dec
|
||||
<< it.second->TotalTime << ", " << std::dec << it.second->TotalCalls << ", " << std::dec
|
||||
<< ((double)it.second->TotalTime / (double)it.second->TotalCalls) << std::endl;
|
||||
}
|
||||
Output.close();
|
||||
LogMan::Msg::DFmt("Dumped {} blocks of sampling data", SamplingMap.size());
|
||||
}
|
||||
|
||||
BlockSamplingData::BlockData* BlockSamplingData::GetBlockData(uint64_t RIP) {
|
||||
auto it = SamplingMap.find(RIP);
|
||||
if (it != SamplingMap.end()) {
|
||||
return it->second;
|
||||
}
|
||||
BlockData* NewData = new BlockData {};
|
||||
memset(NewData, 0, sizeof(BlockData));
|
||||
NewData->Min = ~0ULL;
|
||||
SamplingMap[RIP] = NewData;
|
||||
return NewData;
|
||||
}
|
||||
|
||||
BlockSamplingData::~BlockSamplingData() {
|
||||
DumpBlockData();
|
||||
for (auto it : SamplingMap) {
|
||||
delete it.second;
|
||||
}
|
||||
SamplingMap.clear();
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -14,7 +14,7 @@ public:
|
||||
uint64_t TotalCalls;
|
||||
};
|
||||
|
||||
BlockData *GetBlockData(uint64_t RIP);
|
||||
BlockData* GetBlockData(uint64_t RIP);
|
||||
~BlockSamplingData();
|
||||
|
||||
void DumpBlockData();
|
||||
@@ -22,4 +22,4 @@ public:
|
||||
private:
|
||||
std::unordered_map<uint64_t, BlockData*> SamplingMap;
|
||||
};
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -2,8 +2,8 @@
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <sys/prctl.h>
|
||||
@@ -12,347 +12,326 @@
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
|
||||
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
|
||||
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
|
||||
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT
|
||||
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
|
||||
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
|
||||
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
|
||||
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
|
||||
{0x0706'0504'0302'0100ULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0111B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1001B
|
||||
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
|
||||
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
// Expectation for this LUT is to simulate PSHUFLW with ARM's TBL (single register) instruction
|
||||
// PSHUFLW behaviour:
|
||||
// 16-bit words in [63:48], [47:32], [31:16], [15:0] are selected using the 8-bit Index.
|
||||
// For 128-bit PSHUFLW, bits [127:64] are identity copied.
|
||||
constexpr uint64_t IdentityCopyUpper = 0x0f'0e'0d'0c'0b'0a'09'08;
|
||||
std::array<LUTType, 256> TotalLUT{};
|
||||
uint64_t WordSelection[4] = {
|
||||
0x01'00,
|
||||
0x03'02,
|
||||
0x05'04,
|
||||
0x07'06,
|
||||
};
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
const auto Word0 = (i >> 0) & 0b11;
|
||||
const auto Word1 = (i >> 2) & 0b11;
|
||||
const auto Word2 = (i >> 4) & 0b11;
|
||||
const auto Word3 = (i >> 6) & 0b11;
|
||||
|
||||
LUT.Val[0] =
|
||||
(WordSelection[Word0] << 0) |
|
||||
(WordSelection[Word1] << 16) |
|
||||
(WordSelection[Word2] << 32) |
|
||||
(WordSelection[Word3] << 48);
|
||||
|
||||
LUT.Val[1] = IdentityCopyUpper;
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFHW_LUT {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
// Expectation for this LUT is to simulate PSHUFHW with ARM's TBL (single register) instruction
|
||||
// PSHUFHW behaviour:
|
||||
// 16-bit words in [127:112], [111:96], [95:80], [79:64] are selected using the 8-bit Index.
|
||||
// Incoming words come from bits [127:64] of the source.
|
||||
// Bits [63:0] are identity copied.
|
||||
constexpr uint64_t IdentityCopyLower = 0x07'06'05'04'03'02'01'00;
|
||||
std::array<LUTType, 256> TotalLUT{};
|
||||
uint64_t WordSelection[4] = {
|
||||
0x09'08,
|
||||
0x0b'0a,
|
||||
0x0d'0c,
|
||||
0x0f'0e,
|
||||
};
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
const auto Word0 = (i >> 0) & 0b11;
|
||||
const auto Word1 = (i >> 2) & 0b11;
|
||||
const auto Word2 = (i >> 4) & 0b11;
|
||||
const auto Word3 = (i >> 6) & 0b11;
|
||||
|
||||
LUT.Val[0] = IdentityCopyLower;
|
||||
|
||||
LUT.Val[1] =
|
||||
(WordSelection[Word0] << 0) |
|
||||
(WordSelection[Word1] << 16) |
|
||||
(WordSelection[Word2] << 32) |
|
||||
(WordSelection[Word3] << 48);
|
||||
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFD_LUT {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
// Expectation for this LUT is to simulate PSHUFD with ARM's TBL (single register) instruction
|
||||
// PSHUFD behaviour:
|
||||
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
|
||||
std::array<LUTType, 256> TotalLUT{};
|
||||
uint64_t WordSelection[4] = {
|
||||
0x03'02'01'00,
|
||||
0x07'06'05'04,
|
||||
0x0b'0a'09'08,
|
||||
0x0f'0e'0d'0c,
|
||||
};
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
const auto Word0 = (i >> 0) & 0b11;
|
||||
const auto Word1 = (i >> 2) & 0b11;
|
||||
const auto Word2 = (i >> 4) & 0b11;
|
||||
const auto Word3 = (i >> 6) & 0b11;
|
||||
|
||||
LUT.Val[0] =
|
||||
(WordSelection[Word0] << 0) |
|
||||
(WordSelection[Word1] << 32);
|
||||
|
||||
LUT.Val[1] =
|
||||
(WordSelection[Word2] << 0) |
|
||||
(WordSelection[Word3] << 32);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto SHUFPS_LUT {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
|
||||
// Expectation for this LUT is to simulate SHUFPS with ARM's TBL (two register) instruction.
|
||||
// SHUFPS behaviour:
|
||||
// Two 32-bits words from each source are selected from each source in the lower and upper halves of the 128-bit destination.
|
||||
// Dest[31:0] = Src1[<Word0>]
|
||||
// Dest[63:32] = Src1[<Word1>]
|
||||
// Dest[95:64] = Src2[<Word2>]
|
||||
// Dest[127:96] = Src2[<Word3>]
|
||||
|
||||
std::array<LUTType, 256> TotalLUT{};
|
||||
const uint64_t WordSelectionSrc1[4] = {
|
||||
0x03'02'01'00,
|
||||
0x07'06'05'04,
|
||||
0x0b'0a'09'08,
|
||||
0x0f'0e'0d'0c,
|
||||
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
|
||||
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
|
||||
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
|
||||
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT
|
||||
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
|
||||
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
|
||||
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
|
||||
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
|
||||
{0x0706'0504'0302'0100ULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0111B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1001B
|
||||
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
|
||||
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
|
||||
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB
|
||||
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB_UPPER
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_ONE
|
||||
{0xD49A'784B'CD1B'8AFEULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_LOG2_10
|
||||
{0xB8AA'3B29'5C17'F0BCULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_LOG2_E
|
||||
{0xC90F'DAA2'2168'C235ULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_PI
|
||||
{0x9A20'9A84'FBCF'F799ULL, 0x0000'0000'0000'3FFDULL}, // NAMED_VECTOR_X87_LOG10_2
|
||||
{0xB172'17F7'D1CF'79ACULL, 0x0000'0000'0000'3FFEULL}, // NAMED_VECTOR_X87_LOG_2
|
||||
};
|
||||
|
||||
// Src2 needs to offset each byte index by 16-bytes to pull from the second source.
|
||||
const uint64_t WordSelectionSrc2[4] = {
|
||||
0x03'02'01'00 + (0x10101010),
|
||||
0x07'06'05'04 + (0x10101010),
|
||||
0x0b'0a'09'08 + (0x10101010),
|
||||
0x0f'0e'0d'0c + (0x10101010),
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
const auto Word0 = (i >> 0) & 0b11;
|
||||
const auto Word1 = (i >> 2) & 0b11;
|
||||
const auto Word2 = (i >> 4) & 0b11;
|
||||
const auto Word3 = (i >> 6) & 0b11;
|
||||
|
||||
LUT.Val[0] =
|
||||
(WordSelectionSrc1[Word0] << 0) |
|
||||
(WordSelectionSrc1[Word1] << 32);
|
||||
|
||||
LUT.Val[1] =
|
||||
(WordSelectionSrc2[Word2] << 0) |
|
||||
(WordSelectionSrc2[Word3] << 32);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto DPPS_MASK {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint32_t Val[4];
|
||||
};
|
||||
|
||||
std::array<LUTType, 16> TotalLUT{};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1U;
|
||||
}
|
||||
return 0U;
|
||||
constexpr static auto PSHUFLW_LUT {[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
LUT.Val[2] = GetLUT(i, 2);
|
||||
LUT.Val[3] = GetLUT(i, 3);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto DPPD_MASK {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
|
||||
std::array<LUTType, 4> TotalLUT{};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1ULL;
|
||||
}
|
||||
return 0ULL;
|
||||
// Expectation for this LUT is to simulate PSHUFLW with ARM's TBL (single register) instruction
|
||||
// PSHUFLW behaviour:
|
||||
// 16-bit words in [63:48], [47:32], [31:16], [15:0] are selected using the 8-bit Index.
|
||||
// For 128-bit PSHUFLW, bits [127:64] are identity copied.
|
||||
constexpr uint64_t IdentityCopyUpper = 0x0f'0e'0d'0c'0b'0a'09'08;
|
||||
std::array<LUTType, 256> TotalLUT {};
|
||||
uint64_t WordSelection[4] = {
|
||||
0x01'00,
|
||||
0x03'02,
|
||||
0x05'04,
|
||||
0x07'06,
|
||||
};
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto& LUT = TotalLUT[i];
|
||||
const auto Word0 = (i >> 0) & 0b11;
|
||||
const auto Word1 = (i >> 2) & 0b11;
|
||||
const auto Word2 = (i >> 4) & 0b11;
|
||||
const auto Word3 = (i >> 6) & 0b11;
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
LUT.Val[0] = (WordSelection[Word0] << 0) | (WordSelection[Word1] << 16) | (WordSelection[Word2] << 32) | (WordSelection[Word3] << 48);
|
||||
|
||||
constexpr static auto PBLENDW_LUT {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint16_t Val[8];
|
||||
};
|
||||
// 16-bit words in [127:112], [111:96], [95:80], [79:64], [63:48], [47:32], [31:16], [15:0] are selected using 8-bit swizzle.
|
||||
// Expectation for this LUT is to simulate PBLENDW with ARM's TBX (one register) instruction.
|
||||
// PBLENDW behaviour:
|
||||
// 16-bit words from the source is moved in to the destination based on the bit in the swizzle.
|
||||
// Dest[15:0] = Swizzle[0] ? Src[15:0] : Dest[15:0]
|
||||
// Dest[31:16] = Swizzle[1] ? Src[31:16] : Dest[31:16]
|
||||
// Dest[47:32] = Swizzle[2] ? Src[47:32] : Dest[47:32]
|
||||
// Dest[63:48] = Swizzle[3] ? Src[63:48] : Dest[63:48]
|
||||
// Dest[79:64] = Swizzle[4] ? Src[79:64] : Dest[79:64]
|
||||
// Dest[95:80] = Swizzle[5] ? Src[95:80] : Dest[95:80]
|
||||
// Dest[111:96] = Swizzle[6] ? Src[111:96] : Dest[111:96]
|
||||
// Dest[127:112] = Swizzle[7] ? Src[127:112] : Dest[127:112]
|
||||
|
||||
std::array<LUTType, 256> TotalLUT{};
|
||||
const uint16_t WordSelectionSrc[8] = {
|
||||
0x01'00,
|
||||
0x03'02,
|
||||
0x05'04,
|
||||
0x07'06,
|
||||
0x09'08,
|
||||
0x0B'0A,
|
||||
0x0D'0C,
|
||||
0x0F'0E,
|
||||
};
|
||||
|
||||
constexpr uint16_t OriginalDest = 0xFF'FF;
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
for (size_t j = 0; j < 8; ++j) {
|
||||
LUT.Val[j] = ((i >> j) & 1) ? WordSelectionSrc[j] : OriginalDest;
|
||||
LUT.Val[1] = IdentityCopyUpper;
|
||||
}
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {
|
||||
constexpr static auto PSHUFHW_LUT {[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
// Expectation for this LUT is to simulate PSHUFHW with ARM's TBL (single register) instruction
|
||||
// PSHUFHW behaviour:
|
||||
// 16-bit words in [127:112], [111:96], [95:80], [79:64] are selected using the 8-bit Index.
|
||||
// Incoming words come from bits [127:64] of the source.
|
||||
// Bits [63:0] are identity copied.
|
||||
constexpr uint64_t IdentityCopyLower = 0x07'06'05'04'03'02'01'00;
|
||||
std::array<LUTType, 256> TotalLUT {};
|
||||
uint64_t WordSelection[4] = {
|
||||
0x09'08,
|
||||
0x0b'0a,
|
||||
0x0d'0c,
|
||||
0x0f'0e,
|
||||
};
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto& LUT = TotalLUT[i];
|
||||
const auto Word0 = (i >> 0) & 0b11;
|
||||
const auto Word1 = (i >> 2) & 0b11;
|
||||
const auto Word2 = (i >> 4) & 0b11;
|
||||
const auto Word3 = (i >> 6) & 0b11;
|
||||
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
LUT.Val[0] = IdentityCopyLower;
|
||||
|
||||
// Initialize named vector constants.
|
||||
for (size_t i = 0; i < FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX; ++i) {
|
||||
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
|
||||
}
|
||||
LUT.Val[1] = (WordSelection[Word0] << 0) | (WordSelection[Word1] << 16) | (WordSelection[Word2] << 32) | (WordSelection[Word3] << 48);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
// Copy named vector constants.
|
||||
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
|
||||
constexpr static auto PSHUFD_LUT {[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
// Expectation for this LUT is to simulate PSHUFD with ARM's TBL (single register) instruction
|
||||
// PSHUFD behaviour:
|
||||
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
|
||||
std::array<LUTType, 256> TotalLUT {};
|
||||
uint64_t WordSelection[4] = {
|
||||
0x03'02'01'00,
|
||||
0x07'06'05'04,
|
||||
0x0b'0a'09'08,
|
||||
0x0f'0e'0d'0c,
|
||||
};
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto& LUT = TotalLUT[i];
|
||||
const auto Word0 = (i >> 0) & 0b11;
|
||||
const auto Word1 = (i >> 2) & 0b11;
|
||||
const auto Word2 = (i >> 4) & 0b11;
|
||||
const auto Word3 = (i >> 6) & 0b11;
|
||||
|
||||
// Initialize Indexed named vector constants.
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] = reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] = reinterpret_cast<uint64_t>(DPPS_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] = reinterpret_cast<uint64_t>(DPPD_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] = reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
|
||||
LUT.Val[0] = (WordSelection[Word0] << 0) | (WordSelection[Word1] << 32);
|
||||
|
||||
LUT.Val[1] = (WordSelection[Word2] << 0) | (WordSelection[Word3] << 32);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
constexpr static auto SHUFPS_LUT {[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
|
||||
// Expectation for this LUT is to simulate SHUFPS with ARM's TBL (two register) instruction.
|
||||
// SHUFPS behaviour:
|
||||
// Two 32-bits words from each source are selected from each source in the lower and upper halves of the 128-bit destination.
|
||||
// Dest[31:0] = Src1[<Word0>]
|
||||
// Dest[63:32] = Src1[<Word1>]
|
||||
// Dest[95:64] = Src2[<Word2>]
|
||||
// Dest[127:96] = Src2[<Word3>]
|
||||
|
||||
std::array<LUTType, 256> TotalLUT {};
|
||||
const uint64_t WordSelectionSrc1[4] = {
|
||||
0x03'02'01'00,
|
||||
0x07'06'05'04,
|
||||
0x0b'0a'09'08,
|
||||
0x0f'0e'0d'0c,
|
||||
};
|
||||
|
||||
// Src2 needs to offset each byte index by 16-bytes to pull from the second source.
|
||||
const uint64_t WordSelectionSrc2[4] = {
|
||||
0x03'02'01'00 + (0x10101010),
|
||||
0x07'06'05'04 + (0x10101010),
|
||||
0x0b'0a'09'08 + (0x10101010),
|
||||
0x0f'0e'0d'0c + (0x10101010),
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto& LUT = TotalLUT[i];
|
||||
const auto Word0 = (i >> 0) & 0b11;
|
||||
const auto Word1 = (i >> 2) & 0b11;
|
||||
const auto Word2 = (i >> 4) & 0b11;
|
||||
const auto Word3 = (i >> 6) & 0b11;
|
||||
|
||||
LUT.Val[0] = (WordSelectionSrc1[Word0] << 0) | (WordSelectionSrc1[Word1] << 32);
|
||||
|
||||
LUT.Val[1] = (WordSelectionSrc2[Word2] << 0) | (WordSelectionSrc2[Word3] << 32);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
constexpr static auto DPPS_MASK {[]() consteval {
|
||||
struct LUTType {
|
||||
uint32_t Val[4];
|
||||
};
|
||||
|
||||
std::array<LUTType, 16> TotalLUT {};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto& LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1U;
|
||||
}
|
||||
return 0U;
|
||||
};
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
LUT.Val[2] = GetLUT(i, 2);
|
||||
LUT.Val[3] = GetLUT(i, 3);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
constexpr static auto DPPD_MASK {[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
|
||||
std::array<LUTType, 4> TotalLUT {};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto& LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1ULL;
|
||||
}
|
||||
return 0ULL;
|
||||
};
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
constexpr static auto PBLENDW_LUT {[]() consteval {
|
||||
struct LUTType {
|
||||
uint16_t Val[8];
|
||||
};
|
||||
// 16-bit words in [127:112], [111:96], [95:80], [79:64], [63:48], [47:32], [31:16], [15:0] are selected using 8-bit swizzle.
|
||||
// Expectation for this LUT is to simulate PBLENDW with ARM's TBX (one register) instruction.
|
||||
// PBLENDW behaviour:
|
||||
// 16-bit words from the source is moved in to the destination based on the bit in the swizzle.
|
||||
// Dest[15:0] = Swizzle[0] ? Src[15:0] : Dest[15:0]
|
||||
// Dest[31:16] = Swizzle[1] ? Src[31:16] : Dest[31:16]
|
||||
// Dest[47:32] = Swizzle[2] ? Src[47:32] : Dest[47:32]
|
||||
// Dest[63:48] = Swizzle[3] ? Src[63:48] : Dest[63:48]
|
||||
// Dest[79:64] = Swizzle[4] ? Src[79:64] : Dest[79:64]
|
||||
// Dest[95:80] = Swizzle[5] ? Src[95:80] : Dest[95:80]
|
||||
// Dest[111:96] = Swizzle[6] ? Src[111:96] : Dest[111:96]
|
||||
// Dest[127:112] = Swizzle[7] ? Src[127:112] : Dest[127:112]
|
||||
|
||||
std::array<LUTType, 256> TotalLUT {};
|
||||
const uint16_t WordSelectionSrc[8] = {
|
||||
0x01'00, 0x03'02, 0x05'04, 0x07'06, 0x09'08, 0x0B'0A, 0x0D'0C, 0x0F'0E,
|
||||
};
|
||||
|
||||
constexpr uint16_t OriginalDest = 0xFF'FF;
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto& LUT = TotalLUT[i];
|
||||
for (size_t j = 0; j < 8; ++j) {
|
||||
LUT.Val[j] = ((i >> j) & 1) ? WordSelectionSrc[j] : OriginalDest;
|
||||
}
|
||||
}
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState)
|
||||
, InitialCodeSize(InitialCodeSize)
|
||||
, MaxCodeSize(MaxCodeSize) {
|
||||
|
||||
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
// Initialize named vector constants.
|
||||
for (size_t i = 0; i < FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX; ++i) {
|
||||
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
|
||||
}
|
||||
|
||||
// Copy named vector constants.
|
||||
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
|
||||
|
||||
// Initialize Indexed named vector constants.
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] =
|
||||
reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] =
|
||||
reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] =
|
||||
reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] =
|
||||
reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] =
|
||||
reinterpret_cast<uint64_t>(DPPS_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] =
|
||||
reinterpret_cast<uint64_t>(DPPD_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] =
|
||||
reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
|
||||
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
// Fill in telemetry values
|
||||
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
|
||||
auto &Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
|
||||
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(Telem.GetAddr());
|
||||
}
|
||||
// Fill in telemetry values
|
||||
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
|
||||
auto& Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
|
||||
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(Telem.GetAddr());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
CPUBackend::~CPUBackend() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
}
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
|
||||
if (CodeBuffers.empty()) {
|
||||
CPUBackend::~CPUBackend() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
}
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
|
||||
if (CodeBuffers.empty()) {
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
} else {
|
||||
if (CodeBuffers.size() > 1) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (size_t i = 1; i < CodeBuffers.size(); i++) {
|
||||
FreeCodeBuffer(CodeBuffers[i]);
|
||||
}
|
||||
CodeBuffers.resize(1);
|
||||
}
|
||||
// Set the current code buffer to the initial
|
||||
CurrentCodeBuffer = &CodeBuffers[0];
|
||||
|
||||
if (CurrentCodeBuffer->Size != MaxCodeSize) {
|
||||
FreeCodeBuffer(*CurrentCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
|
||||
|
||||
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
} else {
|
||||
if (CodeBuffers.size() > 1) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (size_t i = 1; i < CodeBuffers.size(); i++) {
|
||||
FreeCodeBuffer(CodeBuffers[i]);
|
||||
}
|
||||
CodeBuffers.resize(1);
|
||||
}
|
||||
// Set the current code buffer to the initial
|
||||
CurrentCodeBuffer = &CodeBuffers[0];
|
||||
|
||||
if (CurrentCodeBuffer->Size != MaxCodeSize) {
|
||||
FreeCodeBuffer(*CurrentCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
|
||||
|
||||
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
|
||||
return CurrentCodeBuffer;
|
||||
}
|
||||
|
||||
return CurrentCodeBuffer;
|
||||
}
|
||||
|
||||
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
#ifndef _WIN32
|
||||
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
|
||||
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
|
||||
@@ -372,40 +351,39 @@ auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
#ifndef PR_GET_MDWE
|
||||
#define PR_GET_MDWE 66
|
||||
#endif
|
||||
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
|
||||
if (MDWE != -1 && MDWE != 0) {
|
||||
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
|
||||
}
|
||||
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
|
||||
if (MDWE != -1 && MDWE != 0) {
|
||||
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
|
||||
}
|
||||
#endif
|
||||
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t *>(
|
||||
FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
|
||||
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
for (auto &Buffer: CodeBuffers) {
|
||||
auto start = (uintptr_t)Buffer.Ptr;
|
||||
auto end = start + Buffer.Size;
|
||||
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
|
||||
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
for (auto& Buffer : CodeBuffers) {
|
||||
auto start = (uintptr_t)Buffer.Ptr;
|
||||
auto end = start + Buffer.Size;
|
||||
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
} // namespace CPU
|
||||
} // namespace FEXCore
|
||||
+25
-19
@@ -20,14 +20,14 @@ namespace FEXCore {
|
||||
namespace IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
}
|
||||
} // namespace IR
|
||||
|
||||
namespace Core {
|
||||
struct DebugData;
|
||||
struct ThreadState;
|
||||
struct CpuStateFrame;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
} // namespace Core
|
||||
|
||||
namespace CodeSerialize {
|
||||
struct CodeObjectFileSection;
|
||||
@@ -43,21 +43,22 @@ namespace CPU {
|
||||
class CPUBackend {
|
||||
public:
|
||||
struct CodeBuffer {
|
||||
uint8_t *Ptr;
|
||||
uint8_t* Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
/**
|
||||
* @param InitialCodeSize - Initial size for the code buffers
|
||||
* @param MaxCodeSize - Max size for the code buffers
|
||||
*/
|
||||
CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize);
|
||||
*/
|
||||
CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize);
|
||||
|
||||
virtual ~CPUBackend();
|
||||
/**
|
||||
* @return The name of this backend
|
||||
*/
|
||||
[[nodiscard]] virtual fextl::string GetName() = 0;
|
||||
[[nodiscard]]
|
||||
virtual fextl::string GetName() = 0;
|
||||
|
||||
struct CompiledCode {
|
||||
// Where this code block begins.
|
||||
@@ -137,10 +138,9 @@ namespace CPU {
|
||||
*
|
||||
* @return Information about the compiled code block.
|
||||
*/
|
||||
[[nodiscard]] virtual CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) = 0;
|
||||
[[nodiscard]]
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
FEXCore::IR::RegisterAllocationData* RAData) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
@@ -150,14 +150,18 @@ namespace CPU {
|
||||
*
|
||||
* @return An executable function pointer relocated from the cache object
|
||||
*/
|
||||
[[nodiscard]] virtual void *RelocateJITObjectCode(uint64_t Entry, CodeSerialize::CodeObjectFileSection const *SerializationData) { return nullptr; }
|
||||
[[nodiscard]]
|
||||
virtual void* RelocateJITObjectCode(uint64_t Entry, const CodeSerialize::CodeObjectFileSection* SerializationData) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Function for mapping memory in to the CPUBackend's visible space. Allows setting up virtual mappings if required
|
||||
*
|
||||
* @return Currently unused
|
||||
*/
|
||||
[[nodiscard]] virtual void *MapRegion(void *HostPtr, uint64_t GuestPtr, uint64_t Size) = 0;
|
||||
[[nodiscard]]
|
||||
virtual void* MapRegion(void* HostPtr, uint64_t GuestPtr, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief Lets FEXCore know if this CPUBackend needs IR and DebugData for CompileCode
|
||||
@@ -168,7 +172,8 @@ namespace CPU {
|
||||
*
|
||||
* @return true if it needs the IR
|
||||
*/
|
||||
[[nodiscard]] virtual bool NeedsOpDispatch() = 0;
|
||||
[[nodiscard]]
|
||||
virtual bool NeedsOpDispatch() = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
@@ -184,13 +189,14 @@ namespace CPU {
|
||||
// to be able to handle a 256-bit vector store to a slot.
|
||||
constexpr static uint32_t MaxSpillSlotSize = 32;
|
||||
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
FEXCore::Core::InternalThreadState* ThreadState;
|
||||
|
||||
size_t InitialCodeSize, MaxCodeSize;
|
||||
[[nodiscard]] CodeBuffer *GetEmptyCodeBuffer();
|
||||
[[nodiscard]]
|
||||
CodeBuffer* GetEmptyCodeBuffer();
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer *CurrentCodeBuffer{};
|
||||
CodeBuffer* CurrentCodeBuffer {};
|
||||
|
||||
private:
|
||||
CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
@@ -202,8 +208,8 @@ namespace CPU {
|
||||
|
||||
// This is the array of code buffers. Unless signals force us to keep more than
|
||||
// buffer, there will be only one entry here
|
||||
fextl::vector<CodeBuffer> CodeBuffers{};
|
||||
fextl::vector<CodeBuffer> CodeBuffers {};
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
} // namespace CPU
|
||||
} // namespace FEXCore
|
||||
File diff suppressed because it is too large.
Load diff
@@ -30,7 +30,7 @@ private:
|
||||
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
|
||||
|
||||
public:
|
||||
CPUIDEmu(FEXCore::Context::ContextImpl const *ctx);
|
||||
CPUIDEmu(const FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
// X86 cacheline size effectively has to be hardcoded to 64
|
||||
// if we report anything differently then applications are likely to break
|
||||
@@ -58,12 +58,13 @@ public:
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) const {
|
||||
if (Function == 0x8000'0002U)
|
||||
if (Function == 0x8000'0002U) {
|
||||
return Function_8000_0002h(Leaf, CPU % PerCPUData.size());
|
||||
else if (Function == 0x8000'0003U)
|
||||
} else if (Function == 0x8000'0003U) {
|
||||
return Function_8000_0003h(Leaf, CPU % PerCPUData.size());
|
||||
else
|
||||
} else {
|
||||
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) const {
|
||||
@@ -113,11 +114,12 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl const *CTX;
|
||||
bool Hybrid{};
|
||||
uint32_t Cores{};
|
||||
const FEXCore::Context::ContextImpl* CTX;
|
||||
bool Hybrid {};
|
||||
uint32_t Cores {};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
// Mask that configures what features are enabled on the CPU.
|
||||
@@ -148,10 +150,7 @@ private:
|
||||
.SHA = 1,
|
||||
};
|
||||
|
||||
uint64_t XCR0 {
|
||||
XCR0_X87 |
|
||||
XCR0_SSE
|
||||
};
|
||||
uint64_t XCR0 {XCR0_X87 | XCR0_SSE};
|
||||
|
||||
uint32_t SupportsAVX() const {
|
||||
return (XCR0 & XCR0_AVX) ? 1 : 0;
|
||||
@@ -160,13 +159,13 @@ private:
|
||||
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf) const;
|
||||
|
||||
struct CPUData {
|
||||
const char *ProductName{};
|
||||
const char* ProductName {};
|
||||
#ifdef _M_ARM_64
|
||||
uint32_t MIDR{};
|
||||
uint32_t MIDR {};
|
||||
#endif
|
||||
bool IsBig{};
|
||||
bool IsBig {};
|
||||
};
|
||||
fextl::vector<CPUData> PerCPUData{};
|
||||
fextl::vector<CPUData> PerCPUData {};
|
||||
|
||||
// Functions
|
||||
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf) const;
|
||||
@@ -277,74 +276,74 @@ private:
|
||||
|
||||
static constexpr std::array<FunctionConstant, PRIMARY_FUNCTION_COUNT> Primary_Constant = {{
|
||||
// 0: Highest function parameter and ID
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 1: Processor info
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 2: Cache and TLB info
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 3: Serial Number(previously), now reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#ifndef CPUID_AMD
|
||||
// 4: Deterministic cache parameters for each level
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
|
||||
#else
|
||||
// 4: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#endif
|
||||
// 5: Monitor/mwait
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 6: Thermal and power management
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 7: Extended feature flags
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
|
||||
// 0x08: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 9: Direct Cache Access information
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x0A: Architectural performance monitoring
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x0B: Extended topology enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x0C: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x0D: Processor extended state enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
|
||||
// 0x0E: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x0F: Intel RDT monitoring
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x10: Intel RDT allocation enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x12: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x12: Intel SGX capability enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x13: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x14: Intel Processor trace
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#ifndef CPUID_AMD
|
||||
// 0x15: Timestamp counter information
|
||||
// Doesn't exist on AMD hardware
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#else
|
||||
// 0x15: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#endif
|
||||
// 0x16: Processor frequency information
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x17: SoC vendor attribute enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x18: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x19: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#ifndef CPUID_AMD
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#else
|
||||
// 0x1A: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#endif
|
||||
}};
|
||||
|
||||
@@ -357,9 +356,9 @@ private:
|
||||
|
||||
static constexpr std::array<FunctionConstant, HYPERVISOR_FUNCTION_COUNT> Hypervisor_Constant = {{
|
||||
// Hypervisor CPUID information leaf
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// FEX-Emu specific leaf
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
|
||||
}};
|
||||
|
||||
static constexpr std::array<FunctionHandler, EXTENDED_FUNCTION_COUNT> Extended = {
|
||||
@@ -439,79 +438,79 @@ private:
|
||||
|
||||
static constexpr std::array<FunctionConstant, EXTENDED_FUNCTION_COUNT> Extended_Constant = {{
|
||||
// Largest extended function number
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// Processor vendor
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// Processor brand string
|
||||
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// Processor brand string continued
|
||||
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// Processor brand string continued
|
||||
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#ifdef CPUID_AMD
|
||||
// 0x8000'0005: L1 Cache and TLB identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#else
|
||||
// 0x8000'0005: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#endif
|
||||
// 0x8000'0006: L2 Cache identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0007: Advanced power management information
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0008: Virtual and physical address sizes
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0009: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'000A: SVM Revision
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'000B: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'000C: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'000D: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'000E: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'000F: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0010: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0011: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0012: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0013: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0014: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0015: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0016: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0017: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0018: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'0019: TLB 1GB page identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'001A: Performance optimization identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'001B: Instruction based sampling identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'001C: Lightweight profiling capabilities
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#ifdef CPUID_AMD
|
||||
// 0x8000'001D: Cache properties
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
|
||||
#else
|
||||
// 0x8000'001D: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#endif
|
||||
// 0x8000'001E: Extended APIC ID
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
}};
|
||||
};
|
||||
}
|
||||
} // namespace FEXCore
|
||||
File diff suppressed because it is too large.
Load diff
@@ -1,6 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
}
|
||||
@@ -25,13 +25,13 @@
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
static void SleepThread(FEXCore::Context::ContextImpl *CTX, FEXCore::Core::CpuStateFrame *Frame) {
|
||||
static void SleepThread(FEXCore::Context::ContextImpl* CTX, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
CTX->SyscallHandler->SleepThread(CTX, Frame);
|
||||
}
|
||||
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
|
||||
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx)
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl* ctx)
|
||||
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
|
||||
, CTX {ctx} {
|
||||
EmitDispatcher();
|
||||
@@ -61,6 +61,9 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ARMEmitter::ForwardLabel l_CTX;
|
||||
ARMEmitter::SingleUseForwardLabel l_Sleep;
|
||||
#ifdef _M_ARM_64EC
|
||||
ARMEmitter::SingleUseForwardLabel ExitEC;
|
||||
#endif
|
||||
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
|
||||
|
||||
// Push all the register we need to save
|
||||
@@ -82,16 +85,17 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// We want to ensure that we are 16 byte aligned at the top of this loop
|
||||
Align16B();
|
||||
ARMEmitter::BiDirectionalLabel FullLookup{};
|
||||
ARMEmitter::BiDirectionalLabel CallBlock{};
|
||||
ARMEmitter::BackwardLabel LoopTop{};
|
||||
ARMEmitter::BiDirectionalLabel FullLookup {};
|
||||
ARMEmitter::BiDirectionalLabel CallBlock {};
|
||||
ARMEmitter::BackwardLabel LoopTop {};
|
||||
|
||||
Bind(&LoopTop);
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
|
||||
|
||||
// IMPORTANT: Pointers.Common.ExitFunctionEC callsites/implementations need to be
|
||||
// adjusted accordingly if this changes.
|
||||
auto RipReg = TMP3;
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
@@ -99,7 +103,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL , 4);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP1, TMP1, 0);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
|
||||
@@ -117,8 +121,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
|
||||
}
|
||||
@@ -134,6 +137,10 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// If page pointer is zero then we have no block
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
#ifdef _M_ARM_64EC
|
||||
// The LSB of an L2 page entry indicates if this page contains EC code
|
||||
tbnz(TMP1, 0, &ExitEC);
|
||||
#endif
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
@@ -167,6 +174,15 @@ void Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
{
|
||||
Bind(&ExitEC);
|
||||
// Target PC is already loaded into TMP3 at the start of the dispatcher
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
}
|
||||
#endif
|
||||
|
||||
{
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
SpillStaticRegs(TMP1);
|
||||
@@ -193,9 +209,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void *, void *>(ARMEmitter::Reg::r2);
|
||||
}
|
||||
else {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
@@ -237,9 +252,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(ARMEmitter::XReg::x4, &l_CompileBlock);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void *, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
}
|
||||
else {
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP, MaxInst }
|
||||
}
|
||||
|
||||
@@ -285,7 +299,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
// Guest SIGTRAP handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
@@ -308,8 +322,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::r0, 0);
|
||||
PopCalleeSavedRegisters();
|
||||
ret();
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
}
|
||||
@@ -328,9 +341,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, &l_Sleep);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void *, void *>(ARMEmitter::Reg::r2);
|
||||
}
|
||||
else {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
@@ -405,8 +417,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(ARMEmitter::XReg::x3, R, Offset);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
// Result is now in x0
|
||||
@@ -463,23 +474,23 @@ void Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
void Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
|
||||
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(DispatchPtr));
|
||||
Simulator.RunFrom(reinterpret_cast< const vixl::aarch64::Instruction*>(DispatchPtr));
|
||||
}
|
||||
|
||||
void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
|
||||
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
|
||||
Simulator.WriteXRegister(1, RIP);
|
||||
Simulator.RunFrom(reinterpret_cast<vixl::aarch64::Instruction const*>(CallbackPtr));
|
||||
Simulator.RunFrom(reinterpret_cast< const vixl::aarch64::Instruction*>(CallbackPtr));
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
auto& Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
@@ -492,7 +503,7 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread)
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
|
||||
|
||||
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
|
||||
auto& AArch64 = Thread->CurrentFrame->Pointers.AArch64;
|
||||
AArch64.LUDIVHandler = LUDIVHandlerAddress;
|
||||
AArch64.LDIVHandler = LDIVHandlerAddress;
|
||||
AArch64.LUREMHandler = LUREMHandlerAddress;
|
||||
@@ -500,8 +511,8 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread)
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl *CTX) {
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl* CTX) {
|
||||
return fextl::make_unique<Dispatcher>(CTX);
|
||||
}
|
||||
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -2,8 +2,8 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -23,7 +23,7 @@ struct GuestSigAction;
|
||||
namespace FEXCore::Core {
|
||||
struct CpuStateFrame;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
@@ -31,51 +31,50 @@ class ContextImpl;
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
|
||||
|
||||
class Dispatcher final : public Arm64Emitter {
|
||||
public:
|
||||
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl *CTX);
|
||||
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl* CTX);
|
||||
|
||||
Dispatcher(FEXCore::Context::ContextImpl *ctx);
|
||||
Dispatcher(FEXCore::Context::ContextImpl* ctx);
|
||||
~Dispatcher();
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
uint64_t ThreadStopHandlerAddress{};
|
||||
uint64_t ThreadStopHandlerAddressSpillSRA{};
|
||||
uint64_t AbsoluteLoopTopAddress{};
|
||||
uint64_t AbsoluteLoopTopAddressFillSRA{};
|
||||
uint64_t ThreadPauseHandlerAddress{};
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA{};
|
||||
uint64_t ExitFunctionLinkerAddress{};
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t SignalHandlerReturnAddressRT{};
|
||||
uint64_t GuestSignal_SIGILL{};
|
||||
uint64_t GuestSignal_SIGTRAP{};
|
||||
uint64_t GuestSignal_SIGSEGV{};
|
||||
uint64_t IntCallbackReturnAddress{};
|
||||
uint64_t ThreadStopHandlerAddress {};
|
||||
uint64_t ThreadStopHandlerAddressSpillSRA {};
|
||||
uint64_t AbsoluteLoopTopAddress {};
|
||||
uint64_t AbsoluteLoopTopAddressFillSRA {};
|
||||
uint64_t ThreadPauseHandlerAddress {};
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA {};
|
||||
uint64_t ExitFunctionLinkerAddress {};
|
||||
uint64_t SignalHandlerReturnAddress {};
|
||||
uint64_t SignalHandlerReturnAddressRT {};
|
||||
uint64_t GuestSignal_SIGILL {};
|
||||
uint64_t GuestSignal_SIGTRAP {};
|
||||
uint64_t GuestSignal_SIGSEGV {};
|
||||
uint64_t IntCallbackReturnAddress {};
|
||||
|
||||
uint64_t PauseReturnInstruction{};
|
||||
uint64_t PauseReturnInstruction {};
|
||||
|
||||
/** @} */
|
||||
|
||||
uint64_t Start{};
|
||||
uint64_t End{};
|
||||
uint64_t Start {};
|
||||
uint64_t End {};
|
||||
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) ;
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
#else
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
DispatchPtr(Frame);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
#endif
|
||||
@@ -103,21 +102,21 @@ public:
|
||||
}
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
|
||||
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame);
|
||||
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
|
||||
AsmDispatch DispatchPtr;
|
||||
JITCallback CallbackPtr;
|
||||
private:
|
||||
// Long division helpers
|
||||
uint64_t LUDIVHandlerAddress{};
|
||||
uint64_t LDIVHandlerAddress{};
|
||||
uint64_t LUREMHandlerAddress{};
|
||||
uint64_t LREMHandlerAddress{};
|
||||
uint64_t LUDIVHandlerAddress {};
|
||||
uint64_t LDIVHandlerAddress {};
|
||||
uint64_t LUREMHandlerAddress {};
|
||||
uint64_t LREMHandlerAddress {};
|
||||
|
||||
void EmitDispatcher();
|
||||
};
|
||||
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
File diff suppressed because it is too large.
Load diff
@@ -21,10 +21,10 @@ class Decoder final {
|
||||
public:
|
||||
// New Frontend decoding
|
||||
struct DecodedBlocks final {
|
||||
uint64_t Entry{};
|
||||
uint64_t NumInstructions{};
|
||||
FEXCore::X86Tables::DecodedInst *DecodedInstructions;
|
||||
bool HasInvalidInstruction{};
|
||||
uint64_t Entry {};
|
||||
uint64_t NumInstructions {};
|
||||
FEXCore::X86Tables::DecodedInst* DecodedInstructions;
|
||||
bool HasInvalidInstruction {};
|
||||
};
|
||||
|
||||
struct DecodedBlockInformation final {
|
||||
@@ -32,19 +32,24 @@ public:
|
||||
fextl::vector<DecodedBlocks> Blocks;
|
||||
};
|
||||
|
||||
Decoder(FEXCore::Context::ContextImpl *ctx);
|
||||
Decoder(FEXCore::Context::ContextImpl* ctx);
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, uint64_t MaxInst, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
void DecodeInstructionsAtEntry(const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst,
|
||||
std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
|
||||
DecodedBlockInformation const *GetDecodedBlockInfo() const {
|
||||
const DecodedBlockInformation* GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
}
|
||||
|
||||
uint64_t DecodedMinAddress {};
|
||||
uint64_t DecodedMaxAddress {~0ULL};
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
|
||||
void SetExternalBranches(fextl::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
void SetSectionMaxAddress(uint64_t v) {
|
||||
SectionMaxAddress = v;
|
||||
}
|
||||
void SetExternalBranches(fextl::set<uint64_t>* v) {
|
||||
ExternalBranches = v;
|
||||
}
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
@@ -59,8 +64,8 @@ private:
|
||||
bool L; // VEX.L bit (if set then 256 bit operation, if unset then scalar or 128-bit operation)
|
||||
};
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
const FEXCore::HLE::SyscallOSABI OSABI{};
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
const FEXCore::HLE::SyscallOSABI OSABI {};
|
||||
|
||||
bool DecodeInstruction(uint64_t PC);
|
||||
|
||||
@@ -70,22 +75,24 @@ private:
|
||||
uint8_t ReadByte();
|
||||
uint8_t PeekByte(uint8_t Offset) const;
|
||||
uint64_t ReadData(uint8_t Size);
|
||||
void SkipBytes(uint8_t Size) { InstructionSize += Size; }
|
||||
void SkipBytes(uint8_t Size) {
|
||||
InstructionSize += Size;
|
||||
}
|
||||
|
||||
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op, DecodedHeader Options = {});
|
||||
bool NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
|
||||
bool NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
|
||||
bool NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
|
||||
|
||||
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
|
||||
FEXCore::X86Tables::DecodedInst *DecodedBuffer{};
|
||||
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
|
||||
uint8_t const *InstStream;
|
||||
const uint8_t* InstStream;
|
||||
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize;
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
FEXCore::X86Tables::DecodedInst *DecodeInst;
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
// This is for multiblock data tracking
|
||||
bool SymbolAvailable {false};
|
||||
@@ -99,21 +106,21 @@ private:
|
||||
DecodedBlockInformation BlockInfo;
|
||||
fextl::set<uint64_t> BlocksToDecode;
|
||||
fextl::set<uint64_t> HasBlocks;
|
||||
fextl::set<uint64_t> *ExternalBranches {nullptr};
|
||||
fextl::set<uint64_t>* ExternalBranches {nullptr};
|
||||
|
||||
// ModRM rm decoding
|
||||
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
void DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
void DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM);
|
||||
using DecodeModRMPtr = void (FEXCore::Frontend::Decoder::*)(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
|
||||
void DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
|
||||
void DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModRMDecoded ModRM);
|
||||
|
||||
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp{
|
||||
static constexpr std::array<DecodeModRMPtr, 2> DecodeModRMs_Disp {
|
||||
&FEXCore::Frontend::Decoder::DecodeModRM_64,
|
||||
&FEXCore::Frontend::Decoder::DecodeModRM_16,
|
||||
};
|
||||
|
||||
const uint8_t *AdjustAddrForSpecialRegion(uint8_t const* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
|
||||
FEXCORE_TELEMETRY_INIT(VEXOpTelem, TYPE_USES_VEX_OPS);
|
||||
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
|
||||
};
|
||||
}
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -28,23 +28,21 @@ namespace FEXCore {
|
||||
[[maybe_unused]] constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
[[maybe_unused]] static uint32_t GetDCZID() {
|
||||
uint64_t Result{};
|
||||
__asm("mrs %[Res], DCZID_EL0"
|
||||
: [Res] "=r" (Result));
|
||||
[[maybe_unused]]
|
||||
static uint32_t GetDCZID() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], DCZID_EL0" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static uint32_t GetFPCR() {
|
||||
uint64_t Result{};
|
||||
__asm ("mrs %[Res], FPCR"
|
||||
: [Res] "=r" (Result));
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], FPCR" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static void SetFPCR(uint64_t Value) {
|
||||
__asm ("msr FPCR, %[Value]"
|
||||
:: [Value] "r" (Value));
|
||||
__asm("msr FPCR, %[Value]" ::[Value] "r"(Value));
|
||||
}
|
||||
#else
|
||||
static uint32_t GetDCZID() {
|
||||
@@ -53,7 +51,7 @@ static uint32_t GetDCZID() {
|
||||
}
|
||||
#endif
|
||||
|
||||
static void OverrideFeatures(HostFeatures *Features) {
|
||||
static void OverrideFeatures(HostFeatures* Features) {
|
||||
// Override features if the user has specifically called for it.
|
||||
FEX_CONFIG_OPT(HostFeatures, HOSTFEATURES);
|
||||
if (!HostFeatures()) {
|
||||
@@ -62,19 +60,19 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
}
|
||||
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features->FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features->FeatureName = Result; \
|
||||
} while (0)
|
||||
do { \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features->FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features->FeatureName = Result; \
|
||||
} while (0)
|
||||
|
||||
#define GET_SINGLE_OPTION(name, enum_name) \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
|
||||
|
||||
ENABLE_DISABLE_OPTION(SupportsAVX, AVX, AVX);
|
||||
ENABLE_DISABLE_OPTION(SupportsAVX2, AVX2, AVX2);
|
||||
@@ -102,8 +100,7 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
Features->SupportsCRC = true;
|
||||
Features->SupportsSHA = true;
|
||||
Features->SupportsPMULL_128Bit = true;
|
||||
}
|
||||
else if (DisableCrypto) {
|
||||
} else if (DisableCrypto) {
|
||||
Features->SupportsAES = false;
|
||||
Features->SupportsCRC = false;
|
||||
Features->SupportsSHA = false;
|
||||
@@ -127,8 +124,7 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) &&
|
||||
Features.Has(vixl::CPUFeatures::Feature::kSHA2);
|
||||
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) && Features.Has(vixl::CPUFeatures::Feature::kSHA2);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
|
||||
|
||||
@@ -151,8 +147,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsAVX = true;
|
||||
#else
|
||||
SupportsSVE = Features.Has(vixl::CPUFeatures::Feature::kSVE);
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
|
||||
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) && vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
#endif
|
||||
// TODO: AVX2 is currently unsupported. Disable until the remaining features are implemented.
|
||||
SupportsAVX2 = false;
|
||||
@@ -173,21 +168,19 @@ HostFeatures::HostFeatures() {
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
__asm volatile ("mrs %[ctr], ctr_el0"
|
||||
: [ctr] "=r"(CTR));
|
||||
__asm volatile("mrs %[ctr], ctr_el0" : [ctr] "=r"(CTR));
|
||||
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps =
|
||||
(1U << 8) | // Invalid Operation float exception trap enable
|
||||
(1U << 9) | // Divide by zero float exception trap enable
|
||||
(1U << 10) | // Overflow float exception trap enable
|
||||
(1U << 11) | // Underflow float exception trap enable
|
||||
(1U << 12) | // Inexact float exception trap enable
|
||||
(1U << 15); // Input Denormal float exception trap enable
|
||||
constexpr uint32_t ExceptionEnableTraps = (1U << 8) | // Invalid Operation float exception trap enable
|
||||
(1U << 9) | // Divide by zero float exception trap enable
|
||||
(1U << 10) | // Overflow float exception trap enable
|
||||
(1U << 11) | // Underflow float exception trap enable
|
||||
(1U << 12) | // Inexact float exception trap enable
|
||||
(1U << 15); // Input Denormal float exception trap enable
|
||||
|
||||
uint32_t OriginalFPCR = GetFPCR();
|
||||
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
|
||||
@@ -222,7 +215,7 @@ HostFeatures::HostFeatures() {
|
||||
ICacheLineSize = 64U;
|
||||
|
||||
#if !defined(VIXL_SIMULATOR)
|
||||
Xbyak::util::Cpu X86Features{};
|
||||
Xbyak::util::Cpu X86Features {};
|
||||
SupportsAES = X86Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = X86Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsRAND = X86Features.has(Xbyak::util::Cpu::tRDRAND) && X86Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
@@ -256,4 +249,4 @@ HostFeatures::HostFeatures() {
|
||||
SupportsPreserveAllABI = FEXCORE_HAS_PRESERVE_ALL_ATTR;
|
||||
OverrideFeatures(this);
|
||||
}
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -2,48 +2,36 @@
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/SoftFloat-3e/softfloat.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static void LoadDeferredFCW(uint16_t NewFCW) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static void LoadDeferredFCW(uint16_t NewFCW) {
|
||||
auto PC = (NewFCW >> 8) & 3;
|
||||
switch(PC) {
|
||||
case 0: extF80_roundingPrecision = 32; break;
|
||||
case 2: extF80_roundingPrecision = 64; break;
|
||||
case 3: extF80_roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
switch (PC) {
|
||||
case 0: extF80_roundingPrecision = 32; break;
|
||||
case 2: extF80_roundingPrecision = 64; break;
|
||||
case 3: extF80_roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
}
|
||||
|
||||
auto RC = (NewFCW >> 10) & 3;
|
||||
switch(RC) {
|
||||
case 0:
|
||||
softfloat_roundingMode = softfloat_round_near_even;
|
||||
break;
|
||||
case 1:
|
||||
softfloat_roundingMode = softfloat_round_min;
|
||||
break;
|
||||
case 2:
|
||||
softfloat_roundingMode = softfloat_round_max;
|
||||
break;
|
||||
case 3:
|
||||
softfloat_roundingMode = softfloat_round_minMag;
|
||||
break;
|
||||
switch (RC) {
|
||||
case 0: softfloat_roundingMode = softfloat_round_near_even; break;
|
||||
case 1: softfloat_roundingMode = softfloat_round_min; break;
|
||||
case 2: softfloat_roundingMode = softfloat_round_max; break;
|
||||
case 3: softfloat_roundingMode = softfloat_round_minMag; break;
|
||||
}
|
||||
}
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle4(uint16_t NewFCW, float src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t NewFCW, float src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle8(uint16_t NewFCW, double src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle8(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
@@ -52,24 +40,20 @@ struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CMP> {
|
||||
template<uint32_t Flags>
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint64_t handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
|
||||
bool eq, lt, nan;
|
||||
uint64_t ResultFlags = 0;
|
||||
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
lt) {
|
||||
if (Flags & (1 << IR::FCMP_FLAG_LT) && lt) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
|
||||
nan) {
|
||||
if (Flags & (1 << IR::FCMP_FLAG_UNORDERED) && nan) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
|
||||
}
|
||||
if (Flags & (1 << IR::FCMP_FLAG_EQ) &&
|
||||
eq) {
|
||||
if (Flags & (1 << IR::FCMP_FLAG_EQ) && eq) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
|
||||
}
|
||||
return ResultFlags;
|
||||
@@ -78,14 +62,12 @@ struct OpHandlers<IR::OP_F80CMP> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static float handle4(uint16_t NewFCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static double handle8(uint16_t NewFCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
@@ -93,26 +75,22 @@ struct OpHandlers<IR::OP_F80CVT> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static int16_t handle2(uint16_t NewFCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static int32_t handle4(uint16_t NewFCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static int64_t handle8(uint16_t NewFCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static int16_t handle2t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
|
||||
@@ -125,14 +103,12 @@ struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static int32_t handle4t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static int64_t handle8t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return extF80_to_i64(src, softfloat_round_minMag, false);
|
||||
}
|
||||
@@ -140,14 +116,12 @@ struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle2(uint16_t NewFCW, int16_t src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle2(uint16_t NewFCW, int16_t src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle4(uint16_t NewFCW, int32_t src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t NewFCW, int32_t src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
@@ -155,8 +129,7 @@ struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FRNDINT(Src1);
|
||||
}
|
||||
@@ -164,8 +137,7 @@ struct OpHandlers<IR::OP_F80ROUND> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::F2XM1(Src1);
|
||||
}
|
||||
@@ -173,8 +145,7 @@ struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FTAN(Src1);
|
||||
}
|
||||
@@ -182,8 +153,7 @@ struct OpHandlers<IR::OP_F80TAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SQRT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSQRT(Src1);
|
||||
}
|
||||
@@ -191,8 +161,7 @@ struct OpHandlers<IR::OP_F80SQRT> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSIN(Src1);
|
||||
}
|
||||
@@ -200,8 +169,7 @@ struct OpHandlers<IR::OP_F80SIN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FCOS(Src1);
|
||||
}
|
||||
@@ -209,8 +177,7 @@ struct OpHandlers<IR::OP_F80COS> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FXTRACT_EXP(Src1);
|
||||
}
|
||||
@@ -218,8 +185,7 @@ struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FXTRACT_SIG(Src1);
|
||||
}
|
||||
@@ -227,8 +193,7 @@ struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ADD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FADD(Src1, Src2);
|
||||
}
|
||||
@@ -236,8 +201,7 @@ struct OpHandlers<IR::OP_F80ADD> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SUB> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSUB(Src1, Src2);
|
||||
}
|
||||
@@ -245,8 +209,7 @@ struct OpHandlers<IR::OP_F80SUB> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80MUL> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FMUL(Src1, Src2);
|
||||
}
|
||||
@@ -254,8 +217,7 @@ struct OpHandlers<IR::OP_F80MUL> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80DIV> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FDIV(Src1, Src2);
|
||||
}
|
||||
@@ -263,8 +225,7 @@ struct OpHandlers<IR::OP_F80DIV> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FYL2X(Src1, Src2);
|
||||
}
|
||||
@@ -272,8 +233,7 @@ struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FATAN(Src1, Src2);
|
||||
}
|
||||
@@ -281,8 +241,7 @@ struct OpHandlers<IR::OP_F80ATAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FREM1(Src1, Src2);
|
||||
}
|
||||
@@ -290,8 +249,7 @@ struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FREM(Src1, Src2);
|
||||
}
|
||||
@@ -299,8 +257,7 @@ struct OpHandlers<IR::OP_F80FPREM> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSCALE(Src1, Src2);
|
||||
}
|
||||
@@ -374,15 +331,14 @@ template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
double trunc = (double)(int64_t)(src2); //truncate
|
||||
double trunc = (double)(int64_t)(src2); // truncate
|
||||
return src1 * exp2(trunc);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
@@ -393,7 +349,7 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
|
||||
uint64_t Tmp = Src1;
|
||||
X80SoftFloat Rv;
|
||||
uint8_t *BCD = reinterpret_cast<uint8_t*>(&Rv);
|
||||
uint8_t* BCD = reinterpret_cast<uint8_t*>(&Rv);
|
||||
memset(BCD, 0, 10);
|
||||
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
@@ -423,11 +379,10 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
|
||||
uint64_t BCD{};
|
||||
uint8_t* Src1 = reinterpret_cast<uint8_t*>(&Src);
|
||||
uint64_t BCD {};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
// Only 0-9 is supported, A-F results in undefined data
|
||||
|
||||
@@ -68,8 +68,7 @@ namespace FEXCore::CPU {
|
||||
//
|
||||
// 5. Done.
|
||||
//
|
||||
template <IR::IROps Op>
|
||||
struct OpHandlers {
|
||||
};
|
||||
template<IR::IROps Op>
|
||||
struct OpHandlers {};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -10,23 +10,23 @@
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_UNKNOWN, (void*)fn, HandlerIndex, false};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64, (void*)fn, HandlerIndex, false};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_F64, (void*)fn, HandlerIndex, false};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
|
||||
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
|
||||
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
|
||||
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4);
|
||||
Info[Core::OPINDEX_F80CVT_8] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8);
|
||||
Info[Core::OPINDEX_F80CVTINT_2] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2);
|
||||
@@ -55,8 +55,8 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F80COS] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80COS>::handle);
|
||||
Info[Core::OPINDEX_F80XTRACT_EXP] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle);
|
||||
Info[Core::OPINDEX_F80XTRACT_SIG] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle);
|
||||
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle);
|
||||
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle);
|
||||
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle);
|
||||
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle);
|
||||
|
||||
// Binary
|
||||
Info[Core::OPINDEX_F80ADD] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80ADD>::handle);
|
||||
@@ -85,126 +85,123 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_VPCMPISTRX] = reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle);
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_Header const *IROp, FallbackInfo *Info) {
|
||||
bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::IROp_Header* IROp, FallbackInfo* Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
switch (IROp->Op) {
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVTINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2, SupportsPreserveAllABI};
|
||||
}
|
||||
else {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4, SupportsPreserveAllABI};
|
||||
}
|
||||
else {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8, SupportsPreserveAllABI};
|
||||
}
|
||||
else {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
static constexpr std::array handlers{
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags), SupportsPreserveAllABI};
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTOINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVTINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
static constexpr std::array handlers {
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>, &FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = {FABI_I64_I16_F80_F80, (void*)handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags),
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTOINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
}
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_UNARY_X87_OP(ROUND)
|
||||
@@ -242,20 +239,20 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_He
|
||||
COMMON_F64_OP(FPREM)
|
||||
COMMON_F64_OP(SCALE)
|
||||
|
||||
// SSE4.2 Fallbacks
|
||||
case IR::OP_VPCMPESTRX:
|
||||
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX, SupportsPreserveAllABI};
|
||||
return true;
|
||||
case IR::OP_VPCMPISTRX:
|
||||
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
|
||||
return true;
|
||||
// SSE4.2 Fallbacks
|
||||
case IR::OP_VPCMPESTRX:
|
||||
*Info = {FABI_I32_I64_I64_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle, Core::OPINDEX_VPCMPESTRX,
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
case IR::OP_VPCMPISTRX:
|
||||
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
|
||||
return true;
|
||||
|
||||
default:
|
||||
break;
|
||||
default: break;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -15,9 +15,9 @@ namespace FEXCore::CPU {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
enum class AggregationOp {
|
||||
EqualAny = 0b00,
|
||||
Ranges = 0b01,
|
||||
EqualEach = 0b10,
|
||||
EqualAny = 0b00,
|
||||
Ranges = 0b01,
|
||||
EqualEach = 0b10,
|
||||
EqualOrdered = 0b11,
|
||||
};
|
||||
|
||||
@@ -35,8 +35,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
NegativeMasked,
|
||||
};
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint32_t handle(uint64_t RAX, uint64_t RDX, __uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(uint64_t RAX, uint64_t RDX, __uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetExplicitLength(RAX, control) - 1;
|
||||
const auto valid_rhs = GetExplicitLength(RDX, control) - 1;
|
||||
@@ -45,8 +44,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
}
|
||||
|
||||
// Main PCMPXSTRX algorithm body. Allows for reuse with both implicit and explicit length variants.
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
|
||||
const uint32_t aggregation = PerformAggregation(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
const int32_t upper_limit = (16 >> (control & 1)) - 1;
|
||||
|
||||
@@ -70,8 +68,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
return result | (flags << 16);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static int32_t GetExplicitLength(uint64_t reg, uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetExplicitLength(uint64_t reg, uint16_t control) {
|
||||
// Bit 8 controls whether or not the reg value is 64-bit or 32-bit.
|
||||
int64_t value = 0;
|
||||
if (((control >> 8) & 1) != 0) {
|
||||
@@ -94,62 +91,50 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
return std::abs(static_cast<int>(value));
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static int32_t GetElement(const __uint128_t& vec, int32_t index, uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetElement(const __uint128_t& vec, int32_t index, uint16_t control) {
|
||||
const auto* vec_ptr = reinterpret_cast<const uint8_t*>(&vec);
|
||||
|
||||
// Control bits [1:0] define the data type being dealt with.
|
||||
switch (static_cast<SourceData>(control & 0b11)) {
|
||||
case SourceData::U8:
|
||||
return static_cast<int32_t>(vec_ptr[index]);
|
||||
case SourceData::U8: return static_cast<int32_t>(vec_ptr[index]);
|
||||
case SourceData::U16: {
|
||||
uint16_t value{};
|
||||
uint16_t value {};
|
||||
std::memcpy(&value, vec_ptr + (sizeof(uint16_t) * static_cast<size_t>(index)), sizeof(value));
|
||||
return value;
|
||||
}
|
||||
case SourceData::S8:
|
||||
return static_cast<int8_t>(vec_ptr[index]);
|
||||
case SourceData::S8: return static_cast<int8_t>(vec_ptr[index]);
|
||||
case SourceData::S16:
|
||||
default: {
|
||||
int16_t value{};
|
||||
int16_t value {};
|
||||
std::memcpy(&value, vec_ptr + (sizeof(int16_t) * static_cast<size_t>(index)), sizeof(value));
|
||||
return value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint32_t PerformAggregation(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
|
||||
PerformAggregation(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
|
||||
switch (static_cast<AggregationOp>((control >> 2) & 0b11)) {
|
||||
case AggregationOp::EqualAny:
|
||||
return HandleEqualAny(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::Ranges:
|
||||
return HandleRanges(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::EqualEach:
|
||||
return HandleEqualEach(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::EqualAny: return HandleEqualAny(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::Ranges: return HandleRanges(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::EqualEach: return HandleEqualEach(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
case AggregationOp::EqualOrdered:
|
||||
default:
|
||||
return HandleEqualOrdered(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
default: return HandleEqualOrdered(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint32_t HandlePolarity(uint32_t value, uint16_t control, int upper_limit, int valid_rhs) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t HandlePolarity(uint32_t value, uint16_t control, int upper_limit, int valid_rhs) {
|
||||
switch (static_cast<Polarity>((control >> 4) & 0b11)) {
|
||||
case Polarity::Negative:
|
||||
return value ^ ((2U << upper_limit) - 1);
|
||||
case Polarity::NegativeMasked:
|
||||
return value ^ ((1U << (valid_rhs + 1)) - 1);
|
||||
case Polarity::Positive:
|
||||
case Polarity::PositiveMasked:
|
||||
default:
|
||||
// Both positive masking and positive polarity are documented
|
||||
// as both being equivalent to "IntRes2 = IntRes1", where IntRes1
|
||||
// is our 'value' parameter, so we don't need to do anything in
|
||||
// these cases except return the same value.
|
||||
return value;
|
||||
case Polarity::Negative: return value ^ ((2U << upper_limit) - 1);
|
||||
case Polarity::NegativeMasked: return value ^ ((1U << (valid_rhs + 1)) - 1);
|
||||
case Polarity::Positive:
|
||||
case Polarity::PositiveMasked:
|
||||
default:
|
||||
// Both positive masking and positive polarity are documented
|
||||
// as both being equivalent to "IntRes2 = IntRes1", where IntRes1
|
||||
// is our 'value' parameter, so we don't need to do anything in
|
||||
// these cases except return the same value.
|
||||
return value;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -175,10 +160,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
// │
|
||||
// 'c' match ────────┘
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint32_t HandleEqualAny(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
|
||||
HandleEqualAny(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
|
||||
uint32_t result = 0;
|
||||
|
||||
for (int j = valid_rhs; j >= 0; j--) {
|
||||
@@ -222,10 +205,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
// │
|
||||
// 'Z' >= 'z' && 'A' <= 'z' ──────────┘
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint32_t HandleRanges(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
|
||||
HandleRanges(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
|
||||
uint32_t result = 0;
|
||||
|
||||
for (int j = valid_rhs; j >= 0; j--) {
|
||||
@@ -275,10 +256,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
// │
|
||||
// 'a' == 'a' ──────────┘
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint32_t HandleEqualEach(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
|
||||
HandleEqualEach(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
|
||||
const auto upper_limit = (16 >> (control & 1)) - 1;
|
||||
const auto max_valid = std::max(valid_lhs, valid_rhs);
|
||||
const auto min_valid = std::min(valid_lhs, valid_rhs);
|
||||
@@ -330,10 +309,8 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
// │
|
||||
// At index 0 ──────────┘
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint32_t HandleEqualOrdered(const __uint128_t& lhs, int32_t valid_lhs,
|
||||
const __uint128_t& rhs, int32_t valid_rhs,
|
||||
uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t
|
||||
HandleEqualOrdered(const __uint128_t& lhs, int32_t valid_lhs, const __uint128_t& rhs, int32_t valid_rhs, uint16_t control) {
|
||||
const auto upper_limit = (16 >> (control & 1)) - 1;
|
||||
|
||||
// Edge case!
|
||||
@@ -345,8 +322,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
}
|
||||
|
||||
uint32_t result = 0;
|
||||
const int initial = valid_rhs == upper_limit ? valid_rhs
|
||||
: valid_rhs - valid_lhs;
|
||||
const int initial = valid_rhs == upper_limit ? valid_rhs : valid_rhs - valid_lhs;
|
||||
for (int j = initial; j >= 0; j--) {
|
||||
result <<= 1;
|
||||
|
||||
@@ -379,8 +355,7 @@ struct OpHandlers<IR::OP_VPCMPISTRX> {
|
||||
// to be the max length possible for the given character size specified
|
||||
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
|
||||
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
|
||||
@@ -388,8 +363,7 @@ struct OpHandlers<IR::OP_VPCMPISTRX> {
|
||||
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR
|
||||
static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
|
||||
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
@@ -399,7 +373,7 @@ struct OpHandlers<IR::OP_VPCMPISTRX> {
|
||||
const auto get_word = [data_u8](int32_t index) {
|
||||
const auto* src = data_u8 + (index * sizeof(uint16_t));
|
||||
|
||||
uint16_t element{};
|
||||
uint16_t element {};
|
||||
std::memcpy(&element, src, sizeof(uint16_t));
|
||||
return element;
|
||||
};
|
||||
|
||||
@@ -7,44 +7,43 @@
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IRListView;
|
||||
struct IROp_Header;
|
||||
}
|
||||
class IRListView;
|
||||
struct IROp_Header;
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
enum FallbackABI {
|
||||
FABI_UNKNOWN,
|
||||
FABI_F80_I16_F32,
|
||||
FABI_F80_I16_F64,
|
||||
FABI_F80_I16_I16,
|
||||
FABI_F80_I16_I32,
|
||||
FABI_F32_I16_F80,
|
||||
FABI_F64_I16_F80,
|
||||
FABI_F64_I16_F64,
|
||||
FABI_F64_I16_F64_F64,
|
||||
FABI_I16_I16_F80,
|
||||
FABI_I32_I16_F80,
|
||||
FABI_I64_I16_F80,
|
||||
FABI_I64_I16_F80_F80,
|
||||
FABI_F80_I16_F80,
|
||||
FABI_F80_I16_F80_F80,
|
||||
FABI_I32_I64_I64_I128_I128_I16,
|
||||
FABI_I32_I128_I128_I16,
|
||||
};
|
||||
enum FallbackABI {
|
||||
FABI_UNKNOWN,
|
||||
FABI_F80_I16_F32,
|
||||
FABI_F80_I16_F64,
|
||||
FABI_F80_I16_I16,
|
||||
FABI_F80_I16_I32,
|
||||
FABI_F32_I16_F80,
|
||||
FABI_F64_I16_F80,
|
||||
FABI_F64_I16_F64,
|
||||
FABI_F64_I16_F64_F64,
|
||||
FABI_I16_I16_F80,
|
||||
FABI_I32_I16_F80,
|
||||
FABI_I64_I16_F80,
|
||||
FABI_I64_I16_F80_F80,
|
||||
FABI_F80_I16_F80,
|
||||
FABI_F80_I16_F80_F80,
|
||||
FABI_I32_I64_I64_I128_I128_I16,
|
||||
FABI_I32_I128_I128_I16,
|
||||
};
|
||||
|
||||
struct FallbackInfo {
|
||||
FallbackABI ABI;
|
||||
void *fn;
|
||||
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
|
||||
bool SupportsPreserveAllABI;
|
||||
};
|
||||
struct FallbackInfo {
|
||||
FallbackABI ABI;
|
||||
void* fn;
|
||||
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
|
||||
bool SupportsPreserveAllABI;
|
||||
};
|
||||
|
||||
class InterpreterOps {
|
||||
public:
|
||||
static void FillFallbackIndexPointers(uint64_t *Info);
|
||||
static bool GetFallbackHandler(bool SupportsPreserveAllABI, IR::IROp_Header const *IROp, FallbackInfo *Info);
|
||||
};
|
||||
class InterpreterOps {
|
||||
public:
|
||||
static void FillFallbackIndexPointers(uint64_t* Info);
|
||||
static bool GetFallbackHandler(bool SupportsPreserveAllABI, const IR::IROp_Header* IROp, FallbackInfo* Info);
|
||||
};
|
||||
} // namespace FEXCore::CPU
|
||||
File diff suppressed because it is too large.
Load diff
@@ -13,21 +13,19 @@ namespace FEXCore::CPU {
|
||||
|
||||
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op)); break;
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum &Sum) {
|
||||
Relocation MoveABI{};
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum) {
|
||||
Relocation MoveABI {};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
|
||||
@@ -43,22 +41,25 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Lit {
|
||||
.Lit = Pointer,
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
.MoveABI =
|
||||
{
|
||||
.NamedSymbolLiteral =
|
||||
{
|
||||
.Header =
|
||||
{
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
|
||||
Bind(&Lit.Loc);
|
||||
@@ -67,10 +68,10 @@ void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
Relocation MoveABI {};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
|
||||
@@ -79,54 +80,54 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
|
||||
const char* EntryRelocations) {
|
||||
size_t DataIndex {};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
dc64(Pointer);
|
||||
// Generate a literal so we can place it
|
||||
dc64(Pointer);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -11,7 +11,7 @@ $end_info$
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Wrong element size");
|
||||
@@ -29,13 +29,20 @@ DEF_OP(CASPair) {
|
||||
caspal(EmitSize, TMP3, TMP4, Desired.first, Desired.second, MemSrc);
|
||||
mov(EmitSize, Dst.first, TMP3.R());
|
||||
mov(EmitSize, Dst.second, TMP4.R());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
// Save NZCV so we don't have to mark this op as clobbering NZCV (the
|
||||
// SupportsAtomics does not clobber atomics and this !SupportsAtomics path
|
||||
// is so slow it's not worth the complexity of splitting the IR op.). We
|
||||
// clobber NZCV inside the hot loop and we can't replace cmp/ccmp/b.ne with
|
||||
// something NZCV-preserving without requiring an extra instruction.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
|
||||
// This instruction sequence must be synced with HandleCASPAL_Armv8.
|
||||
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
|
||||
cmp(EmitSize, TMP2, Expected.first);
|
||||
ccmp(EmitSize, TMP3, Expected.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
@@ -47,13 +54,16 @@ DEF_OP(CASPair) {
|
||||
|
||||
b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst.first, TMP2.R());
|
||||
mov(EmitSize, Dst.second, TMP3.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst.first, TMP2.R());
|
||||
mov(EmitSize, Dst.second, TMP3.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopExpected);
|
||||
|
||||
// Restore
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -71,16 +81,16 @@ DEF_OP(CAS) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(EmitSize, TMP2, Expected);
|
||||
casal(SubEmitSize, TMP2, Desired, MemSrc);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
@@ -88,11 +98,9 @@ DEF_OP(CAS) {
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (OpSize == 1) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
|
||||
}
|
||||
else if (OpSize == 2) {
|
||||
} else if (OpSize == 2) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTH, 0);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
cmp(EmitSize, TMP2, Expected);
|
||||
}
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
@@ -101,11 +109,11 @@ DEF_OP(CAS) {
|
||||
mov(EmitSize, GetReg(Node), Expected);
|
||||
b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopExpected);
|
||||
}
|
||||
}
|
||||
@@ -120,14 +128,14 @@ DEF_OP(AtomicAdd) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
staddl(SubEmitSize, Src, MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -147,15 +155,15 @@ DEF_OP(AtomicSub) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(EmitSize, TMP2, Src);
|
||||
staddl(SubEmitSize, TMP2, MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -175,15 +183,15 @@ DEF_OP(AtomicAnd) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(EmitSize, TMP2, Src);
|
||||
stclrl(SubEmitSize, TMP2, MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -203,14 +211,14 @@ DEF_OP(AtomicCLR) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stclrl(SubEmitSize, Src, MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -230,14 +238,14 @@ DEF_OP(AtomicOr) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stsetl(SubEmitSize, Src, MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -257,14 +265,14 @@ DEF_OP(AtomicXor) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
steorl(SubEmitSize, Src, MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -283,9 +291,10 @@ DEF_OP(AtomicNeg) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
@@ -305,14 +314,14 @@ DEF_OP(AtomicSwap) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -332,14 +341,14 @@ DEF_OP(AtomicFetchAdd) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -360,15 +369,15 @@ DEF_OP(AtomicFetchSub) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(EmitSize, TMP2, Src);
|
||||
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -389,15 +398,15 @@ DEF_OP(AtomicFetchAnd) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(EmitSize, TMP2, Src);
|
||||
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -418,14 +427,14 @@ DEF_OP(AtomicFetchCLR) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -446,14 +455,14 @@ DEF_OP(AtomicFetchOr) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -474,14 +483,14 @@ DEF_OP(AtomicFetchXor) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -501,9 +510,10 @@ DEF_OP(AtomicFetchNeg) {
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
@@ -527,8 +537,7 @@ DEF_OP(TelemetrySetValue) {
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stsetl(ARMEmitter::SubRegSize::i64Bit, TMP1, TMP2);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
|
||||
@@ -540,5 +549,4 @@ DEF_OP(TelemetrySetValue) {
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -19,7 +19,7 @@ $end_info$
|
||||
#include <Interface/HLE/Thunks/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
// spill back to CTX
|
||||
@@ -53,14 +53,23 @@ DEF_OP(ExitFunction) {
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
#ifdef _M_ARM_64EC
|
||||
if (RtlIsEcCode(NewRIP)) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, NewRIP);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
} else {
|
||||
#endif
|
||||
ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
ldr(TMP1, &l_BranchHost);
|
||||
blr(TMP1);
|
||||
|
||||
ldr(TMP1, &l_BranchHost);
|
||||
blr(TMP1);
|
||||
|
||||
Bind(&l_BranchHost);
|
||||
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
dc64(NewRIP);
|
||||
Bind(&l_BranchHost);
|
||||
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
dc64(NewRIP);
|
||||
#ifdef _M_ARM_64EC
|
||||
}
|
||||
#endif
|
||||
} else {
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel FullLookup;
|
||||
@@ -94,7 +103,7 @@ DEF_OP(Jump) {
|
||||
|
||||
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
@@ -106,17 +115,15 @@ static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -135,13 +142,12 @@ DEF_OP(CondJump) {
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
|
||||
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ ||
|
||||
Op->Cond.Val == FEXCore::IR::COND_NEQ,
|
||||
"CondJump: Expected simple condition");
|
||||
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ || Op->Cond.Val == FEXCore::IR::COND_NEQ, "CondJump: Expected simple "
|
||||
"condition");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else {
|
||||
} else {
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
}
|
||||
|
||||
@@ -181,7 +187,9 @@ DEF_OP(Syscall) {
|
||||
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) continue;
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
str(GetReg(Op->Header.Args[i].ID()).X(), ARMEmitter::Reg::rsp, i * 8);
|
||||
}
|
||||
|
||||
@@ -193,8 +201,7 @@ DEF_OP(Syscall) {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
@@ -232,20 +239,19 @@ DEF_OP(InlineSyscall) {
|
||||
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
|
||||
|
||||
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
|
||||
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS-1> RegArgs = {{
|
||||
ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5
|
||||
}};
|
||||
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS - 1> RegArgs = {
|
||||
{ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5}};
|
||||
|
||||
bool Intersects{};
|
||||
bool Intersects {};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
if (Reg == ARMEmitter::Reg::r8 ||
|
||||
Reg == ARMEmitter::Reg::r4 ||
|
||||
Reg == ARMEmitter::Reg::r5) {
|
||||
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
|
||||
|
||||
SpillMask |= (1U << Reg.Idx());
|
||||
Intersects = true;
|
||||
@@ -269,8 +275,10 @@ DEF_OP(InlineSyscall) {
|
||||
const auto EmitSize = CTX->Config.Is64BitMode() ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSubSize = CTX->Config.Is64BitMode() ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
|
||||
if (Intersects) {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
@@ -278,21 +286,19 @@ DEF_OP(InlineSyscall) {
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg == ARMEmitter::Reg::r8) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
}
|
||||
else if (Reg == ARMEmitter::Reg::r4) {
|
||||
} else if (Reg == ARMEmitter::Reg::r4) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX]));
|
||||
}
|
||||
else if (Reg == ARMEmitter::Reg::r5) {
|
||||
} else if (Reg == ARMEmitter::Reg::r5) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RCX]));
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
mov(EmitSize, RegArgs[i].R(), Reg);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
} else {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i].ID()));
|
||||
}
|
||||
@@ -333,8 +339,7 @@ DEF_OP(Thunk) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
@@ -345,7 +350,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
const auto *OldCode = (const uint8_t *)&Op->CodeOriginalLow;
|
||||
const auto* OldCode = (const uint8_t*)&Op->CodeOriginalLow;
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
@@ -355,37 +360,33 @@ DEF_OP(ValidateCode) {
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
while (len >= 8)
|
||||
{
|
||||
while (len >= 8) {
|
||||
ldr(ARMEmitter::XReg::x2, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t *)(OldCode + idx));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 8;
|
||||
idx += 8;
|
||||
}
|
||||
while (len >= 4)
|
||||
{
|
||||
while (len >= 4) {
|
||||
ldr(ARMEmitter::WReg::w2, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t *)(OldCode + idx));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 4;
|
||||
idx += 4;
|
||||
}
|
||||
while (len >= 2)
|
||||
{
|
||||
while (len >= 2) {
|
||||
ldrh(TMP3, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint16_t *)(OldCode + idx));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint16_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 2;
|
||||
idx += 2;
|
||||
}
|
||||
while (len >= 1)
|
||||
{
|
||||
while (len >= 1) {
|
||||
ldrb(TMP3, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint8_t *)(OldCode + idx));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint8_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 1;
|
||||
@@ -407,8 +408,7 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
FillStaticRegs();
|
||||
@@ -439,8 +439,7 @@ DEF_OP(CPUID) {
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
@@ -456,7 +455,7 @@ DEF_OP(CPUID) {
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.first, TMP1);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.first, TMP1);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, TMP2);
|
||||
}
|
||||
|
||||
@@ -474,8 +473,7 @@ DEF_OP(XGetBV) {
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
@@ -490,10 +488,9 @@ DEF_OP(XGetBV) {
|
||||
// Results are in x0
|
||||
// Results want to be in a i32v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, TMP1);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, TMP1);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, TMP1, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -9,7 +9,7 @@ $end_info$
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -20,9 +20,10 @@ DEF_OP(VInsGPR) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ElementsPer128Bit = 16 / ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -94,21 +95,17 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
uxtb(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 2:
|
||||
uxth(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 4:
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src);
|
||||
break;
|
||||
case 8:
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
|
||||
case 1:
|
||||
uxtb(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 2:
|
||||
uxth(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 4: fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src); break;
|
||||
case 8: fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -122,14 +119,14 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1,
|
||||
"Unexpected {} element size: {}", __func__, ElementSize);
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1, "Unexpected {} element size: {}",
|
||||
__func__, ElementSize);
|
||||
|
||||
const auto SubEmitSize =
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
dup(SubEmitSize, Dst.Z(), Src);
|
||||
@@ -148,34 +145,33 @@ DEF_OP(Float_FromGPR_S) {
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0204: { // Half <- int32_t
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.H(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0208: { // Half <- int64_t
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.H(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.S(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.S(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.D(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.D(), Src);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
|
||||
Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
case 0x0204: { // Half <- int32_t
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.H(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0208: { // Half <- int64_t
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.H(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.S(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.S(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.D(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.D(), Src);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}", Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -187,31 +183,31 @@ DEF_OP(Float_FToF) {
|
||||
auto Src = GetVReg(Op->Scalar.ID());
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvt(Dst.H(), Src.S());
|
||||
break;
|
||||
}
|
||||
case 0x0208: { // Half <- Double
|
||||
fcvt(Dst.H(), Src.D());
|
||||
break;
|
||||
}
|
||||
case 0x0402: { // Float <- Half
|
||||
fcvt(Dst.S(), Src.H());
|
||||
break;
|
||||
}
|
||||
case 0x0802: { // Double <- Half
|
||||
fcvt(Dst.D(), Src.H());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvt(Dst.D(), Src.S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvt(Dst.S(), Src.D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvt(Dst.H(), Src.S());
|
||||
break;
|
||||
}
|
||||
case 0x0208: { // Half <- Double
|
||||
fcvt(Dst.H(), Src.D());
|
||||
break;
|
||||
}
|
||||
case 0x0402: { // Float <- Half
|
||||
fcvt(Dst.S(), Src.H());
|
||||
break;
|
||||
}
|
||||
case 0x0802: { // Double <- Half
|
||||
fcvt(Dst.D(), Src.H());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvt(Dst.D(), Src.S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvt(Dst.S(), Src.D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,8 +220,9 @@ DEF_OP(Vector_SToF) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -236,19 +233,15 @@ DEF_OP(Vector_SToF) {
|
||||
if (OpSize == ElementSize) {
|
||||
if (ElementSize == 8) {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
|
||||
}
|
||||
else if (ElementSize == 4) {
|
||||
} else if (ElementSize == 4) {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
if (OpSize == 8) {
|
||||
scvtf(SubEmitSize, Dst.D(), Vector.D());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
scvtf(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
@@ -264,8 +257,9 @@ DEF_OP(Vector_FToZS) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -276,19 +270,15 @@ DEF_OP(Vector_FToZS) {
|
||||
if (OpSize == ElementSize) {
|
||||
if (ElementSize == 8) {
|
||||
fcvtzs(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
|
||||
}
|
||||
else if (ElementSize == 4) {
|
||||
} else if (ElementSize == 4) {
|
||||
fcvtzs(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
fcvtzs(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
if (OpSize == 8) {
|
||||
fcvtzs(SubEmitSize, Dst.D(), Vector.D());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
fcvtzs(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
@@ -304,8 +294,9 @@ DEF_OP(Vector_FToS) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -320,8 +311,7 @@ DEF_OP(Vector_FToS) {
|
||||
if (OpSize == 8) {
|
||||
frinti(SubEmitSize, Dst.D(), Vector.D());
|
||||
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
frinti(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
fcvtzs(SubEmitSize, Dst.Q(), Dst.Q());
|
||||
}
|
||||
@@ -338,8 +328,9 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -361,45 +352,41 @@ DEF_OP(Vector_FToF) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0402: { // Float <- Half
|
||||
zip1(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
zip1(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Mask, Vector.Z());
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Vector.Z());
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
case 0x0402: { // Float <- Half
|
||||
zip1(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
zip1(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Mask, Vector.Z());
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Vector.Z());
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
}
|
||||
} else {
|
||||
switch (Conv) {
|
||||
case 0x0402: // Float <- Half
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(SubEmitSize, Dst.D(), Vector.D());
|
||||
break;
|
||||
}
|
||||
case 0x0204: // Half <- Float
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(SubEmitSize, Dst.D(), Vector.D());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
case 0x0402: // Float <- Half
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(SubEmitSize, Dst.D(), Vector.D());
|
||||
break;
|
||||
}
|
||||
case 0x0204: // Half <- Float
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(SubEmitSize, Dst.D(), Vector.D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -413,8 +400,9 @@ DEF_OP(Vector_FToI) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -423,82 +411,51 @@ DEF_OP(Vector_FToI) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z());
|
||||
break;
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
} else {
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
|
||||
if (IsScalar) {
|
||||
// Since we have multiple overloads of the same name (e.g.
|
||||
// frinti having AdvSIMD, AdvSIMD scalar, and an SVE version),
|
||||
// we can't just use a lambda without some seriously ugly casting.
|
||||
// This is fairly self-contained otherwise.
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == 2) { \
|
||||
name(Dst.H(), Vector.H()); \
|
||||
} else if (ElementSize == 4) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == 8) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
}
|
||||
// Since we have multiple overloads of the same name (e.g.
|
||||
// frinti having AdvSIMD, AdvSIMD scalar, and an SVE version),
|
||||
// we can't just use a lambda without some seriously ugly casting.
|
||||
// This is fairly self-contained otherwise.
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == 2) { \
|
||||
name(Dst.H(), Vector.H()); \
|
||||
} else if (ElementSize == 4) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == 8) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
}
|
||||
|
||||
switch (Op->Round) {
|
||||
case IR::Round_Nearest.Val:
|
||||
ROUNDING_FN(frintn);
|
||||
break;
|
||||
case IR::Round_Negative_Infinity.Val:
|
||||
ROUNDING_FN(frintm);
|
||||
break;
|
||||
case IR::Round_Positive_Infinity.Val:
|
||||
ROUNDING_FN(frintp);
|
||||
break;
|
||||
case IR::Round_Towards_Zero.Val:
|
||||
ROUNDING_FN(frintz);
|
||||
break;
|
||||
case IR::Round_Host.Val:
|
||||
ROUNDING_FN(frinti);
|
||||
break;
|
||||
case IR::Round_Nearest.Val: ROUNDING_FN(frintn); break;
|
||||
case IR::Round_Negative_Infinity.Val: ROUNDING_FN(frintm); break;
|
||||
case IR::Round_Positive_Infinity.Val: ROUNDING_FN(frintp); break;
|
||||
case IR::Round_Towards_Zero.Val: ROUNDING_FN(frintz); break;
|
||||
case IR::Round_Host.Val: ROUNDING_FN(frinti); break;
|
||||
}
|
||||
|
||||
#undef ROUNDING_FN
|
||||
#undef ROUNDING_FN
|
||||
} else {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
frintn(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
frintm(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
frintp(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
frintz(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
frinti(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
break;
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -10,7 +10,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(VAESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
@@ -26,8 +26,7 @@ DEF_OP(VAESEnc) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -35,8 +34,7 @@ DEF_OP(VAESEnc) {
|
||||
aese(Dst.Q(), ZeroReg.Q());
|
||||
aesmc(Dst.Q(), Dst.Q());
|
||||
eor(Dst.Q(), Dst.Q(), Key.Q());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
mov(VTMP1.Q(), State.Q());
|
||||
aese(VTMP1, ZeroReg.Q());
|
||||
aesmc(VTMP1, VTMP1);
|
||||
@@ -53,16 +51,14 @@ DEF_OP(VAESEncLast) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
// This matches the common case of XMM AES.
|
||||
aese(Dst.Q(), ZeroReg.Q());
|
||||
eor(Dst.Q(), Dst.Q(), Key.Q());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
mov(VTMP1.Q(), State.Q());
|
||||
aese(VTMP1, ZeroReg.Q());
|
||||
eor(Dst.Q(), VTMP1.Q(), Key.Q());
|
||||
@@ -78,8 +74,7 @@ DEF_OP(VAESDec) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -87,8 +82,7 @@ DEF_OP(VAESDec) {
|
||||
aesd(Dst.Q(), ZeroReg.Q());
|
||||
aesimc(Dst.Q(), Dst.Q());
|
||||
eor(Dst.Q(), Dst.Q(), Key.Q());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
mov(VTMP1.Q(), State.Q());
|
||||
aesd(VTMP1, ZeroReg.Q());
|
||||
aesimc(VTMP1, VTMP1);
|
||||
@@ -105,16 +99,14 @@ DEF_OP(VAESDecLast) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
// This matches the common case of XMM AES.
|
||||
aesd(Dst.Q(), ZeroReg.Q());
|
||||
eor(Dst.Q(), Dst.Q(), Key.Q());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
mov(VTMP1.Q(), State.Q());
|
||||
aesd(VTMP1, ZeroReg.Q());
|
||||
eor(Dst.Q(), VTMP1.Q(), Key.Q());
|
||||
@@ -149,8 +141,7 @@ DEF_OP(VAESKeyGenAssist) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->RCON) << 32);
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP2.Q(), TMP1);
|
||||
eor(Dst.Q(), Dst.Q(), VTMP2.Q());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
tbl(Dst.Q(), Dst.Q(), Swizzle.Q());
|
||||
}
|
||||
}
|
||||
@@ -163,19 +154,11 @@ DEF_OP(CRC32) {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
crc32cb(Dst.W(), Src1.W(), Src2.W());
|
||||
break;
|
||||
case 2:
|
||||
crc32ch(Dst.W(), Src1.W(), Src2.W());
|
||||
break;
|
||||
case 4:
|
||||
crc32cw(Dst.W(), Src1.W(), Src2.W());
|
||||
break;
|
||||
case 8:
|
||||
crc32cx(Dst.X(), Src1.X(), Src2.X());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
case 1: crc32cb(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case 2: crc32ch(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case 4: crc32cw(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case 8: crc32cx(Dst.X(), Src1.X(), Src2.X()); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -197,11 +180,10 @@ DEF_OP(VSha256U0) {
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256su0(Dst, Src2);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256su0(VTMP1, Src2);
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -209,17 +191,14 @@ DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
"Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D());
|
||||
break;
|
||||
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
|
||||
case 0b00000001:
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
|
||||
@@ -228,14 +207,10 @@ DEF_OP(PCLMUL) {
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
|
||||
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
|
||||
break;
|
||||
case 0b00010001:
|
||||
pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
case 0b00010001: pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q()); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector); break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -8,12 +8,11 @@ $end_info$
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Value.ID()), Op->Flag, 1);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -73,11 +73,11 @@ static void PrintValue(uint64_t Value) {
|
||||
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
|
||||
}
|
||||
}
|
||||
} // namespace
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(CTX->HostFeatures.SupportsPreserveAllABI, IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
@@ -118,379 +118,347 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
mov(Dst.W(), TMP1.W());
|
||||
};
|
||||
|
||||
switch(Info.ABI) {
|
||||
case FABI_F80_I16_F32:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
switch (Info.ABI) {
|
||||
case FABI_F80_I16_F32: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
fmov(ARMEmitter::SReg::s0, Src1.S());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, float>(ARMEmitter::Reg::r1);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
fmov(ARMEmitter::SReg::s0, Src1.S());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, float>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_I16_F64:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
case FABI_F80_I16_F64: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
FillF80Result();
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_I16_I16:
|
||||
case FABI_F80_I16_I32: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16_I16) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint32_t>(ARMEmitter::Reg::r2);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
case FABI_F80_I16_I16:
|
||||
case FABI_F80_I16_I32: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
FillF80Result();
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16_I16) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
} else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F32_I16_F80:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<float, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VTMP1.S(), ARMEmitter::SReg::s0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
fmov(Dst.S(), VTMP1.S());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint32_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_I16_F80:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
case FABI_F32_I16_F80: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
FillF64Result();
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<float, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_I16_F64: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VTMP1.S(), ARMEmitter::SReg::s0);
|
||||
}
|
||||
break;
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
case FABI_F64_I16_F64_F64: {
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
const auto Dst = GetVReg(Node);
|
||||
fmov(Dst.S(), VTMP1.S());
|
||||
} break;
|
||||
|
||||
mov(VTMP1.D(), Src1.D());
|
||||
mov(VTMP2.D(), Src2.D());
|
||||
case FABI_F64_I16_F80: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::DReg::d0, VTMP1.D());
|
||||
mov(ARMEmitter::DReg::d1, VTMP2.D());
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double>(ARMEmitter::Reg::r1);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_I16_F80:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
case FABI_F64_I16_F64: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
sxth(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I16_F80:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
case FABI_F64_I16_F64_F64: {
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
mov(VTMP1.D(), Src1.D());
|
||||
mov(VTMP2.D(), Src2.D());
|
||||
|
||||
FillI32Result();
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::DReg::d0, VTMP1.D());
|
||||
mov(ARMEmitter::DReg::d1, VTMP2.D());
|
||||
}
|
||||
break;
|
||||
case FABI_I64_I16_F80:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_I16_F80_F80:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
case FABI_I16_I16_F80: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_I16_F80:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_I16_F80_F80:{
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
const auto Dst = GetReg(Node);
|
||||
sxth(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_I32_I16_F80: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I64_I64_I128_I128_I16: {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
mov(TMP1, SrcRAX.X());
|
||||
mov(TMP2, SrcRDX.X());
|
||||
FillI32Result();
|
||||
} break;
|
||||
case FABI_I64_I16_F80: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP3, true);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
const auto Control = Op->Control;
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x0, TMP1);
|
||||
mov(ARMEmitter::XReg::x1, TMP2);
|
||||
}
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r4, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r5, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r6, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x7, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r7);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r7);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Control = Op->Control;
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r4, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r5);
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
break;
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_I64_I16_F80_F80: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
}
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_F80_I16_F80: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
case FABI_F80_I16_F80_F80: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
case FABI_I32_I64_I64_I128_I128_I16: {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
mov(TMP1, SrcRAX.X());
|
||||
mov(TMP2, SrcRDX.X());
|
||||
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP3, true);
|
||||
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::XReg::x0, TMP1);
|
||||
mov(ARMEmitter::XReg::x1, TMP2);
|
||||
}
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r4, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r5, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r6, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x7, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r7);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r7);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
} break;
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Control = Op->Control;
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r4, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r5);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
} break;
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}",
|
||||
FEXCore::IR::GetName(IROp->Op), ToUnderlying(Info.ABI));
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}", FEXCore::IR::GetName(IROp->Op), ToUnderlying(Info.ABI));
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
@@ -498,9 +466,9 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record) {
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
uintptr_t branch = (uintptr_t)(Record) - 8;
|
||||
uintptr_t branch = (uintptr_t)(Record)-8;
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 8);
|
||||
FEXCore::ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
emit.ldr(TMP1, &l_BranchHost);
|
||||
@@ -510,12 +478,12 @@ static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Co
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 8);
|
||||
}
|
||||
|
||||
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record) {
|
||||
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
Record->HostBranch = LinkerAddress;
|
||||
}
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, FEXCore::Context::ExitFunctionLinkData *Record) {
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = Record->GuestRIP;
|
||||
|
||||
@@ -526,9 +494,9 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(Record) - 8;
|
||||
uintptr_t branch = (uintptr_t)(Record)-8;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
auto offset = HostCode / 4 - branch / 4;
|
||||
if (vixl::IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
@@ -549,16 +517,15 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::NodeID Node) {}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128{ctx->HostFeatures.SupportsSVE}
|
||||
, HostSupportsSVE256{ctx->HostFeatures.SupportsAVX}
|
||||
, HostSupportsRPRES{ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP{ctx->HostFeatures.SupportsAFP}
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsAVX}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP}
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
@@ -573,7 +540,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
RAPass->AddRegisters(FEXCore::IR::ComplexClass, 1);
|
||||
|
||||
for (uint32_t i = 0; i < GeneralPairRegisters.size(); ++i) {
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2, FEXCore::IR::GPRPairClass, i);
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2, FEXCore::IR::GPRPairClass, i);
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2 + 1, FEXCore::IR::GPRPairClass, i);
|
||||
}
|
||||
|
||||
@@ -581,7 +548,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
// Set up pointers that the JIT needs to load
|
||||
|
||||
// Common
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
@@ -609,7 +576,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
|
||||
|
||||
// Platform Specific
|
||||
auto &AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
auto& AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
@@ -624,8 +591,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
if (ParanoidTSO()) {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
|
||||
}
|
||||
@@ -645,9 +611,7 @@ void Arm64JITCore::ClearCache() {
|
||||
EmitDetectionString();
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {
|
||||
|
||||
}
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
|
||||
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
@@ -704,10 +668,8 @@ bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
return Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
@@ -727,9 +689,9 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Put the code header at the start of the data block.
|
||||
ARMEmitter::BackwardLabel JITCodeHeaderLabel{};
|
||||
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
|
||||
Bind(&JITCodeHeaderLabel);
|
||||
JITCodeHeader *CodeHeader = GetCursorAddress<JITCodeHeader *>();
|
||||
JITCodeHeader* CodeHeader = GetCursorAddress<JITCodeHeader*>();
|
||||
CursorIncrement(sizeof(JITCodeHeader));
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
@@ -766,11 +728,11 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) -
|
||||
offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
// LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
@@ -794,14 +756,13 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t *>();
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t*>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
|
||||
// if there's a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second)
|
||||
{
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second) {
|
||||
b(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
@@ -812,43 +773,40 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
#define REGISTER_OP(op, x) case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
#define REGISTER_OP_RT(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
#define REGISTER_OP(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
|
||||
#define IROP_DISPATCH_DISPATCH
|
||||
#include <FEXCore/IR/IRDefines_Dispatch.inc>
|
||||
#undef REGISTER_OP
|
||||
|
||||
default:
|
||||
Op_Unhandled(IROp, ID);
|
||||
break;
|
||||
default: Op_Unhandled(IROp, ID); break;
|
||||
}
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockEntry),
|
||||
static_cast<uint32_t>(GetCursorAddress<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
DebugData->Subblocks.push_back({static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockEntry),
|
||||
static_cast<uint32_t>(GetCursorAddress<uint8_t*>() - BlockStartHostCode)});
|
||||
}
|
||||
}
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
if (PendingTargetLabel)
|
||||
{
|
||||
if (PendingTargetLabel) {
|
||||
b(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
// CodeSize not including the tail data.
|
||||
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
// Add the JitCodeTail
|
||||
auto JITBlockTailLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITBlockTailLocation = GetCursorAddress<uint8_t*>();
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t*>();
|
||||
auto JITRIPEntries = GetCursorAddress<JITRIPReconstructEntries*>();
|
||||
|
||||
CursorIncrement(sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
@@ -867,8 +825,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
const auto& GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto& RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
@@ -878,7 +836,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
@@ -933,7 +891,7 @@ void Arm64JITCore::ResetStack() {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread) {
|
||||
return fextl::make_unique<Arm64JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
@@ -945,4 +903,4 @@ CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
};
|
||||
}
|
||||
|
||||
}
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -9,16 +9,17 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
@@ -29,48 +30,60 @@ $end_info$
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
public:
|
||||
explicit Arm64JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
explicit Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]] fextl::string GetName() override { return "JIT"; }
|
||||
[[nodiscard]]
|
||||
fextl::string GetName() override {
|
||||
return "JIT";
|
||||
}
|
||||
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
FEXCore::IR::RegisterAllocationData* RAData) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
[[nodiscard]]
|
||||
void* MapRegion(void* HostPtr, uint64_t, uint64_t) override {
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
[[nodiscard]]
|
||||
bool NeedsOpDispatch() override {
|
||||
return true;
|
||||
}
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
void ClearRelocations() override {
|
||||
Relocations.clear();
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
|
||||
const bool HostSupportsSVE128{};
|
||||
const bool HostSupportsSVE256{};
|
||||
const bool HostSupportsRPRES{};
|
||||
const bool HostSupportsAFP{};
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
const bool HostSupportsRPRES {};
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
const FEXCore::IR::IRListView* IR;
|
||||
uint64_t Entry;
|
||||
CPUBackend::CompiledCode CodeData{};
|
||||
CPUBackend::CompiledCode CodeData {};
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
@@ -84,7 +97,8 @@ private:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
@@ -98,7 +112,8 @@ private:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]] std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
|
||||
[[nodiscard]]
|
||||
std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
@@ -106,9 +121,11 @@ private:
|
||||
return GeneralPairRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]] IR::PhysicalRegister GetPhys(IR::NodeID Node) const {
|
||||
[[nodiscard]]
|
||||
IR::PhysicalRegister GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
@@ -116,7 +133,8 @@ private:
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Src, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
|
||||
@@ -128,36 +146,38 @@ private:
|
||||
|
||||
// Converts IR-base shift type to ARMEmitter shift type.
|
||||
// Will be a no-op, only a type conversion since the two definitions match.
|
||||
[[nodiscard]] ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
|
||||
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
|
||||
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
|
||||
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
|
||||
ARMEmitter::ShiftType::ROR;
|
||||
ARMEmitter::ShiftType::ROR;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize,
|
||||
FEXCore::ARMEmitter::Register Base,
|
||||
IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(
|
||||
uint8_t AccessSize, FEXCore::ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
// the limits of the scalar plus immediate variant of SVE load/stores.
|
||||
//
|
||||
// TMP1 is safe to use again once this memory operand is used with its
|
||||
// equivalent loads or stores that this was called for.
|
||||
[[nodiscard]] FEXCore::ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize,
|
||||
FEXCore::ARMEmitter::Register Base,
|
||||
IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize, FEXCore::ARMEmitter::Register Base,
|
||||
IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
[[nodiscard]]
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
[[nodiscard]]
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
|
||||
struct LiveRange {
|
||||
uint32_t Begin;
|
||||
@@ -166,81 +186,83 @@ private:
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
FEXCore::Core::DebugData *DebugData;
|
||||
IR::RegisterAllocationPass* RAPass;
|
||||
IR::RegisterAllocationData* RAData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
|
||||
void ResetStack();
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
ARMEmitter::ForwardLabel Loc;
|
||||
uint64_t Lit;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
ARMEmitter::ForwardLabel Loc;
|
||||
uint64_t Lit;
|
||||
Relocation MoveABI {};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum &Sum);
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
uint32_t SpillSlots {};
|
||||
using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::NodeID Node);
|
||||
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
|
||||
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
// Dynamic Dispatcher supporting operations
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -17,7 +17,7 @@ $end_info$
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
@@ -28,16 +28,10 @@ DEF_OP(GuestOpcode) {
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
case IR::Fence_Load.Val:
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::LD);
|
||||
break;
|
||||
case IR::Fence_LoadStore.Val:
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::SY);
|
||||
break;
|
||||
case IR::Fence_Store.Val:
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ST);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
case IR::Fence_Load.Val: dmb(FEXCore::ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::Fence_LoadStore.Val: dmb(FEXCore::ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::Fence_Store.Val: dmb(FEXCore::ARMEmitter::BarrierScope::ST); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -55,7 +49,7 @@ DEF_OP(Break) {
|
||||
.err_code = Op->Reason.ErrorRegister,
|
||||
};
|
||||
|
||||
uint64_t Constant{};
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant);
|
||||
@@ -136,8 +130,7 @@ DEF_OP(Print) {
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue));
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value.ID()), false);
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value.ID()), true);
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
|
||||
@@ -146,12 +139,10 @@ DEF_OP(Print) {
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
@@ -231,8 +222,7 @@ DEF_OP(RDRAND) {
|
||||
|
||||
if (Op->GetReseeded) {
|
||||
mrs(Dst.first, ARMEmitter::SystemRegister::RNDRRS);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
mrs(Dst.first, ARMEmitter::SystemRegister::RNDR);
|
||||
}
|
||||
|
||||
@@ -245,5 +235,4 @@ DEF_OP(Yield) {
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -8,7 +8,7 @@ $end_info$
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
LOGMAN_THROW_AA_FMT(Op->Header.Size == 4 || Op->Header.Size == 8, "Invalid size");
|
||||
@@ -43,5 +43,4 @@ DEF_OP(CreateElementPair) {
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
File diff suppressed because it is too large.
Load diff
@@ -1,7 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
@@ -15,8 +15,8 @@ struct InternalThreadState;
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]] fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
[[nodiscard]]
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -13,8 +13,8 @@ $end_info$
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
namespace FEXCore {
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl *CTX)
|
||||
: BlockLinks_mbr { fextl::pmr::get_default_resource() }
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
: BlockLinks_mbr {fextl::pmr::get_default_resource()}
|
||||
, ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
@@ -78,5 +78,4 @@ void LookupCache::ClearCache() {
|
||||
BlockList.clear();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -13,6 +13,9 @@
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <mutex>
|
||||
#ifdef _M_ARM_64EC
|
||||
#include <winnt.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
@@ -23,12 +26,12 @@ public:
|
||||
uintptr_t GuestCode;
|
||||
};
|
||||
|
||||
LookupCache(FEXCore::Context::ContextImpl *CTX);
|
||||
LookupCache(FEXCore::Context::ContextImpl* CTX);
|
||||
~LookupCache();
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
// Try L1, no lock needed
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
@@ -37,7 +40,7 @@ public:
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize -1)) >> 12;
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
@@ -48,8 +51,7 @@ public:
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == Address)
|
||||
{
|
||||
if (BlockPointers[PageOffset].GuestCode == Address) {
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
return L1Entry.HostCode;
|
||||
@@ -68,6 +70,24 @@ public:
|
||||
return 0;
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
bool CheckPageEC(uint64_t Address) {
|
||||
if (!RtlIsEcCode(Address)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Mark L2 entry for this page as EC by setting the LSB, this can then be
|
||||
// checked by the dispatcher to see if it needs to perform a call/return to
|
||||
// EC code.
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
Pointers[PageIndex] |= 1;
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
||||
|
||||
// Appends Block {Address} to CodePages [Start, Start + Length)
|
||||
@@ -77,8 +97,8 @@ public:
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length -1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
auto &CodePage = CodePages[CurrentPage];
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
auto& CodePage = CodePages[CurrentPage];
|
||||
rv |= CodePage.size() == 0;
|
||||
CodePage.push_back(Address);
|
||||
}
|
||||
@@ -87,7 +107,7 @@ public:
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode) {
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
@@ -95,18 +115,18 @@ public:
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
void Erase(FEXCore::Core::CpuStateFrame *Frame, uint64_t Address) {
|
||||
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData *>(UINTPTR_MAX)});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second(Frame, it->first.HostLink);
|
||||
}
|
||||
@@ -115,7 +135,7 @@ public:
|
||||
BlockList.erase(Address);
|
||||
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = 0;
|
||||
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
|
||||
@@ -124,11 +144,11 @@ public:
|
||||
}
|
||||
|
||||
// Do full map
|
||||
Address = Address & (VirtualMemSize -1);
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
|
||||
uintptr_t *Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// Page for this code didn't even exist, nothing to do
|
||||
@@ -141,7 +161,7 @@ public:
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData * HostLink, const FEXCore::Context::BlockDelinkerFunc &delinker) {
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
@@ -150,14 +170,20 @@ public:
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
|
||||
uintptr_t GetL1Pointer() const { return L1Pointer; }
|
||||
uintptr_t GetPagePointer() const { return PagePointer; }
|
||||
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
|
||||
uintptr_t GetL1Pointer() const {
|
||||
return L1Pointer;
|
||||
}
|
||||
uintptr_t GetPagePointer() const {
|
||||
return PagePointer;
|
||||
}
|
||||
uintptr_t GetVirtualMemorySize() const {
|
||||
return VirtualMemSize;
|
||||
}
|
||||
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
// All other operations must be done from the owning thread.
|
||||
@@ -169,17 +195,17 @@ public:
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
// Do L1
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = HostCode;
|
||||
|
||||
// Do ful map
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize -1);
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
uintptr_t *Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
@@ -223,15 +249,16 @@ private:
|
||||
|
||||
struct BlockLinkTag {
|
||||
uint64_t GuestDestination;
|
||||
FEXCore::Context::ExitFunctionLinkData *HostLink;
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink;
|
||||
|
||||
bool operator <(const BlockLinkTag& other) const {
|
||||
if (GuestDestination < other.GuestDestination)
|
||||
bool operator<(const BlockLinkTag& other) const {
|
||||
if (GuestDestination < other.GuestDestination) {
|
||||
return true;
|
||||
else if (GuestDestination == other.GuestDestination)
|
||||
} else if (GuestDestination == other.GuestDestination) {
|
||||
return HostLink < other.HostLink;
|
||||
else
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -244,7 +271,7 @@ private:
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType *BlockLinks;
|
||||
BlockLinksMapType* BlockLinks;
|
||||
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
@@ -256,7 +283,7 @@ private:
|
||||
|
||||
size_t AllocateOffset {};
|
||||
|
||||
FEXCore::Context::ContextImpl *ctx;
|
||||
uint64_t VirtualMemSize{};
|
||||
FEXCore::Context::ContextImpl* ctx;
|
||||
uint64_t VirtualMemSize {};
|
||||
};
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -6,79 +6,81 @@
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// If any of the config options mismatch on load then the cache won't be used
|
||||
// Any of these will result in codegen changes
|
||||
struct
|
||||
FEX_PACKED
|
||||
CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
uint64_t Cookie{};
|
||||
// If any of the config options mismatch on load then the cache won't be used
|
||||
// Any of these will result in codegen changes
|
||||
struct FEX_PACKED CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
uint64_t Cookie {};
|
||||
|
||||
// Instructions per block configuration
|
||||
int32_t MaxInstPerBlock{};
|
||||
// Instructions per block configuration
|
||||
int32_t MaxInstPerBlock {};
|
||||
|
||||
// Follows CPUID 4000_0001_EAX[3:0]
|
||||
unsigned Arch : 4;
|
||||
// Follows CPUID 4000_0001_EAX[3:0]
|
||||
unsigned Arch : 4;
|
||||
|
||||
// Multiblock enabled
|
||||
unsigned MultiBlock : 1;
|
||||
// Multiblock enabled
|
||||
unsigned MultiBlock : 1;
|
||||
|
||||
// Hardware TSO enabled
|
||||
unsigned HardwareTSOEnabled : 1;
|
||||
// Hardware TSO enabled
|
||||
unsigned HardwareTSOEnabled : 1;
|
||||
|
||||
// TSO enabled
|
||||
unsigned TSOEnabled : 1;
|
||||
// TSO enabled
|
||||
unsigned TSOEnabled : 1;
|
||||
|
||||
// ABI local flag unsafe optimization
|
||||
unsigned ABILocalFlags : 1;
|
||||
// ABI local flag unsafe optimization
|
||||
unsigned ABILocalFlags : 1;
|
||||
|
||||
// Paranoid TSO mode enabled
|
||||
unsigned ParanoidTSO : 1;
|
||||
// Paranoid TSO mode enabled
|
||||
unsigned ParanoidTSO : 1;
|
||||
|
||||
// Guest code execution mode (We don't support live mode switch)
|
||||
unsigned Is64BitMode : 1;
|
||||
// Guest code execution mode (We don't support live mode switch)
|
||||
unsigned Is64BitMode : 1;
|
||||
|
||||
// SMC checks style
|
||||
unsigned SMCChecks : 2;
|
||||
// SMC checks style
|
||||
unsigned SMCChecks : 2;
|
||||
|
||||
// x87 reduced precision
|
||||
unsigned x87ReducedPrecision : 1;
|
||||
// x87 reduced precision
|
||||
unsigned x87ReducedPrecision : 1;
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 19;
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 19;
|
||||
|
||||
bool operator==(CodeObjectSerializationConfig const &other) const {
|
||||
return Cookie == other.Cookie &&
|
||||
MaxInstPerBlock == other.MaxInstPerBlock &&
|
||||
Arch == other.Arch &&
|
||||
MultiBlock == other.MultiBlock &&
|
||||
HardwareTSOEnabled == other.HardwareTSOEnabled &&
|
||||
TSOEnabled == other.TSOEnabled &&
|
||||
ABILocalFlags == other.ABILocalFlags &&
|
||||
ParanoidTSO == other.ParanoidTSO &&
|
||||
Is64BitMode == other.Is64BitMode &&
|
||||
SMCChecks == other.SMCChecks &&
|
||||
x87ReducedPrecision == other.x87ReducedPrecision;
|
||||
}
|
||||
static uint64_t GetHash(CodeObjectSerializationConfig const &other) {
|
||||
// For < 64-bits of data just pack directly
|
||||
// Skip the cookie
|
||||
uint64_t Hash{};
|
||||
Hash <<= 32; Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1; Hash |= other.Arch;
|
||||
Hash <<= 1; Hash |= other.MultiBlock;
|
||||
Hash <<= 1; Hash |= other.HardwareTSOEnabled;
|
||||
Hash <<= 1; Hash |= other.TSOEnabled;
|
||||
Hash <<= 1; Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1; Hash |= other.ParanoidTSO;
|
||||
Hash <<= 1; Hash |= other.Is64BitMode;
|
||||
Hash <<= 2; Hash |= other.SMCChecks;
|
||||
Hash <<= 1; Hash |= other.x87ReducedPrecision;
|
||||
return Hash;
|
||||
}
|
||||
};
|
||||
bool operator==(const CodeObjectSerializationConfig& other) const {
|
||||
return Cookie == other.Cookie && MaxInstPerBlock == other.MaxInstPerBlock && Arch == other.Arch && MultiBlock == other.MultiBlock &&
|
||||
HardwareTSOEnabled == other.HardwareTSOEnabled && TSOEnabled == other.TSOEnabled && ABILocalFlags == other.ABILocalFlags &&
|
||||
ParanoidTSO == other.ParanoidTSO && Is64BitMode == other.Is64BitMode && SMCChecks == other.SMCChecks &&
|
||||
x87ReducedPrecision == other.x87ReducedPrecision;
|
||||
}
|
||||
static uint64_t GetHash(const CodeObjectSerializationConfig& other) {
|
||||
// For < 64-bits of data just pack directly
|
||||
// Skip the cookie
|
||||
uint64_t Hash {};
|
||||
Hash <<= 32;
|
||||
Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1;
|
||||
Hash |= other.Arch;
|
||||
Hash <<= 1;
|
||||
Hash |= other.MultiBlock;
|
||||
Hash <<= 1;
|
||||
Hash |= other.HardwareTSOEnabled;
|
||||
Hash <<= 1;
|
||||
Hash |= other.TSOEnabled;
|
||||
Hash <<= 1;
|
||||
Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1;
|
||||
Hash |= other.ParanoidTSO;
|
||||
Hash <<= 1;
|
||||
Hash |= other.Is64BitMode;
|
||||
Hash <<= 2;
|
||||
Hash |= other.SMCChecks;
|
||||
Hash <<= 1;
|
||||
Hash |= other.x87ReducedPrecision;
|
||||
return Hash;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is generated!");
|
||||
}
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is "
|
||||
"generated!");
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -11,120 +11,112 @@
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename) {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
|
||||
#ifndef _WIN32
|
||||
// This function adds a named region *JOB* to our named region handler
|
||||
// This needs to be as fast as possible to keep out of the way of the JIT
|
||||
// This function adds a named region *JOB* to our named region handler
|
||||
// This needs to be as fast as possible to keep out of the way of the JIT
|
||||
|
||||
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
|
||||
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
|
||||
|
||||
if (!BaseFilename.empty()) {
|
||||
// Create a new entry that once set up will be put in to our section object map
|
||||
auto Entry = fextl::make_unique<CodeRegionEntry>(
|
||||
Base,
|
||||
Size,
|
||||
Offset,
|
||||
filename,
|
||||
NamedRegionHandler->DefaultCodeHeader(Base, Offset)
|
||||
);
|
||||
if (!BaseFilename.empty()) {
|
||||
// Create a new entry that once set up will be put in to our section object map
|
||||
auto Entry = fextl::make_unique<CodeRegionEntry>(Base, Size, Offset, filename, NamedRegionHandler->DefaultCodeHeader(Base, Offset));
|
||||
|
||||
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
|
||||
Entry->NamedJobRefCountMutex.lock();
|
||||
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
|
||||
Entry->NamedJobRefCountMutex.lock();
|
||||
|
||||
CodeRegionMapType::iterator EntryIterator;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
|
||||
auto it = EntryMap.emplace(Base, std::move(Entry));
|
||||
if (!it.second) {
|
||||
// This happens when an application overwrites a previous region without unmapping what was there
|
||||
|
||||
// Lock this entry's Named job reference counter.
|
||||
// Once this passes then we know that this section has been loaded.
|
||||
it.first->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Finalize anything the region needs to do first.
|
||||
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
|
||||
|
||||
// munmap the file that was mapped
|
||||
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
|
||||
}
|
||||
|
||||
// Now overwrite the entry in the map
|
||||
it = EntryMap.insert_or_assign(Base, std::move(Entry));
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
else {
|
||||
// No overwrite, just insert
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
}
|
||||
|
||||
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
|
||||
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
|
||||
// do the loading for us.
|
||||
//
|
||||
// Create the async work queue job now so it can load
|
||||
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
#ifndef _WIN32
|
||||
// Removing a named region through the job system
|
||||
// We need to find the entry that we are deleting first
|
||||
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
CodeRegionMapType::iterator EntryIterator;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto &EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
auto it = EntryMap.find(Base);
|
||||
if (it != EntryMap.end()) {
|
||||
// Lock the job ref counter since we are erasing it
|
||||
// Once this passes it will have been loaded
|
||||
it->second->NamedJobRefCountMutex.lock();
|
||||
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
|
||||
// Take the pointer from the map
|
||||
EntryPointer = std::move(it->second);
|
||||
auto it = EntryMap.emplace(Base, std::move(Entry));
|
||||
if (!it.second) {
|
||||
// This happens when an application overwrites a previous region without unmapping what was there
|
||||
|
||||
// We can now unmap the file data
|
||||
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
|
||||
// Lock this entry's Named job reference counter.
|
||||
// Once this passes then we know that this section has been loaded.
|
||||
it.first->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Remove this from the entry map
|
||||
EntryMap.erase(it);
|
||||
// Finalize anything the region needs to do first.
|
||||
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
|
||||
|
||||
// munmap the file that was mapped
|
||||
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Tried to remove something that wasn't in our code object tracking
|
||||
return;
|
||||
}
|
||||
|
||||
// Create the async work queue job now so it can finalize what it needs to do
|
||||
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
// Now overwrite the entry in the map
|
||||
it = EntryMap.insert_or_assign(Base, std::move(Entry));
|
||||
EntryIterator = it.first;
|
||||
} else {
|
||||
// No overwrite, just insert
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
|
||||
// XXX: Actually add serialization job
|
||||
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
|
||||
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
|
||||
// do the loading for us.
|
||||
//
|
||||
// Create the async work queue job now so it can load
|
||||
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
#ifndef _WIN32
|
||||
// Removing a named region through the job system
|
||||
// We need to find the entry that we are deleting first
|
||||
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
auto it = EntryMap.find(Base);
|
||||
if (it != EntryMap.end()) {
|
||||
// Lock the job ref counter since we are erasing it
|
||||
// Once this passes it will have been loaded
|
||||
it->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Take the pointer from the map
|
||||
EntryPointer = std::move(it->second);
|
||||
|
||||
// We can now unmap the file data
|
||||
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
|
||||
|
||||
// Remove this from the entry map
|
||||
EntryMap.erase(it);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
|
||||
}
|
||||
} else {
|
||||
// Tried to remove something that wasn't in our code object tracking
|
||||
return;
|
||||
}
|
||||
|
||||
// Create the async work queue job now so it can finalize what it needs to do
|
||||
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
|
||||
// XXX: Actually add serialization job
|
||||
}
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -7,66 +7,67 @@
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl *ctx) {
|
||||
DefaultSerializationConfig.Cookie = CODE_COOKIE;
|
||||
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx) {
|
||||
DefaultSerializationConfig.Cookie = CODE_COOKIE;
|
||||
|
||||
// Initialize the Arch from CPUID
|
||||
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
|
||||
DefaultSerializationConfig.Arch = Arch;
|
||||
// Initialize the Arch from CPUID
|
||||
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
|
||||
DefaultSerializationConfig.Arch = Arch;
|
||||
|
||||
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
|
||||
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
|
||||
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
|
||||
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
|
||||
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
|
||||
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
|
||||
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
|
||||
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
|
||||
}
|
||||
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
|
||||
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
|
||||
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
|
||||
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
|
||||
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
|
||||
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
|
||||
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
|
||||
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string &base_filename, const fextl::string &filename, bool Executable) {
|
||||
// XXX: Add named region objects
|
||||
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename,
|
||||
const fextl::string& filename, bool Executable) {
|
||||
// XXX: Add named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->second->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->second->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
// XXX: Remove named region objects
|
||||
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
// XXX: Remove named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
|
||||
// Walk through all of our jobs sequentially until the work queue is empty
|
||||
while (NamedWorkQueueJobs.load()) {
|
||||
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
|
||||
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
|
||||
// Walk through all of our jobs sequentially until the work queue is empty
|
||||
while (NamedWorkQueueJobs.load()) {
|
||||
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
|
||||
|
||||
{
|
||||
// Lock the work queue mutex for a short moment and grab an item from the list
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
size_t WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
WorkItem = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
// Atomically update the number of jobs
|
||||
--NamedWorkQueueJobs;
|
||||
{
|
||||
// Lock the work queue mutex for a short moment and grab an item from the list
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
size_t WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
WorkItem = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
if (WorkItem) {
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
|
||||
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion *>(WorkItem.get());
|
||||
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
|
||||
}
|
||||
// Atomically update the number of jobs
|
||||
--NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
|
||||
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion *>(WorkItem.get());
|
||||
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
|
||||
}
|
||||
if (WorkItem) {
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
|
||||
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion*>(WorkItem.get());
|
||||
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
|
||||
}
|
||||
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
|
||||
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion*>(WorkItem.get());
|
||||
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -6,80 +6,80 @@
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
namespace {
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::CodeSerialize::CodeObjectSerializeService *This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
static void* ThreadHandler(void* Arg) {
|
||||
FEXCore::CodeSerialize::CodeObjectSerializeService* This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::ContextImpl *ctx)
|
||||
: CTX {ctx}
|
||||
, AsyncHandler { &NamedRegionHandler , this }
|
||||
, NamedRegionHandler { ctx } {
|
||||
Initialize();
|
||||
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx}
|
||||
, AsyncHandler {&NamedRegionHandler, this}
|
||||
, NamedRegionHandler {ctx} {
|
||||
Initialize();
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Shutdown() {
|
||||
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
return;
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Shutdown() {
|
||||
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
return;
|
||||
}
|
||||
WorkerThreadShuttingDown = true;
|
||||
|
||||
WorkerThreadShuttingDown = true;
|
||||
// Kick the working thread
|
||||
WorkAvailable.NotifyAll();
|
||||
|
||||
// Kick the working thread
|
||||
WorkAvailable.NotifyAll();
|
||||
|
||||
if (WorkerThread->joinable()) {
|
||||
// Wait for worker thread to close down
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Initialize() {
|
||||
// Add a canary so we don't crash on empty map iterator handling
|
||||
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
|
||||
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it) {
|
||||
if (Base == ~0ULL) {
|
||||
// Don't do closure on canary
|
||||
return;
|
||||
}
|
||||
// XXX: Do code region closure
|
||||
}
|
||||
|
||||
CodeObjectFileSection const *CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
|
||||
// XXX: Actually fetch code objects from cache
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
|
||||
while (WorkerThreadShuttingDown.load() != true) {
|
||||
// Wait for work
|
||||
WorkAvailable.Wait();
|
||||
|
||||
// Handle named region async jobs first. Highest priority
|
||||
NamedRegionHandler.HandleNamedRegionObjectJobs();
|
||||
|
||||
// XXX: Handle code serialization jobs second.
|
||||
}
|
||||
|
||||
// Do final code region closures on thread shutdown
|
||||
for (auto &it : AddressToEntryMap) {
|
||||
DoCodeRegionClosure(it.first, it.second.get());
|
||||
}
|
||||
|
||||
// Safely clear our maps now
|
||||
AddressToEntryMap.clear();
|
||||
UnrelocatedAddressToEntryMap.clear();
|
||||
if (WorkerThread->joinable()) {
|
||||
// Wait for worker thread to close down
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Initialize() {
|
||||
// Add a canary so we don't crash on empty map iterator handling
|
||||
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
|
||||
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it) {
|
||||
if (Base == ~0ULL) {
|
||||
// Don't do closure on canary
|
||||
return;
|
||||
}
|
||||
// XXX: Do code region closure
|
||||
}
|
||||
|
||||
const CodeObjectFileSection* CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
|
||||
// XXX: Actually fetch code objects from cache
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
|
||||
while (WorkerThreadShuttingDown.load() != true) {
|
||||
// Wait for work
|
||||
WorkAvailable.Wait();
|
||||
|
||||
// Handle named region async jobs first. Highest priority
|
||||
NamedRegionHandler.HandleNamedRegionObjectJobs();
|
||||
|
||||
// XXX: Handle code serialization jobs second.
|
||||
}
|
||||
|
||||
// Do final code region closures on thread shutdown
|
||||
for (auto& it : AddressToEntryMap) {
|
||||
DoCodeRegionClosure(it.first, it.second.get());
|
||||
}
|
||||
|
||||
// Safely clear our maps now
|
||||
AddressToEntryMap.clear();
|
||||
UnrelocatedAddressToEntryMap.clear();
|
||||
}
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -17,445 +17,441 @@
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// XXX: Does this need to be signal safe?
|
||||
using CodeSerializationMutex = std::shared_mutex;
|
||||
struct CodeSerializationData {
|
||||
};
|
||||
// XXX: Does this need to be signal safe?
|
||||
using CodeSerializationMutex = std::shared_mutex;
|
||||
struct CodeSerializationData {};
|
||||
|
||||
struct CodeObjectFileSection {
|
||||
bool Serialized;
|
||||
bool Invalid;
|
||||
const CodeSerializationData *Data;
|
||||
const char *HostCode;
|
||||
uint64_t NumRelocations;
|
||||
const char *Relocations;
|
||||
};
|
||||
struct CodeObjectFileSection {
|
||||
bool Serialized;
|
||||
bool Invalid;
|
||||
const CodeSerializationData* Data;
|
||||
const char* HostCode;
|
||||
uint64_t NumRelocations;
|
||||
const char* Relocations;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is the file header that lives at the start of an object cache file
|
||||
*
|
||||
* This header is updated from multiple processes!
|
||||
* Care must be taken to use OS locks when updating the file backing including this header
|
||||
*/
|
||||
struct CodeObjectSerializationHeader {
|
||||
// The configuration that this file has
|
||||
CodeObjectSerializationConfig Config;
|
||||
// The original RIP that this object section was mapped at
|
||||
uint64_t OriginalBase {};
|
||||
// The original offset in to the file that this object section was loaded from
|
||||
uint64_t OriginalOffset {};
|
||||
// Total amount of code that should be in this file
|
||||
uint64_t TotalCodeSize {};
|
||||
// Used to reserve the TSL map
|
||||
uint64_t NumCodeEntries {};
|
||||
// The number of relocations that point to this section
|
||||
uint64_t NumRelocationsTo {};
|
||||
// Total relocations in this file
|
||||
uint64_t TotalRelocationsCount {};
|
||||
};
|
||||
|
||||
struct CodeRegionEntry {
|
||||
/**
|
||||
* @name Threaded initialization objects for the initial object creation
|
||||
* @{ */
|
||||
// Base address in memory where the code region is at
|
||||
uint64_t Base {};
|
||||
|
||||
// Size of this code entry
|
||||
uint64_t Size {};
|
||||
|
||||
// The offset inside the file that is mapped to Base
|
||||
uint64_t Offset {};
|
||||
|
||||
// Filename of the object
|
||||
fextl::string Filename {};
|
||||
|
||||
CodeObjectSerializationHeader EntryHeader {};
|
||||
/** @} */
|
||||
|
||||
// The filename of the object cache for this entry
|
||||
fextl::string ObjectEntrySourceFilename {};
|
||||
|
||||
// In the case of file corruption that we can detect, we can disable serialization early for an entry
|
||||
// We should be resiliant to corruption but things happen
|
||||
bool StillSerializing {true};
|
||||
|
||||
// Long lived FD for serialization if we have multiple jobs to serialize
|
||||
// Bursts of code entries are common and this reduces file lock overhead
|
||||
//
|
||||
// Especially useful over network mounts where file locks are very slow
|
||||
int CurrentSerializedFD {-1};
|
||||
|
||||
/**
|
||||
* @brief This is the file header that lives at the start of an object cache file
|
||||
*
|
||||
* This header is updated from multiple processes!
|
||||
* Care must be taken to use OS locks when updating the file backing including this header
|
||||
*/
|
||||
struct CodeObjectSerializationHeader {
|
||||
// The configuration that this file has
|
||||
CodeObjectSerializationConfig Config;
|
||||
// The original RIP that this object section was mapped at
|
||||
uint64_t OriginalBase{};
|
||||
// The original offset in to the file that this object section was loaded from
|
||||
uint64_t OriginalOffset{};
|
||||
// Total amount of code that should be in this file
|
||||
uint64_t TotalCodeSize{};
|
||||
// Used to reserve the TSL map
|
||||
uint64_t NumCodeEntries{};
|
||||
// The number of relocations that point to this section
|
||||
uint64_t NumRelocationsTo{};
|
||||
// Total relocations in this file
|
||||
uint64_t TotalRelocationsCount{};
|
||||
};
|
||||
* @name Objects required to sync objects between threads
|
||||
* @{ */
|
||||
// Refcount for the number of outstanding code entries waiting to be written for this object section
|
||||
CodeSerializationMutex ObjectJobRefCountMutex;
|
||||
|
||||
struct CodeRegionEntry {
|
||||
/**
|
||||
* @name Threaded initialization objects for the initial object creation
|
||||
* @{ */
|
||||
// Base address in memory where the code region is at
|
||||
uint64_t Base{};
|
||||
|
||||
// Size of this code entry
|
||||
uint64_t Size{};
|
||||
|
||||
// The offset inside the file that is mapped to Base
|
||||
uint64_t Offset{};
|
||||
|
||||
// Filename of the object
|
||||
fextl::string Filename{};
|
||||
|
||||
CodeObjectSerializationHeader EntryHeader{};
|
||||
/** @} */
|
||||
|
||||
// The filename of the object cache for this entry
|
||||
fextl::string ObjectEntrySourceFilename{};
|
||||
|
||||
// In the case of file corruption that we can detect, we can disable serialization early for an entry
|
||||
// We should be resiliant to corruption but things happen
|
||||
bool StillSerializing {true};
|
||||
|
||||
// Long lived FD for serialization if we have multiple jobs to serialize
|
||||
// Bursts of code entries are common and this reduces file lock overhead
|
||||
//
|
||||
// Especially useful over network mounts where file locks are very slow
|
||||
int CurrentSerializedFD {-1};
|
||||
|
||||
/**
|
||||
* @name Objects required to sync objects between threads
|
||||
* @{ */
|
||||
// Refcount for the number of outstanding code entries waiting to be written for this object section
|
||||
CodeSerializationMutex ObjectJobRefCountMutex;
|
||||
|
||||
// Refcount for outstanding named object region entry loading itself
|
||||
// Will block JIT code cache look up when this has a unique_lock held
|
||||
CodeSerializationMutex NamedJobRefCountMutex;
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Object Entry data management
|
||||
* @{ */
|
||||
|
||||
/**
|
||||
* @name This is the raw file data that we loaded from the code region entry file
|
||||
* @{ */
|
||||
char *CodeData{};
|
||||
size_t FileSize{};
|
||||
|
||||
fextl::vector<CodeObjectFileSection> FileCodeSections;
|
||||
/** @} */
|
||||
|
||||
// This per section map takes the most time to load and needs to be quick
|
||||
// This is the map of all code segments for this entry
|
||||
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap{};
|
||||
/** @} */
|
||||
|
||||
// Default initialization
|
||||
CodeRegionEntry() = default;
|
||||
|
||||
// Initializer specifically for threaded loading
|
||||
CodeRegionEntry(uint64_t Base,
|
||||
uint64_t Size,
|
||||
uint64_t Offset,
|
||||
fextl::string const &Filename,
|
||||
CodeObjectSerializationHeader const &DefaultHeader)
|
||||
: Base {Base}
|
||||
, Size {Size}
|
||||
, Offset {Offset}
|
||||
, Filename {Filename}
|
||||
, EntryHeader {DefaultHeader} {
|
||||
}
|
||||
};
|
||||
|
||||
// Map type must use an interator that isn't invalidation on erase/insert
|
||||
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
|
||||
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
|
||||
|
||||
class NamedRegionObjectHandler;
|
||||
class CodeObjectSerializeService;
|
||||
|
||||
class AsyncJobHandler final {
|
||||
public:
|
||||
/**
|
||||
* @brief Structure containing all the data required to async serialize code objects
|
||||
*/
|
||||
struct SerializationJobData {
|
||||
uint64_t GuestRIP; ///< The RIP for the guest
|
||||
// XXX: Support multiblock
|
||||
uint64_t GuestCodeLength; ///< The Guest's code length
|
||||
uint64_t GuestCodeHash; ///< Hash of the guest code
|
||||
|
||||
void *HostCodeBegin; ///< Host JIT code starting memory address
|
||||
size_t HostCodeLength; ///< Host JIT code length
|
||||
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
|
||||
|
||||
// This is the thread specific ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
|
||||
// This way it will wait until the async job handler is complete with it.
|
||||
CodeSerializationMutex *ThreadJobRefCount;
|
||||
|
||||
// These are the reolocations for this serialization job
|
||||
// Relatively small number of entries most of the time
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
/**
|
||||
* @name Objects filled in from the Code Object Serialization service when a job is added
|
||||
* @{ */
|
||||
// This is the code region's ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
|
||||
CodeSerializationMutex *ObjectJobRefCountMutexPtr;
|
||||
|
||||
// This is the code region iterator to reduce the number of map lookups
|
||||
// This will remain valid while jobs are outstanding for this region
|
||||
CodeRegionMapType::iterator CodeRegionIterator;
|
||||
/** @} */
|
||||
};
|
||||
|
||||
AsyncJobHandler(NamedRegionObjectHandler *NamedRegionHandler, CodeObjectSerializeService *CodeObjectCacheService)
|
||||
: NamedRegionHandler {NamedRegionHandler}
|
||||
, CodeObjectCacheService {CodeObjectCacheService} {}
|
||||
|
||||
protected:
|
||||
friend class CodeObjectSerializeService;
|
||||
friend class NamedRegionObjectHandler;
|
||||
/**
|
||||
* @name Async job submission functions
|
||||
* @{ */
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename);
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Async named region handling
|
||||
* @{ */
|
||||
/**
|
||||
* @brief The async named region jobs to handle.
|
||||
*
|
||||
* Only two, Code serialization goes in to a different queue.
|
||||
*/
|
||||
enum class NamedRegionJobType {
|
||||
JOB_ADD_NAMED_REGION,
|
||||
JOB_REMOVE_NAMED_REGION,
|
||||
};
|
||||
|
||||
class NamedRegionWorkItem {
|
||||
public:
|
||||
NamedRegionJobType GetType() const { return Type; }
|
||||
|
||||
protected:
|
||||
friend class WorkItemAddNamedRegion;
|
||||
NamedRegionWorkItem(NamedRegionJobType type)
|
||||
: Type {type} {}
|
||||
|
||||
private:
|
||||
NamedRegionJobType Type;
|
||||
};
|
||||
|
||||
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemAddNamedRegion(const fextl::string &base, const fextl::string &filename, bool executable, CodeRegionMapType::iterator entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
|
||||
, BaseFilename {base}
|
||||
, Filename {filename}
|
||||
, Executable {executable}
|
||||
, Entry {entry}
|
||||
{}
|
||||
const fextl::string BaseFilename;
|
||||
const fextl::string Filename;
|
||||
bool Executable;
|
||||
CodeRegionMapType::iterator Entry;
|
||||
};
|
||||
|
||||
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
|
||||
, Base {base}
|
||||
, Size {size}
|
||||
, Entry {std::move(entry)} {}
|
||||
|
||||
uint64_t Base;
|
||||
uint64_t Size;
|
||||
fextl::unique_ptr<CodeRegionEntry> Entry;
|
||||
};
|
||||
/** @} */
|
||||
|
||||
private:
|
||||
NamedRegionObjectHandler *NamedRegionHandler;
|
||||
CodeObjectSerializeService *CodeObjectCacheService;
|
||||
};
|
||||
|
||||
class NamedRegionObjectHandler final {
|
||||
public:
|
||||
NamedRegionObjectHandler(FEXCore::Context::ContextImpl *ctx);
|
||||
|
||||
void HandleNamedRegionObjectJobs();
|
||||
|
||||
CodeObjectSerializationConfig const &GetDefaultSerializationConfig() const {
|
||||
return DefaultSerializationConfig;
|
||||
}
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
// Return a default code header based off the default serialization config
|
||||
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
|
||||
return CodeObjectSerializationHeader {
|
||||
.Config = DefaultSerializationConfig,
|
||||
.OriginalBase = Base,
|
||||
.OriginalOffset = Offset,
|
||||
.NumCodeEntries = 0,
|
||||
.NumRelocationsTo = 0,
|
||||
.TotalRelocationsCount = 0,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds an asynchronous add named region work item to the object queue
|
||||
*
|
||||
* This adds the job that will do the loading of file resources and data tracking.
|
||||
*/
|
||||
void AsyncAddNamedRegionWorkItem(const fextl::string &base, const fextl::string &filename, bool executable, CodeRegionMapType::iterator entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion> (
|
||||
base,
|
||||
filename,
|
||||
executable,
|
||||
entry
|
||||
));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion> (
|
||||
Base,
|
||||
Size,
|
||||
std::move(Entry)
|
||||
));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
private:
|
||||
// Code version. If the code emission changes then this needs to increment
|
||||
constexpr static uint32_t CODE_VERSION = 0x0;
|
||||
|
||||
// Default cookie header for the file header
|
||||
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
|
||||
|
||||
// Code serialization config for our current process configuration
|
||||
CodeObjectSerializationConfig DefaultSerializationConfig;
|
||||
|
||||
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
|
||||
std::atomic<uint64_t> NamedWorkQueueJobs{};
|
||||
|
||||
// Mutex for ading new jobs to the work queue
|
||||
std::mutex NamedWorkQueueMutex{};
|
||||
|
||||
// The job queue itself
|
||||
// Jobs get consumed as a FIFO
|
||||
// Jobs always get appended to the end
|
||||
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue{};
|
||||
|
||||
/**
|
||||
* @name Named Region object handling
|
||||
* @{ */
|
||||
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string &base_filename, const fextl::string &filename, bool Executable);
|
||||
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
|
||||
/** @} */
|
||||
};
|
||||
// Refcount for outstanding named object region entry loading itself
|
||||
// Will block JIT code cache look up when this has a unique_lock held
|
||||
CodeSerializationMutex NamedJobRefCountMutex;
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @brief Context specific code object serialization class
|
||||
*
|
||||
* Contains everything required for FEXCore to serialize code objects
|
||||
* @name Object Entry data management
|
||||
* @{ */
|
||||
|
||||
/**
|
||||
* @name This is the raw file data that we loaded from the code region entry file
|
||||
* @{ */
|
||||
char* CodeData {};
|
||||
size_t FileSize {};
|
||||
|
||||
fextl::vector<CodeObjectFileSection> FileCodeSections;
|
||||
/** @} */
|
||||
|
||||
// This per section map takes the most time to load and needs to be quick
|
||||
// This is the map of all code segments for this entry
|
||||
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap {};
|
||||
/** @} */
|
||||
|
||||
// Default initialization
|
||||
CodeRegionEntry() = default;
|
||||
|
||||
// Initializer specifically for threaded loading
|
||||
CodeRegionEntry(uint64_t Base, uint64_t Size, uint64_t Offset, const fextl::string& Filename, const CodeObjectSerializationHeader& DefaultHeader)
|
||||
: Base {Base}
|
||||
, Size {Size}
|
||||
, Offset {Offset}
|
||||
, Filename {Filename}
|
||||
, EntryHeader {DefaultHeader} {}
|
||||
};
|
||||
|
||||
// Map type must use an interator that isn't invalidation on erase/insert
|
||||
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
|
||||
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
|
||||
|
||||
class NamedRegionObjectHandler;
|
||||
class CodeObjectSerializeService;
|
||||
|
||||
class AsyncJobHandler final {
|
||||
public:
|
||||
/**
|
||||
* @brief Structure containing all the data required to async serialize code objects
|
||||
*/
|
||||
class CodeObjectSerializeService final {
|
||||
public:
|
||||
CodeObjectSerializeService(FEXCore::Context::ContextImpl *ctx);
|
||||
struct SerializationJobData {
|
||||
uint64_t GuestRIP; ///< The RIP for the guest
|
||||
// XXX: Support multiblock
|
||||
uint64_t GuestCodeLength; ///< The Guest's code length
|
||||
uint64_t GuestCodeHash; ///< Hash of the guest code
|
||||
|
||||
/**
|
||||
* @brief Initialize the internal interface
|
||||
*
|
||||
* Is a public interface to allow the service to reinitialize after forking
|
||||
*/
|
||||
void Initialize();
|
||||
void* HostCodeBegin; ///< Host JIT code starting memory address
|
||||
size_t HostCodeLength; ///< Host JIT code length
|
||||
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
|
||||
|
||||
/**
|
||||
* @brief Safely shut down the Code Object serialization service.
|
||||
*
|
||||
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
|
||||
*/
|
||||
void Shutdown();
|
||||
// This is the thread specific ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
|
||||
// This way it will wait until the async job handler is complete with it.
|
||||
CodeSerializationMutex* ThreadJobRefCount;
|
||||
|
||||
/**
|
||||
* @name Async interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Loads a named region in to the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address that this named region is loaded
|
||||
* @param Size - The size of the region
|
||||
* @param Offset - The offset from the file
|
||||
* @param filename - The filename itself
|
||||
*/
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string &filename) {
|
||||
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
|
||||
}
|
||||
// These are the reolocations for this serialization job
|
||||
// Relatively small number of entries most of the time
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
/**
|
||||
* @brief Unloads a named region from the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address of the named region
|
||||
* @param Size - The size of the region
|
||||
*/
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
|
||||
}
|
||||
/**
|
||||
* @name Objects filled in from the Code Object Serialization service when a job is added
|
||||
* @{ */
|
||||
// This is the code region's ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
|
||||
CodeSerializationMutex* ObjectJobRefCountMutexPtr;
|
||||
|
||||
/**
|
||||
* @brief Adds a code object serialization job. As async as possible.
|
||||
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
|
||||
*
|
||||
* @param Data - A fully filled out struct containing all the code serialization
|
||||
*/
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
|
||||
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
|
||||
}
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Synchronous interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Synchronously waits for this thread's job queue to become empty.
|
||||
*
|
||||
* This is necessary for when a thread is shutting down
|
||||
*
|
||||
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
|
||||
*/
|
||||
static void WaitForEmptyJobQueue(CodeSerializationMutex *ThreadJobRefCount) {
|
||||
// Once the shared mutex is empty this unique lock will be gained
|
||||
std::unique_lock lk {*ThreadJobRefCount};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Fetches object code from the Code Object Cache for JIT.
|
||||
*
|
||||
* @param GuestRIP - Which GuestRIP to search the cache for
|
||||
*
|
||||
* @return Data required for the JIT to relocate the Object code.
|
||||
*/
|
||||
CodeObjectFileSection const *FetchCodeObjectFromCache(uint64_t GuestRIP);
|
||||
/** @} */
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
/**
|
||||
* @brief Safely closes out code object regions from the map
|
||||
*
|
||||
* @param it - iterator to do a closure on
|
||||
*/
|
||||
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry *it);
|
||||
|
||||
CodeSerializationMutex &GetEntryMapMutex() { return EntryMapMutex; }
|
||||
CodeSerializationMutex &GetUnrelocatedEntryMapMutex() { return EntryMapMutex; }
|
||||
|
||||
CodeRegionMapType &GetEntryMap() { return AddressToEntryMap; }
|
||||
CodeRegionPtrMapType &GetUnrelocatedEntryMap() { return UnrelocatedAddressToEntryMap; }
|
||||
|
||||
/**
|
||||
* @brief Notify the async thread that it has work to do
|
||||
*/
|
||||
void NotifyWork() { WorkAvailable.NotifyOne(); }
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
|
||||
Event WorkAvailable{};
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::atomic_bool WorkerThreadShuttingDown {false};
|
||||
AsyncJobHandler AsyncHandler;
|
||||
NamedRegionObjectHandler NamedRegionHandler;
|
||||
|
||||
// Mutex to hold when modifying the entry maps
|
||||
CodeSerializationMutex EntryMapMutex;
|
||||
CodeSerializationMutex UnrelocatedEntryMapMutex;
|
||||
|
||||
// Entry maps
|
||||
CodeRegionMapType AddressToEntryMap;
|
||||
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
|
||||
// This is the code region iterator to reduce the number of map lookups
|
||||
// This will remain valid while jobs are outstanding for this region
|
||||
CodeRegionMapType::iterator CodeRegionIterator;
|
||||
/** @} */
|
||||
};
|
||||
}
|
||||
|
||||
AsyncJobHandler(NamedRegionObjectHandler* NamedRegionHandler, CodeObjectSerializeService* CodeObjectCacheService)
|
||||
: NamedRegionHandler {NamedRegionHandler}
|
||||
, CodeObjectCacheService {CodeObjectCacheService} {}
|
||||
|
||||
protected:
|
||||
friend class CodeObjectSerializeService;
|
||||
friend class NamedRegionObjectHandler;
|
||||
/**
|
||||
* @name Async job submission functions
|
||||
* @{ */
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename);
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Async named region handling
|
||||
* @{ */
|
||||
/**
|
||||
* @brief The async named region jobs to handle.
|
||||
*
|
||||
* Only two, Code serialization goes in to a different queue.
|
||||
*/
|
||||
enum class NamedRegionJobType {
|
||||
JOB_ADD_NAMED_REGION,
|
||||
JOB_REMOVE_NAMED_REGION,
|
||||
};
|
||||
|
||||
class NamedRegionWorkItem {
|
||||
public:
|
||||
NamedRegionJobType GetType() const {
|
||||
return Type;
|
||||
}
|
||||
|
||||
protected:
|
||||
friend class WorkItemAddNamedRegion;
|
||||
NamedRegionWorkItem(NamedRegionJobType type)
|
||||
: Type {type} {}
|
||||
|
||||
private:
|
||||
NamedRegionJobType Type;
|
||||
};
|
||||
|
||||
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemAddNamedRegion(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
|
||||
, BaseFilename {base}
|
||||
, Filename {filename}
|
||||
, Executable {executable}
|
||||
, Entry {entry} {}
|
||||
const fextl::string BaseFilename;
|
||||
const fextl::string Filename;
|
||||
bool Executable;
|
||||
CodeRegionMapType::iterator Entry;
|
||||
};
|
||||
|
||||
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
|
||||
, Base {base}
|
||||
, Size {size}
|
||||
, Entry {std::move(entry)} {}
|
||||
|
||||
uint64_t Base;
|
||||
uint64_t Size;
|
||||
fextl::unique_ptr<CodeRegionEntry> Entry;
|
||||
};
|
||||
/** @} */
|
||||
|
||||
private:
|
||||
NamedRegionObjectHandler* NamedRegionHandler;
|
||||
CodeObjectSerializeService* CodeObjectCacheService;
|
||||
};
|
||||
|
||||
class NamedRegionObjectHandler final {
|
||||
public:
|
||||
NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
void HandleNamedRegionObjectJobs();
|
||||
|
||||
const CodeObjectSerializationConfig& GetDefaultSerializationConfig() const {
|
||||
return DefaultSerializationConfig;
|
||||
}
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
// Return a default code header based off the default serialization config
|
||||
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
|
||||
return CodeObjectSerializationHeader {
|
||||
.Config = DefaultSerializationConfig,
|
||||
.OriginalBase = Base,
|
||||
.OriginalOffset = Offset,
|
||||
.NumCodeEntries = 0,
|
||||
.NumRelocationsTo = 0,
|
||||
.TotalRelocationsCount = 0,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds an asynchronous add named region work item to the object queue
|
||||
*
|
||||
* This adds the job that will do the loading of file resources and data tracking.
|
||||
*/
|
||||
void AsyncAddNamedRegionWorkItem(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion>(base, filename, executable, entry));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion>(Base, Size, std::move(Entry)));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
private:
|
||||
// Code version. If the code emission changes then this needs to increment
|
||||
constexpr static uint32_t CODE_VERSION = 0x0;
|
||||
|
||||
// Default cookie header for the file header
|
||||
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
|
||||
|
||||
// Code serialization config for our current process configuration
|
||||
CodeObjectSerializationConfig DefaultSerializationConfig;
|
||||
|
||||
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
|
||||
std::atomic<uint64_t> NamedWorkQueueJobs {};
|
||||
|
||||
// Mutex for ading new jobs to the work queue
|
||||
std::mutex NamedWorkQueueMutex {};
|
||||
|
||||
// The job queue itself
|
||||
// Jobs get consumed as a FIFO
|
||||
// Jobs always get appended to the end
|
||||
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue {};
|
||||
|
||||
/**
|
||||
* @name Named Region object handling
|
||||
* @{ */
|
||||
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename, const fextl::string& filename, bool Executable);
|
||||
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
|
||||
/** @} */
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Context specific code object serialization class
|
||||
*
|
||||
* Contains everything required for FEXCore to serialize code objects
|
||||
*/
|
||||
class CodeObjectSerializeService final {
|
||||
public:
|
||||
CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
/**
|
||||
* @brief Initialize the internal interface
|
||||
*
|
||||
* Is a public interface to allow the service to reinitialize after forking
|
||||
*/
|
||||
void Initialize();
|
||||
|
||||
/**
|
||||
* @brief Safely shut down the Code Object serialization service.
|
||||
*
|
||||
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
|
||||
*/
|
||||
void Shutdown();
|
||||
|
||||
/**
|
||||
* @name Async interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Loads a named region in to the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address that this named region is loaded
|
||||
* @param Size - The size of the region
|
||||
* @param Offset - The offset from the file
|
||||
* @param filename - The filename itself
|
||||
*/
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
|
||||
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Unloads a named region from the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address of the named region
|
||||
* @param Size - The size of the region
|
||||
*/
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds a code object serialization job. As async as possible.
|
||||
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
|
||||
*
|
||||
* @param Data - A fully filled out struct containing all the code serialization
|
||||
*/
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
|
||||
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
|
||||
}
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Synchronous interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Synchronously waits for this thread's job queue to become empty.
|
||||
*
|
||||
* This is necessary for when a thread is shutting down
|
||||
*
|
||||
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
|
||||
*/
|
||||
static void WaitForEmptyJobQueue(CodeSerializationMutex* ThreadJobRefCount) {
|
||||
// Once the shared mutex is empty this unique lock will be gained
|
||||
std::unique_lock lk {*ThreadJobRefCount};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Fetches object code from the Code Object Cache for JIT.
|
||||
*
|
||||
* @param GuestRIP - Which GuestRIP to search the cache for
|
||||
*
|
||||
* @return Data required for the JIT to relocate the Object code.
|
||||
*/
|
||||
const CodeObjectFileSection* FetchCodeObjectFromCache(uint64_t GuestRIP);
|
||||
/** @} */
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
/**
|
||||
* @brief Safely closes out code object regions from the map
|
||||
*
|
||||
* @param it - iterator to do a closure on
|
||||
*/
|
||||
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it);
|
||||
|
||||
CodeSerializationMutex& GetEntryMapMutex() {
|
||||
return EntryMapMutex;
|
||||
}
|
||||
CodeSerializationMutex& GetUnrelocatedEntryMapMutex() {
|
||||
return EntryMapMutex;
|
||||
}
|
||||
|
||||
CodeRegionMapType& GetEntryMap() {
|
||||
return AddressToEntryMap;
|
||||
}
|
||||
CodeRegionPtrMapType& GetUnrelocatedEntryMap() {
|
||||
return UnrelocatedAddressToEntryMap;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Notify the async thread that it has work to do
|
||||
*/
|
||||
void NotifyWork() {
|
||||
WorkAvailable.NotifyOne();
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
Event WorkAvailable {};
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::atomic_bool WorkerThreadShuttingDown {false};
|
||||
AsyncJobHandler AsyncHandler;
|
||||
NamedRegionObjectHandler NamedRegionHandler;
|
||||
|
||||
// Mutex to hold when modifying the entry maps
|
||||
CodeSerializationMutex EntryMapMutex;
|
||||
CodeSerializationMutex UnrelocatedEntryMapMutex;
|
||||
|
||||
// Entry maps
|
||||
CodeRegionMapType AddressToEntryMap;
|
||||
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
|
||||
};
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -3,77 +3,77 @@
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
enum class RelocationTypes : uint8_t {
|
||||
// 8 byte literal in memory for symbol
|
||||
// Aligned to struct RelocNamedSymbolLiteral
|
||||
RELOC_NAMED_SYMBOL_LITERAL,
|
||||
enum class RelocationTypes : uint8_t {
|
||||
// 8 byte literal in memory for symbol
|
||||
// Aligned to struct RelocNamedSymbolLiteral
|
||||
RELOC_NAMED_SYMBOL_LITERAL,
|
||||
|
||||
// Fixed size named thunk move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocNamedThunkMove
|
||||
RELOC_NAMED_THUNK_MOVE,
|
||||
// Fixed size named thunk move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocNamedThunkMove
|
||||
RELOC_NAMED_THUNK_MOVE,
|
||||
|
||||
// Fixed size guest RIP move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocGuestRIPMove
|
||||
RELOC_GUEST_RIP_MOVE,
|
||||
// Fixed size guest RIP move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocGuestRIPMove
|
||||
RELOC_GUEST_RIP_MOVE,
|
||||
};
|
||||
|
||||
struct RelocationTypeHeader final {
|
||||
RelocationTypes Type;
|
||||
};
|
||||
|
||||
struct RelocNamedSymbolLiteral final {
|
||||
enum class NamedSymbol : uint8_t {
|
||||
///< Thread specific relocations
|
||||
// JIT Literal pointers
|
||||
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
|
||||
};
|
||||
|
||||
struct RelocationTypeHeader final {
|
||||
RelocationTypes Type;
|
||||
};
|
||||
RelocationTypeHeader Header {};
|
||||
|
||||
struct RelocNamedSymbolLiteral final {
|
||||
enum class NamedSymbol : uint8_t {
|
||||
///< Thread specific relocations
|
||||
// JIT Literal pointers
|
||||
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
|
||||
};
|
||||
NamedSymbol Symbol;
|
||||
|
||||
RelocationTypeHeader Header{};
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
};
|
||||
|
||||
NamedSymbol Symbol;
|
||||
struct RelocNamedThunkMove final {
|
||||
RelocationTypeHeader Header {};
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset{};
|
||||
};
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
struct RelocNamedThunkMove final {
|
||||
RelocationTypeHeader Header{};
|
||||
// The thunk SHA256 hash
|
||||
IR::SHA256Sum Symbol;
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
};
|
||||
|
||||
// The thunk SHA256 hash
|
||||
IR::SHA256Sum Symbol;
|
||||
struct RelocGuestRIPMove final {
|
||||
RelocationTypeHeader Header {};
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset{};
|
||||
};
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
struct RelocGuestRIPMove final {
|
||||
RelocationTypeHeader Header{};
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
// The unrelocated RIP that is being moved
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset{};
|
||||
union Relocation {
|
||||
RelocationTypeHeader Header {};
|
||||
|
||||
// The unrelocated RIP that is being moved
|
||||
uint64_t GuestRIP;
|
||||
};
|
||||
RelocNamedSymbolLiteral NamedSymbolLiteral;
|
||||
// This makes our union of relocations at least 48 bytes
|
||||
// It might be more efficient to not use a union
|
||||
RelocNamedThunkMove NamedThunkMove;
|
||||
|
||||
union Relocation {
|
||||
RelocationTypeHeader Header{};
|
||||
|
||||
RelocNamedSymbolLiteral NamedSymbolLiteral;
|
||||
// This makes our union of relocations at least 48 bytes
|
||||
// It might be more efficient to not use a union
|
||||
RelocNamedThunkMove NamedThunkMove;
|
||||
|
||||
RelocGuestRIPMove GuestRIPMove;
|
||||
};
|
||||
}
|
||||
RelocGuestRIPMove GuestRIPMove;
|
||||
};
|
||||
} // namespace FEXCore::CPU
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -22,10 +22,10 @@ class OrderedNode;
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode *RotatedNode{};
|
||||
OrderedNode* RotatedNode {};
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
@@ -34,8 +34,7 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto Sha1HRotated = _VSha1H(Duplicated);
|
||||
RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
// SHA1 extension missing, manually rotate.
|
||||
// Emulate rotate.
|
||||
auto ShiftLeft = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Dest, 30);
|
||||
@@ -48,20 +47,20 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode *NewVec = _VExtr(16, 8, Dest, Src, 1);
|
||||
OrderedNode* NewVec = _VExtr(16, 8, Dest, Src, 1);
|
||||
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
OrderedNode *Result = _VXor(16, 1, Dest, NewVec);
|
||||
OrderedNode* Result = _VXor(16, 1, Dest, NewVec);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
|
||||
// Do all the work without it.
|
||||
@@ -91,41 +90,43 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(),
|
||||
"Src1 needs to be literal here to indicate function and constants");
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here to indicate function and constants");
|
||||
|
||||
using FnType = OrderedNode* (*)(OpDispatchBuilder&, OrderedNode*, OrderedNode*, OrderedNode*);
|
||||
|
||||
const auto f0 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
const auto f0 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
|
||||
return Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._Andn(OpSize::i32Bit, D, B));
|
||||
};
|
||||
const auto f1 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
const auto f1 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
const auto f2 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
const auto f2 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
|
||||
return Self.BitwiseAtLeastTwo(B, C, D);
|
||||
};
|
||||
const auto f3 = [](OpDispatchBuilder &Self, OrderedNode *B, OrderedNode *C, OrderedNode *D) -> OrderedNode* {
|
||||
const auto f3 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
|
||||
constexpr std::array<uint32_t, 4> k_array{
|
||||
constexpr std::array<uint32_t, 4> k_array {
|
||||
0x5A827999U,
|
||||
0x6ED9EBA1U,
|
||||
0x8F1BBCDCU,
|
||||
0xCA62C1D6U,
|
||||
};
|
||||
|
||||
constexpr std::array<FnType, 4> fn_array{
|
||||
f0, f1, f2, f3,
|
||||
constexpr std::array<FnType, 4> fn_array {
|
||||
f0,
|
||||
f1,
|
||||
f2,
|
||||
f3,
|
||||
};
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Data.Literal.Value & 0b11;
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(32, k_array[Imm8]);
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(16, 4, Src, 3);
|
||||
|
||||
@@ -137,7 +138,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
auto C = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto D = _VExtractToGPR(16, 4, Dest, 0);
|
||||
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W0E), K);
|
||||
auto A1 =
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
|
||||
auto D1 = C;
|
||||
@@ -145,13 +147,14 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](OrderedNode *A, OrderedNode *B, OrderedNode *C,
|
||||
OrderedNode *D, OrderedNode *E, OrderedNode *Src, unsigned W_idx) -> RoundResult {
|
||||
const auto Round1To3 = [&](OrderedNode* A, OrderedNode* B, OrderedNode* C, OrderedNode* D, OrderedNode* E, OrderedNode* Src,
|
||||
unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(16, 4, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
|
||||
auto ANext = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), Q), K);
|
||||
auto ANext =
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
|
||||
auto DNext = C;
|
||||
@@ -163,9 +166,9 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
auto [A1, B1, C1, D1, E1] = Round0();
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
|
||||
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
|
||||
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(16, 4, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(16, 4, 1, Dest2, std::get<2>(Final));
|
||||
auto Dest0 = _VInsGPR(16, 4, 0, Dest1, std::get<3>(Final));
|
||||
@@ -174,17 +177,17 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode *Result{};
|
||||
OrderedNode* Result {};
|
||||
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
Result = _VSha256U0(Dest, Src);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
};
|
||||
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
@@ -209,11 +212,12 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
const auto Sigma1 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))), _Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 3);
|
||||
@@ -230,36 +234,38 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::BitwiseAtLeastTwo(OrderedNode *A, OrderedNode *B, OrderedNode *C) {
|
||||
// Returns whether at least 2/3 of A/B/C is true.
|
||||
// Expressed as (A & (B | C)) | (B & C)
|
||||
//
|
||||
// Equivalent to expression in SHA calculations: (A & B) ^ (A & C) ^ (B & C)
|
||||
auto And = _And(OpSize::i32Bit, B, C);
|
||||
auto Or = _Or(OpSize::i32Bit, B, C);
|
||||
return _Or(OpSize::i32Bit, _And(OpSize::i32Bit, A, Or), And);
|
||||
OrderedNode* OpDispatchBuilder::BitwiseAtLeastTwo(OrderedNode* A, OrderedNode* B, OrderedNode* C) {
|
||||
// Returns whether at least 2/3 of A/B/C is true.
|
||||
// Expressed as (A & (B | C)) | (B & C)
|
||||
//
|
||||
// Equivalent to expression in SHA calculations: (A & B) ^ (A & C) ^ (B & C)
|
||||
auto And = _And(OpSize::i32Bit, B, C);
|
||||
auto Or = _Or(OpSize::i32Bit, B, C);
|
||||
return _Or(OpSize::i32Bit, _And(OpSize::i32Bit, A, Or), And);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
const auto Ch = [this](OrderedNode *E, OrderedNode *F, OrderedNode *G) -> OrderedNode* {
|
||||
const auto Ch = [this](OrderedNode* E, OrderedNode* F, OrderedNode* G) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](OrderedNode *A) -> OrderedNode* {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), A, ShiftType::ROR, 13), A, ShiftType::ROR, 22);
|
||||
const auto Sigma0 = [this](OrderedNode* A) -> OrderedNode* {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), A, ShiftType::ROR, 13), A,
|
||||
ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](OrderedNode *E) -> OrderedNode* {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), E, ShiftType::ROR, 11), E, ShiftType::ROR, 25);
|
||||
const auto Sigma1 = [this](OrderedNode* E) -> OrderedNode* {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), E, ShiftType::ROR, 11), E,
|
||||
ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto E0 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto F0 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
OrderedNode *Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
OrderedNode* Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
|
||||
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
@@ -275,7 +281,7 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
auto D0 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
|
||||
OrderedNode * Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
OrderedNode* Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
|
||||
auto WK1 = _VExtractToGPR(16, 4, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
@@ -299,16 +305,16 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Result = _VAESImc(Src);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEnc(16, Dest, Src, ZeroRegister);
|
||||
OrderedNode* Result = _VAESEnc(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -319,19 +325,19 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
|
||||
OrderedNode* Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
|
||||
OrderedNode* Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -342,19 +348,19 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
|
||||
OrderedNode* Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDec(16, Dest, Src, ZeroRegister);
|
||||
OrderedNode* Result = _VAESDec(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -365,19 +371,19 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDec(DstSize, State, Key, ZeroRegister);
|
||||
OrderedNode* Result = _VAESDec(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
|
||||
OrderedNode* Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -388,16 +394,16 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
|
||||
OrderedNode* Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t RCON = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
@@ -407,15 +413,15 @@ OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
OrderedNode *Result = AESKeyGenAssistImpl(Op);
|
||||
OrderedNode* Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Data.Literal.Value);
|
||||
|
||||
auto Res = _PCLMUL(16, Dest, Src, Selector);
|
||||
@@ -427,12 +433,12 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
OrderedNode* Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
|
||||
|
||||
OrderedNode *Res = _PCLMUL(DstSize, Src1, Src2, Selector);
|
||||
OrderedNode* Res = _PCLMUL(DstSize, Src1, Src2, Selector);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -13,90 +13,27 @@ $end_info$
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
FEXCore::X86State::RFLAG_CF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_PF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_AF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_SF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_TF_LOC,
|
||||
FEXCore::X86State::RFLAG_IF_LOC,
|
||||
FEXCore::X86State::RFLAG_DF_LOC,
|
||||
FEXCore::X86State::RFLAG_OF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_IOPL_LOC,
|
||||
FEXCore::X86State::RFLAG_NT_LOC,
|
||||
FEXCore::X86State::RFLAG_RF_LOC,
|
||||
FEXCore::X86State::RFLAG_VM_LOC,
|
||||
FEXCore::X86State::RFLAG_AC_LOC,
|
||||
FEXCore::X86State::RFLAG_VIF_LOC,
|
||||
FEXCore::X86State::RFLAG_VIP_LOC,
|
||||
FEXCore::X86State::RFLAG_ID_LOC,
|
||||
FEXCore::X86State::RFLAG_CF_RAW_LOC, FEXCore::X86State::RFLAG_PF_RAW_LOC, FEXCore::X86State::RFLAG_AF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_RAW_LOC, FEXCore::X86State::RFLAG_SF_RAW_LOC, FEXCore::X86State::RFLAG_TF_LOC,
|
||||
FEXCore::X86State::RFLAG_IF_LOC, FEXCore::X86State::RFLAG_DF_RAW_LOC, FEXCore::X86State::RFLAG_OF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_IOPL_LOC, FEXCore::X86State::RFLAG_NT_LOC, FEXCore::X86State::RFLAG_RF_LOC,
|
||||
FEXCore::X86State::RFLAG_VM_LOC, FEXCore::X86State::RFLAG_AC_LOC, FEXCore::X86State::RFLAG_VIF_LOC,
|
||||
FEXCore::X86State::RFLAG_VIP_LOC, FEXCore::X86State::RFLAG_ID_LOC,
|
||||
};
|
||||
|
||||
void OpDispatchBuilder::ZeroMultipleFlags(uint32_t FlagsMask) {
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
if (ContainsNZCV(FlagsMask)) {
|
||||
// NZCV is stored packed together.
|
||||
// It's more optimal to zero NZCV with move+bic instead of multiple bics.
|
||||
auto NZCVFlagsMask = FlagsMask & FullNZCVMask;
|
||||
if (NZCVFlagsMask == FullNZCVMask) {
|
||||
ZeroNZCV();
|
||||
}
|
||||
else {
|
||||
const auto IndexMask = NZCVIndexMask(FlagsMask);
|
||||
|
||||
if (std::popcount(NZCVFlagsMask) == 1) {
|
||||
// It's more optimal to store only one here.
|
||||
|
||||
for (size_t i = 0; NZCVFlagsMask && i < FlagOffsets.size(); ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
const auto FlagMask = 1U << FlagOffset;
|
||||
if (!(FlagMask & NZCVFlagsMask)) {
|
||||
continue;
|
||||
}
|
||||
SetRFLAG(ZeroConst, FlagOffset);
|
||||
NZCVFlagsMask &= ~(FlagMask);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto IndexMaskConstant = _Constant(IndexMask);
|
||||
auto NewNZCV = _Andn(OpSize::i64Bit, GetNZCV(), IndexMaskConstant);
|
||||
SetNZCV(NewNZCV);
|
||||
}
|
||||
// Unset the possibly set bits.
|
||||
PossiblySetNZCVBits &= ~IndexMask;
|
||||
}
|
||||
|
||||
// Handled NZCV, so remove it from the mask.
|
||||
FlagsMask &= ~FullNZCVMask;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ZeroPF_AF() {
|
||||
// PF is stored inverted, so invert it when we zero.
|
||||
if (FlagsMask & (1u << X86State::RFLAG_PF_RAW_LOC)) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Constant(1));
|
||||
FlagsMask &= ~(1u << X86State::RFLAG_PF_RAW_LOC);
|
||||
}
|
||||
|
||||
// Handle remaining masks.
|
||||
for (size_t i = 0; FlagsMask && i < FlagOffsets.size(); ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
const auto FlagMask = 1U << FlagOffset;
|
||||
if (!(FlagMask & FlagsMask)) {
|
||||
continue;
|
||||
}
|
||||
SetRFLAG(ZeroConst, FlagOffset);
|
||||
FlagsMask &= ~(FlagMask);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Constant(1));
|
||||
SetAF(0);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode* Src) {
|
||||
size_t NumFlags = FlagOffsets.size();
|
||||
if (Lower8) {
|
||||
// Calculate flags early.
|
||||
@@ -104,8 +41,7 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
// This is only a partial overwrite of flags since OF isn't stored here.
|
||||
CalculateDeferredFlags();
|
||||
NumFlags = 5;
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
// We are overwriting all RFLAGS. Invalidate the deferred flag state.
|
||||
InvalidateDeferredFlags();
|
||||
}
|
||||
@@ -125,7 +61,7 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
SetRFLAG(Src, FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
} else if (FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
// PF is stored parity flipped
|
||||
OrderedNode *Tmp = _Bfe(OpSize::i32Bit, 1, FlagOffset, Src);
|
||||
OrderedNode* Tmp = _Bfe(OpSize::i32Bit, 1, FlagOffset, Src);
|
||||
Tmp = _Xor(OpSize::i32Bit, Tmp, _Constant(1));
|
||||
SetRFLAG(Tmp, FlagOffset);
|
||||
} else {
|
||||
@@ -134,15 +70,14 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
OrderedNode* OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
OrderedNode *Original = _Constant(0);
|
||||
OrderedNode* Original = _Constant(0);
|
||||
|
||||
// SF/ZF and N/Z are together on both arm64 and x86_64, so we special case that.
|
||||
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_RAW_LOC)) &&
|
||||
(FlagsMask & (1 << FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_RAW_LOC)) && (FlagsMask & (1 << FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
|
||||
// Handle CF first, since it's at bit 0 and hence doesn't need shift or OR.
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_RAW_LOC)) {
|
||||
@@ -156,21 +91,20 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if ((GetNZ && (FlagOffset == FEXCore::X86State::RFLAG_SF_RAW_LOC ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_ZF_RAW_LOC)) ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_CF_RAW_LOC ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
if ((GetNZ && (FlagOffset == FEXCore::X86State::RFLAG_SF_RAW_LOC || FlagOffset == FEXCore::X86State::RFLAG_ZF_RAW_LOC)) ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_CF_RAW_LOC || FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
// Already handled
|
||||
continue;
|
||||
}
|
||||
|
||||
// Note that the Bfi only considers the bottom bit of the flag, the rest of
|
||||
// the byte is allowed to be garbage.
|
||||
OrderedNode *Flag;
|
||||
if (FlagOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC)
|
||||
OrderedNode* Flag;
|
||||
if (FlagOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
Flag = LoadAF();
|
||||
else
|
||||
} else {
|
||||
Flag = GetRFLAG(FlagOffset);
|
||||
}
|
||||
|
||||
Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset);
|
||||
}
|
||||
@@ -180,7 +114,7 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// instead.
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_RAW_LOC)) {
|
||||
// Set every bit except the bottommost.
|
||||
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(), _Constant(~1ull));
|
||||
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false), _Constant(~1ull));
|
||||
|
||||
// Rotate the bottom bit to the appropriate location for PF, so we get
|
||||
// something like 111P1111. Then invert that to get 000p0000. Then OR that
|
||||
@@ -198,16 +132,17 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
}
|
||||
|
||||
// The constant is OR'ed in at the end, to avoid a pointless or xzr, #2.
|
||||
if ((1U << X86State::RFLAG_RESERVED_LOC) & FlagsMask)
|
||||
if ((1U << X86State::RFLAG_RESERVED_LOC) & FlagsMask) {
|
||||
Original = _Or(OpSize::i64Bit, Original, _Constant(2));
|
||||
}
|
||||
|
||||
return Original;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool Sub) {
|
||||
void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, OrderedNode* Src2, bool Sub) {
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
uint64_t SignBit = (SrcSize * 8) - 1;
|
||||
OrderedNode *Anded = nullptr;
|
||||
OrderedNode* Anded = nullptr;
|
||||
|
||||
// For add, OF is set iff the sources have the same sign but the destination
|
||||
// sign differs. If we know a source sign, we can simplify the expression: if
|
||||
@@ -221,38 +156,44 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNo
|
||||
if (IsValueConstant(WrapNode(Src2), &Const)) {
|
||||
bool Negative = (Const & (1ull << SignBit)) != 0;
|
||||
|
||||
if (Negative ^ Sub)
|
||||
if (Negative ^ Sub) {
|
||||
Anded = _Andn(OpSize, Src1, Res);
|
||||
else
|
||||
} else {
|
||||
Anded = _Andn(OpSize, Res, Src1);
|
||||
}
|
||||
} else {
|
||||
auto XorOp1 = _Xor(OpSize, Src1, Src2);
|
||||
auto XorOp2 = _Xor(OpSize, Res, Src1);
|
||||
|
||||
if (Sub)
|
||||
if (Sub) {
|
||||
Anded = _And(OpSize, XorOp2, XorOp1);
|
||||
else
|
||||
} else {
|
||||
Anded = _Andn(OpSize, XorOp2, XorOp1);
|
||||
}
|
||||
}
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadPFRaw() {
|
||||
OrderedNode* OpDispatchBuilder::LoadPFRaw(bool Invert) {
|
||||
// Read the stored byte. This is the original result (up to 64-bits), it needs
|
||||
// parity calculated.
|
||||
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
// Cast the input to a 32-bit FPR. Logically we only need 8-bit, but that would
|
||||
// generate unwanted an ubfx instruction. VPopcount will ignore the upper bits anyway.
|
||||
auto InputFPR = _VCastFromGPR(4, 4, Result);
|
||||
// Cascade to calculate parity of bottom 8-bits to bottom bit.
|
||||
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 4);
|
||||
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 2);
|
||||
|
||||
// Calculate the popcount.
|
||||
auto Count = _VPopcount(1, 1, InputFPR);
|
||||
return _VExtractToGPR(8, 1, Count, 0);
|
||||
if (Invert) {
|
||||
Result = _XornShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
|
||||
} else {
|
||||
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadAF() {
|
||||
OrderedNode* OpDispatchBuilder::LoadAF() {
|
||||
// Read the stored value. This is the XOR of the arguments.
|
||||
auto AFWord = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
@@ -273,16 +214,29 @@ void OpDispatchBuilder::FixupAF() {
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
OrderedNode *XorRes = _Xor(OpSize::i32Bit, AFRaw, PFRaw);
|
||||
OrderedNode* XorRes = _Xor(OpSize::i32Bit, AFRaw, PFRaw);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculatePF(OrderedNode *Res) {
|
||||
void OpDispatchBuilder::SetAFAndFixup(OrderedNode* AF) {
|
||||
// We have a value of AF, we shift into AF[4]. We need to fixup AF[4] so that
|
||||
// we get the right value when we XOR in PF[4] later. The easiest solution is
|
||||
// to XOR by PF[4], since:
|
||||
//
|
||||
// (AF[4] ^ PF[4]) ^ PF[4] = AF[4]
|
||||
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
OrderedNode* XorRes = _XorShift(OpSize::i32Bit, PFRaw, AF, ShiftType::LSL, 4);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculatePF(OrderedNode* Res) {
|
||||
// Calculation is entirely deferred until load, just store the 8-bit result.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateAF(OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculateAF(OrderedNode* Src1, OrderedNode* Src2) {
|
||||
// We only care about bit 4 in the subsequent XOR. If we'll XOR with 0,
|
||||
// there's no sense XOR'ing at all. If we'll XOR with 1, that's just
|
||||
// inverting.
|
||||
@@ -300,15 +254,16 @@ void OpDispatchBuilder::CalculateAF(OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// We store the XOR of the arguments. At read time, we XOR with the
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit.
|
||||
OrderedNode *XorRes = _Xor(OpSize::i32Bit, Src1, Src2);
|
||||
OrderedNode* XorRes = _Xor(OpSize::i32Bit, Src1, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
if (CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE) {
|
||||
// Nothing to do
|
||||
if (NZCVDirty && CachedNZCV)
|
||||
if (NZCVDirty && CachedNZCV) {
|
||||
_StoreNZCV(CachedNZCV);
|
||||
}
|
||||
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
@@ -316,162 +271,68 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
}
|
||||
|
||||
switch (CurrentDeferredFlags.Type) {
|
||||
case FlagsGenerationType::TYPE_SUB:
|
||||
CalculateFlags_SUB(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src2,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.UpdateCF);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_MUL:
|
||||
CalculateFlags_MUL(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_UMUL:
|
||||
CalculateFlags_UMUL(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LOGICAL:
|
||||
CalculateFlags_Logical(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHL:
|
||||
CalculateFlags_ShiftLeft(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHLI:
|
||||
CalculateFlags_ShiftLeftImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHR:
|
||||
CalculateFlags_ShiftRight(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHRI:
|
||||
CalculateFlags_ShiftRightImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHRDI:
|
||||
CalculateFlags_ShiftRightDoubleImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ASHR:
|
||||
CalculateFlags_SignShiftRight(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ASHRI:
|
||||
CalculateFlags_SignShiftRightImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ROR:
|
||||
CalculateFlags_RotateRight(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_RORI:
|
||||
CalculateFlags_RotateRightImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ROL:
|
||||
CalculateFlags_RotateLeft(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ROLI:
|
||||
CalculateFlags_RotateLeftImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BEXTR:
|
||||
CalculateFlags_BEXTR(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSI:
|
||||
CalculateFlags_BLSI(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSMSK:
|
||||
CalculateFlags_BLSMSK(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSR:
|
||||
CalculateFlags_BLSR(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_POPCOUNT:
|
||||
CalculateFlags_POPCOUNT(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BZHI:
|
||||
CalculateFlags_BZHI(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ZCNT:
|
||||
CalculateFlags_ZCNT(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_RDRAND:
|
||||
CalculateFlags_RDRAND(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_NONE:
|
||||
default: ERROR_AND_DIE_FMT("Unhandled flags type {}", CurrentDeferredFlags.Type);
|
||||
case FlagsGenerationType::TYPE_SUB:
|
||||
CalculateFlags_SUB(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Sources.TwoSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src2, CurrentDeferredFlags.Sources.TwoSrcImmediate.UpdateCF);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_MUL:
|
||||
CalculateFlags_MUL(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res, CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_UMUL: CalculateFlags_UMUL(CurrentDeferredFlags.Res); break;
|
||||
case FlagsGenerationType::TYPE_LOGICAL:
|
||||
CalculateFlags_Logical(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res, CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHLI:
|
||||
CalculateFlags_ShiftLeftImmediate(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1, CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHRI:
|
||||
CalculateFlags_ShiftRightImmediate(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1, CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHRDI:
|
||||
CalculateFlags_ShiftRightDoubleImmediate(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1, CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ASHRI:
|
||||
CalculateFlags_SignShiftRightImmediate(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1, CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BEXTR: CalculateFlags_BEXTR(CurrentDeferredFlags.Res); break;
|
||||
case FlagsGenerationType::TYPE_BLSI: CalculateFlags_BLSI(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res); break;
|
||||
case FlagsGenerationType::TYPE_BLSMSK:
|
||||
CalculateFlags_BLSMSK(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res, CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSR:
|
||||
CalculateFlags_BLSR(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res, CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_POPCOUNT: CalculateFlags_POPCOUNT(CurrentDeferredFlags.Res); break;
|
||||
case FlagsGenerationType::TYPE_BZHI:
|
||||
CalculateFlags_BZHI(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res, CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ZCNT: CalculateFlags_ZCNT(CurrentDeferredFlags.SrcSize, CurrentDeferredFlags.Res); break;
|
||||
case FlagsGenerationType::TYPE_RDRAND: CalculateFlags_RDRAND(CurrentDeferredFlags.Res); break;
|
||||
case FlagsGenerationType::TYPE_NONE:
|
||||
default: ERROR_AND_DIE_FMT("Unhandled flags type {}", CurrentDeferredFlags.Type);
|
||||
}
|
||||
|
||||
// Done calculating
|
||||
CurrentDeferredFlags.Type = FlagsGenerationType::TYPE_NONE;
|
||||
|
||||
if (NZCVDirty && CachedNZCV)
|
||||
if (NZCVDirty && CachedNZCV) {
|
||||
_StoreNZCV(CachedNZCV);
|
||||
}
|
||||
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
OrderedNode* OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
OrderedNode *Res;
|
||||
OrderedNode* Res;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
@@ -479,13 +340,19 @@ OrderedNode *OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode
|
||||
HandleNZCV_RMW();
|
||||
Res = _AdcWithFlags(OpSize, Src1, Src2);
|
||||
} else {
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
Res = _Adc(OpSize, Src1, Src2);
|
||||
// Need to zero-extend for correct comparisons below
|
||||
Src2 = _Bfe(OpSize, SrcSize * 8, 0, Src2);
|
||||
|
||||
// Note that we do not extend Src2PlusCF, since we depend on proper
|
||||
// 32-bit arithmetic to correctly handle the Src2 = 0xffff case.
|
||||
OrderedNode* Src2PlusCF = _Adc(OpSize, _Constant(0), Src2);
|
||||
|
||||
// Need to zero-extend for the comparison.
|
||||
Res = _Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
|
||||
auto SelectOpLT = _Select(FEXCore::IR::COND_ULT, Res, Src2, One, Zero);
|
||||
auto SelectOpLE = _Select(FEXCore::IR::COND_ULE, Res, Src2, One, Zero);
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, One, SelectOpLE, SelectOpLT);
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_ULT, Res, Src2PlusCF, One, Zero);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
|
||||
@@ -496,14 +363,14 @@ OrderedNode *OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode
|
||||
return Res;
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
OrderedNode* OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
OrderedNode *Res;
|
||||
OrderedNode* Res;
|
||||
if (SrcSize >= 4) {
|
||||
// Rectify input carry
|
||||
CarryInvert();
|
||||
@@ -514,13 +381,17 @@ OrderedNode *OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode
|
||||
// Rectify output carry
|
||||
CarryInvert();
|
||||
} else {
|
||||
// Zero extend for correct comparison behaviour with Src1 = 0xffff.
|
||||
Src1 = _Bfe(OpSize, SrcSize * 8, 0, Src1);
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
Res = _Sub(OpSize, Src1, _Add(OpSize, Src2, CF));
|
||||
auto Src1MinusCF = _Sub(OpSize, Src1, CF);
|
||||
|
||||
Res = _Sub(OpSize, Src1MinusCF, Src2);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
|
||||
auto SelectOpLT = _Select(FEXCore::IR::COND_UGT, Res, Src1, One, Zero);
|
||||
auto SelectOpLE = _Select(FEXCore::IR::COND_UGE, Res, Src1, One, Zero);
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, One, SelectOpLE, SelectOpLT);
|
||||
// Need to zero-extend for correct comparisons below
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_ULT, Src1MinusCF, Res, One, Zero);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
|
||||
@@ -531,7 +402,7 @@ OrderedNode *OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode
|
||||
return Res;
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
OrderedNode* OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
@@ -539,7 +410,7 @@ OrderedNode *OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
OrderedNode *Res;
|
||||
OrderedNode* Res;
|
||||
if (SrcSize >= 4) {
|
||||
Res = _SubWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
} else {
|
||||
@@ -551,15 +422,16 @@ OrderedNode *OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode
|
||||
|
||||
// If we're updating CF, we need to invert it for correctness. If we're not
|
||||
// updating CF, we need to restore the CF since we stomped over it.
|
||||
if (UpdateCF)
|
||||
if (UpdateCF) {
|
||||
CarryInvert();
|
||||
else
|
||||
} else {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
|
||||
}
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
OrderedNode* OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
@@ -567,7 +439,7 @@ OrderedNode *OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
OrderedNode *Res;
|
||||
OrderedNode* Res;
|
||||
if (SrcSize >= 4) {
|
||||
Res = _AddWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
} else {
|
||||
@@ -578,13 +450,14 @@ OrderedNode *OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode
|
||||
CalculatePF(Res);
|
||||
|
||||
// We stomped over CF while calculation flags, restore it.
|
||||
if (!UpdateCF)
|
||||
if (!UpdateCF) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
|
||||
}
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High) {
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode* Res, OrderedNode* High) {
|
||||
HandleNZCVWrite();
|
||||
|
||||
// PF/AF/ZF/SF
|
||||
@@ -604,11 +477,11 @@ void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
// If High = SignBit, then sets to nZcv. Else sets to nzCV. Since SF/ZF
|
||||
// undefined, this does what we need.
|
||||
auto Zero = _Constant(0);
|
||||
_CondAddNZCV(OpSize::i64Bit, Zero, Zero, CondClassType{COND_EQ}, 0x3 /* nzCV */);
|
||||
_CondAddNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
|
||||
void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode* High) {
|
||||
HandleNZCVWrite();
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
@@ -629,11 +502,11 @@ void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
|
||||
|
||||
// If High = 0, then sets to nZcv. Else sets to nzCV. Since SF/ZF undefined,
|
||||
// this does what we need.
|
||||
_CondAddNZCV(Size, Zero, Zero, CondClassType{COND_EQ}, 0x3 /* nzCV */);
|
||||
_CondAddNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, OrderedNode* Src2) {
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
@@ -644,76 +517,11 @@ void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, OrderedNode *Res
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res, Src1, Src2](){
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
auto Size = _Constant(SrcSize * 8);
|
||||
auto ShiftAmt = _Sub(OpSize, Size, Src2);
|
||||
auto LastBit = _Lshr(OpSize, Src1, ShiftAmt);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit, 0, true);
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
// When Shift > 1 then OF is undefined
|
||||
auto OFXor = _Xor(OpSize, Src1, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OFXor, SrcSize * 8 - 1, true);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res, Src1, Src2](){
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, _Constant(1));
|
||||
const auto CFSize = IR::SizeToOpSize(std::max<uint8_t>(4u, SrcSize));
|
||||
auto LastBit = _Lshr(CFSize, Src1, ShiftAmt);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit, 0, true);
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// OF flag is set if a sign change occurred
|
||||
auto val = _Xor(OpSize, Src1, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val, SrcSize * 8 - 1, true);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res, Src1, Src2](){
|
||||
// SF/ZF/OF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto CFSize = IR::SizeToOpSize(std::max<uint32_t>(4u, GetOpSize(Src1)));
|
||||
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, _Constant(1));
|
||||
auto LastBit = _Lshr(CFSize, Src1, ShiftAmt);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit, 0, true);
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *UnmaskedRes, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode* UnmaskedRes, OrderedNode* Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
@@ -745,16 +553,18 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Order
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift-1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift - 1, true);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -769,7 +579,7 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
// already zeroed there's nothing to do here.
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
// Set SF and PF. Clobbers OF, but OF only defined for Shift = 1 where it is
|
||||
// set below.
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
@@ -777,7 +587,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift-1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift - 1, true);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -787,9 +597,11 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
CalculateFlags_ShiftRightImmediateCommon(SrcSize, Res, Src1, Shift);
|
||||
|
||||
@@ -803,9 +615,11 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Orde
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
CalculateFlags_ShiftRightImmediateCommon(SrcSize, Res, Src1, Shift);
|
||||
@@ -822,117 +636,15 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
|
||||
auto SizeBits = SrcSize * 8;
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
// Ends up faster overall if we don't have FlagM, slower if we do...
|
||||
// If Shift != 1, OF is undefined so we choose to zero here.
|
||||
if (!CTX->HostFeatures.SupportsFlagM)
|
||||
ZeroCV();
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
|
||||
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, true);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CalculateFlags_ShiftVariable(Src2, [this, SrcSize, Res](){
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto SizeBits = SrcSize * 8;
|
||||
|
||||
// Ends up faster overall if we don't have FlagM, slower if we do...
|
||||
// If Shift != 1, OF is undefined so we choose to zero here.
|
||||
if (!CTX->HostFeatures.SupportsFlagM)
|
||||
ZeroCV();
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
//auto Size = _Constant(GetSrcSize(Res) * 8);
|
||||
//auto ShiftAmt = _Sub(OpSize::i64Bit, Size, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
|
||||
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
|
||||
});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto SizeBits = SrcSize * 8;
|
||||
|
||||
// Ends up faster overall if we don't have FlagM, slower if we do...
|
||||
// If Shift != 1, OF is undefined so we choose to zero here.
|
||||
if (!CTX->HostFeatures.SupportsFlagM)
|
||||
ZeroCV();
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, SizeBits - 1, true);
|
||||
}
|
||||
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, SizeBits - 2, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto SizeBits = SrcSize * 8;
|
||||
|
||||
// Ends up faster overall if we don't have FlagM, slower if we do...
|
||||
// If Shift != 1, OF is undefined so we choose to zero here.
|
||||
if (!CTX->HostFeatures.SupportsFlagM)
|
||||
ZeroCV();
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Res, 0, true);
|
||||
}
|
||||
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, SizeBits - 1);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BEXTR(OrderedNode *Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_BEXTR(OrderedNode* Src) {
|
||||
// ZF is set properly. CF and OF are defined as being set to zero. SF, PF, and
|
||||
// AF are undefined.
|
||||
SetNZ_ZeroCV(GetOpSize(Src), Src);
|
||||
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Result) {
|
||||
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode* Result) {
|
||||
// CF is cleared if Src is zero, otherwise it's set. However, Src is zero iff
|
||||
// Result is zero, so we can test the result instead. So, CF is just the
|
||||
// inverted ZF.
|
||||
@@ -944,14 +656,12 @@ void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Result
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, OrderedNode* Result, OrderedNode* Src) {
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
// CF set according to the Src
|
||||
auto Zero = _Constant(0);
|
||||
@@ -964,7 +674,7 @@ void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, OrderedNode *Resu
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode* Result, OrderedNode* Src) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto CFOp = _Select(IR::COND_EQ, Src, Zero, One, Zero);
|
||||
@@ -973,30 +683,26 @@ void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode *Result
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode *Result) {
|
||||
void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode* Result) {
|
||||
// We need to set ZF while clearing the rest of NZCV. The result of a popcount
|
||||
// is in the range [0, 63]. In particular, it is always positive. So a
|
||||
// combined NZ test will correctly zero SF/CF/OF while setting ZF.
|
||||
SetNZ_ZeroCV(OpSize::i32Bit, Result);
|
||||
|
||||
ZeroMultipleFlags((1U << X86State::RFLAG_AF_RAW_LOC) |
|
||||
(1U << X86State::RFLAG_PF_RAW_LOC));
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode* Result, OrderedNode* Src) {
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, OrderedNode *Result) {
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, OrderedNode* Result) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
// Test ZF of result, SF is undefined so this is ok.
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
@@ -1008,18 +714,13 @@ void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, OrderedNode *Result
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Result, CarryBit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RDRAND(OrderedNode *Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_RDRAND(OrderedNode* Src) {
|
||||
// OF, SF, ZF, AF, PF all zero
|
||||
ZeroNZCV();
|
||||
ZeroPF_AF();
|
||||
|
||||
// CF is set to the incoming source
|
||||
|
||||
uint32_t FlagsMaskToZero =
|
||||
FullNZCVMask |
|
||||
(1U << X86State::RFLAG_AF_RAW_LOC) |
|
||||
(1U << X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
ZeroMultipleFlags(FlagsMaskToZero);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
}
|
||||
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -22,24 +22,24 @@ class OrderedNode;
|
||||
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
//Functions in X87.cpp (no change required)
|
||||
//GetX87Top
|
||||
//SetX87ValidTag
|
||||
//GetX87ValidTag
|
||||
//GetX87Tag (will need changing once special tag handling is implemented)
|
||||
//SetX87FTW
|
||||
//GetX87FTW (will need changing once special tag handling is implemented)
|
||||
//SetX87Top
|
||||
//X87ModifySTP
|
||||
//EMMS
|
||||
//FFREE
|
||||
//FNSTENV
|
||||
//FSTCW
|
||||
//LDSW
|
||||
//FNSTSW
|
||||
//FXCH
|
||||
//FCMOV
|
||||
//FST(register to register)
|
||||
// Functions in X87.cpp (no change required)
|
||||
// GetX87Top
|
||||
// SetX87ValidTag
|
||||
// GetX87ValidTag
|
||||
// GetX87Tag (will need changing once special tag handling is implemented)
|
||||
// SetX87FTW
|
||||
// GetX87FTW (will need changing once special tag handling is implemented)
|
||||
// SetX87Top
|
||||
// X87ModifySTP
|
||||
// EMMS
|
||||
// FFREE
|
||||
// FNSTENV
|
||||
// FSTCW
|
||||
// LDSW
|
||||
// FNSTSW
|
||||
// FXCH
|
||||
// FCMOV
|
||||
// FST(register to register)
|
||||
|
||||
// State loading duplicated from X87.cpp, setting host rounding mode
|
||||
// See issue
|
||||
@@ -64,34 +64,33 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
const auto Size = GetSrcSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
//ignore the rounding precision, we're always 64-bit in F64.
|
||||
//extract rounding mode
|
||||
OrderedNode *roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
OrderedNode* roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
ReconstructX87StateFromFSW(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
//ignore the rounding precision, we're always 64-bit in F64.
|
||||
//extract rounding mode
|
||||
OrderedNode *roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
OrderedNode* NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
OrderedNode* roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
@@ -106,13 +105,13 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
|
||||
|
||||
size_t read_width = (width == 80) ? 16 : width / 8;
|
||||
|
||||
OrderedNode *data{};
|
||||
OrderedNode *converted{};
|
||||
OrderedNode* data {};
|
||||
OrderedNode* converted {};
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Read from memory
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
// Convert to 64bit float
|
||||
if constexpr (width == 32) {
|
||||
converted = _Float_FToF(8, 4, data);
|
||||
} else if constexpr (width == 80) {
|
||||
@@ -120,8 +119,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
|
||||
} else {
|
||||
converted = data;
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
// Implicit arg (does this need to change with width?)
|
||||
auto offset = _Constant(Op->OP & 7);
|
||||
data = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, orig_top, offset), mask);
|
||||
@@ -136,12 +134,9 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
|
||||
_StoreContextIndexed(converted, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64<32>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64<64>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64<80>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FLDF64<32>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FLDF64<64>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FLDF64<80>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
// Update TOP
|
||||
@@ -152,8 +147,8 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
// Read from memory
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode *converted = _F80BCDLoad(data);
|
||||
OrderedNode* data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode* converted = _F80BCDLoad(data);
|
||||
converted = _F80CVT(8, converted);
|
||||
_StoreContextIndexed(converted, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -162,7 +157,7 @@ void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
auto orig_top = GetX87Top();
|
||||
auto data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode *converted = _F80CVTTo(data, 8);
|
||||
OrderedNode* converted = _F80CVTTo(data, 8);
|
||||
converted = _F80BCDStore(converted);
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1);
|
||||
@@ -185,20 +180,13 @@ void OpDispatchBuilder::FLDF64_Const(OpcodeArgs) {
|
||||
_StoreContextIndexed(data, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64_Const<0x3FF0000000000000>(OpcodeArgs); // 1.0
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64_Const<0x400A934F0979A372>(OpcodeArgs); // log2l(10)
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64_Const<0x3FF71547652B82FE>(OpcodeArgs); // log2l(e)
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64_Const<0x400921FB54442D18>(OpcodeArgs); // pi
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64_Const<0x3FD34413509F79FF>(OpcodeArgs); // log10l(2)
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64_Const<0x3FE62E42FEFA39EF>(OpcodeArgs); // log(2)
|
||||
template
|
||||
void OpDispatchBuilder::FLDF64_Const<0>(OpcodeArgs); // 0.0
|
||||
template void OpDispatchBuilder::FLDF64_Const<0x3FF0000000000000>(OpcodeArgs); // 1.0
|
||||
template void OpDispatchBuilder::FLDF64_Const<0x400A934F0979A372>(OpcodeArgs); // log2l(10)
|
||||
template void OpDispatchBuilder::FLDF64_Const<0x3FF71547652B82FE>(OpcodeArgs); // log2l(e)
|
||||
template void OpDispatchBuilder::FLDF64_Const<0x400921FB54442D18>(OpcodeArgs); // pi
|
||||
template void OpDispatchBuilder::FLDF64_Const<0x3FD34413509F79FF>(OpcodeArgs); // log10l(2)
|
||||
template void OpDispatchBuilder::FLDF64_Const<0x3FE62E42FEFA39EF>(OpcodeArgs); // log(2)
|
||||
template void OpDispatchBuilder::FLDF64_Const<0>(OpcodeArgs); // 0.0
|
||||
|
||||
void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
// Update TOP
|
||||
@@ -210,7 +198,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
size_t read_width = GetSrcSize(Op);
|
||||
// Read from memory
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
if(read_width == 2) {
|
||||
if (read_width == 2) {
|
||||
data = _Sbfe(OpSize::i64Bit, read_width * 8, 0, data);
|
||||
}
|
||||
auto converted = _Float_FromGPR_S(8, read_width == 4 ? 4 : 8, data);
|
||||
@@ -223,14 +211,14 @@ void OpDispatchBuilder::FSTF64(OpcodeArgs) {
|
||||
auto orig_top = GetX87Top();
|
||||
auto data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
if constexpr (width == 64) {
|
||||
//Store 64-bit float directly
|
||||
// Store 64-bit float directly
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, data, 8, 1);
|
||||
} else if constexpr (width == 32) {
|
||||
//Convert to 32-bit float and store
|
||||
// Convert to 32-bit float and store
|
||||
auto result = _Float_FToF(4, 8, data);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, result, 4, 1);
|
||||
} else if constexpr (width == 80) {
|
||||
//Convert to 80-bit float
|
||||
// Convert to 80-bit float
|
||||
auto result = _F80CVTTo(data, 8);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, result, 10, 1);
|
||||
}
|
||||
@@ -244,19 +232,16 @@ void OpDispatchBuilder::FSTF64(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FSTF64<32>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FSTF64<64>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FSTF64<80>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSTF64<32>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSTF64<64>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSTF64<80>(OpcodeArgs);
|
||||
|
||||
template<bool Truncate>
|
||||
void OpDispatchBuilder::FISTF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
OrderedNode *data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
if constexpr (Truncate) {
|
||||
data = _Float_ToGPR_ZS(Size == 4 ? 4 : 8, 8, data);
|
||||
} else {
|
||||
@@ -273,18 +258,16 @@ void OpDispatchBuilder::FISTF64(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FISTF64<false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FISTF64<true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FISTF64<false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FISTF64<true>(OpcodeArgs);
|
||||
|
||||
template <size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
|
||||
template<size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FADDF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode *StackLocation = top;
|
||||
OrderedNode* StackLocation = top;
|
||||
|
||||
OrderedNode *arg{};
|
||||
OrderedNode *b{};
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -292,7 +275,7 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
if (width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
@@ -326,26 +309,20 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs) {
|
||||
_StoreContextIndexed(result, StackLocation, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FADDF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FADDF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FADDF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FADDF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FADDF64<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FADDF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FADDF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FADDF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FADDF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template<size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FMULF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode *StackLocation = top;
|
||||
OrderedNode *arg{};
|
||||
OrderedNode *b{};
|
||||
OrderedNode* StackLocation = top;
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -353,7 +330,7 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
if (width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
@@ -390,34 +367,28 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs) {
|
||||
_StoreContextIndexed(result, StackLocation, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FMULF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FMULF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FMULF64<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FMULF64<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FMULF64<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FMULF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FMULF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FMULF64<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FMULF64<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode *StackLocation = top;
|
||||
OrderedNode *arg{};
|
||||
OrderedNode *b{};
|
||||
OrderedNode* StackLocation = top;
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
if (width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
@@ -440,11 +411,10 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode *result{};
|
||||
OrderedNode* result {};
|
||||
if constexpr (reverse) {
|
||||
result = _VFDiv(8, 8, b, a);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
result = _VFDiv(8, 8, a, b);
|
||||
}
|
||||
|
||||
@@ -460,50 +430,38 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
|
||||
_StoreContextIndexed(result, StackLocation, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FDIVF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FDIVF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode *StackLocation = top;
|
||||
OrderedNode *arg{};
|
||||
OrderedNode *b{};
|
||||
OrderedNode* StackLocation = top;
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
if (width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
@@ -526,11 +484,10 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode *result{};
|
||||
OrderedNode* result {};
|
||||
if constexpr (reverse) {
|
||||
result = _VFSub(8, 8, b, a);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
result = _VFSub(8, 8, a, b);
|
||||
}
|
||||
|
||||
@@ -547,35 +504,23 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
|
||||
_StoreContextIndexed(result, StackLocation, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FSUBF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FSUBF64<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::FCHSF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
@@ -598,7 +543,7 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto low = _Constant(0);
|
||||
OrderedNode *data = _VCastFromGPR(8, 8, low);
|
||||
OrderedNode* data = _VCastFromGPR(8, 8, low);
|
||||
|
||||
// We are going to clobber NZCV, make sure it's in a GPR first.
|
||||
GetNZCV();
|
||||
@@ -609,7 +554,7 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
|
||||
//TODO: This should obey rounding mode
|
||||
// TODO: This should obey rounding mode
|
||||
void OpDispatchBuilder::FRNDINTF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -646,14 +591,14 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto mask = _Constant(7);
|
||||
|
||||
OrderedNode *arg{};
|
||||
OrderedNode *b{};
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
if (width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
@@ -679,8 +624,7 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
_FCmp(8, a, b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
// Invalidate deferred flags early
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
@@ -698,8 +642,7 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
// Set the new top now
|
||||
top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
|
||||
SetX87Top(top);
|
||||
}
|
||||
else if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
} else if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
// if we are popping then we must first mark this location as empty
|
||||
SetX87ValidTag(top, false);
|
||||
// Set the new top now
|
||||
@@ -708,24 +651,17 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FCOMIF64<32, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FCOMIF64<32, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FCOMIF64<64, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FCOMIF64<64, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FCOMIF64<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, true>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FCOMIF64<16, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FCOMIF64<16, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::FCOMIF64<32, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::FCOMIF64<32, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs);
|
||||
|
||||
|
||||
void OpDispatchBuilder::FSQRTF64(OpcodeArgs) {
|
||||
@@ -746,8 +682,7 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
|
||||
|
||||
DeriveOp(result, IROp, _F64SIN(a));
|
||||
|
||||
if constexpr (IROp == IR::OP_F64SIN ||
|
||||
IROp == IR::OP_F64COS) {
|
||||
if constexpr (IROp == IR::OP_F64SIN || IROp == IR::OP_F64COS) {
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
@@ -756,12 +691,9 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
|
||||
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64F2XM1>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64SIN>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64COS>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64F2XM1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64SIN>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::X87UnaryOpF64<IR::OP_F64COS>(OpcodeArgs);
|
||||
|
||||
|
||||
template<FEXCore::IR::IROps IROp>
|
||||
@@ -769,16 +701,15 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
|
||||
auto mask = _Constant(7);
|
||||
OrderedNode *st1 = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
|
||||
OrderedNode* st1 = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
|
||||
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
st1 = _LoadContextIndexed(st1, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
DeriveOp(result, IROp, _F64ATAN(a, st1));
|
||||
|
||||
if constexpr (IROp == IR::OP_F64FPREM ||
|
||||
IROp == IR::OP_F64FPREM1) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
if constexpr (IROp == IR::OP_F64FPREM || IROp == IR::OP_F64FPREM1) {
|
||||
// TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
@@ -786,12 +717,9 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM1>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64SCALE>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64FPREM>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::X87BinaryOpF64<IR::OP_F64SCALE>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::X87SinCosF64(OpcodeArgs) {
|
||||
auto orig_top = GetX87Top();
|
||||
@@ -821,8 +749,8 @@ void OpDispatchBuilder::X87FYL2XF64(OpcodeArgs) {
|
||||
auto top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, orig_top, _Constant(1)), _Constant(7));
|
||||
SetX87Top(top);
|
||||
|
||||
OrderedNode *st0 = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode *st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* st0 = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
if (Plus1) {
|
||||
auto one = _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000));
|
||||
@@ -863,7 +791,7 @@ void OpDispatchBuilder::X87ATANF64(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
auto a = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode *st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F64ATAN(st1, a);
|
||||
|
||||
@@ -871,7 +799,7 @@ void OpDispatchBuilder::X87ATANF64(OpcodeArgs) {
|
||||
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
//This function converts to F80 on save for compatibility
|
||||
// This function converts to F80 on save for compatibility
|
||||
|
||||
void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
// 14 bytes for 16bit
|
||||
@@ -893,18 +821,16 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
// 4 bytes : data pointer offset
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
OrderedNode *Top = GetX87Top();
|
||||
const auto Size = GetDstSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
OrderedNode* Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ReconstructFSW(), Size);
|
||||
}
|
||||
|
||||
@@ -912,35 +838,35 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
_StoreMem(GPRClass, Size, MemLocation, GetX87FTW(), Size);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 3));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 3));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 4));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 4));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 5));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 5));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 6));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 6));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
}
|
||||
|
||||
OrderedNode *ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
|
||||
OrderedNode* ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
@@ -968,17 +894,16 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
FNINIT(Op);
|
||||
}
|
||||
|
||||
//This function converts from F80 on load for compatibility
|
||||
// This function converts from F80 on load for compatibility
|
||||
|
||||
void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
const auto Size = GetSrcSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
//ignore the rounding precision, we're always 64-bit in F64.
|
||||
//extract rounding mode
|
||||
OrderedNode *roundingMode = NewFCW;
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
OrderedNode* roundingMode = NewFCW;
|
||||
auto roundShift = _Constant(10);
|
||||
auto roundMask = _Constant(3);
|
||||
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
|
||||
@@ -987,17 +912,17 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
auto Top = ReconstructX87StateFromFSW(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
}
|
||||
|
||||
OrderedNode *ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
|
||||
OrderedNode* ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
@@ -1005,14 +930,14 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
OrderedNode *Mask = _VCastFromGPR(16, 8, low);
|
||||
OrderedNode* Mask = _VCastFromGPR(16, 8, low);
|
||||
Mask = _VInsGPR(16, 8, 1, Mask, high);
|
||||
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
OrderedNode *Reg = _LoadMem(FPRClass, 16, ST0Location, 1);
|
||||
OrderedNode* Reg = _LoadMem(FPRClass, 16, ST0Location, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(16, 16, Reg, Mask);
|
||||
//Convert to double precision
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(8, Reg);
|
||||
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
@@ -1025,20 +950,20 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
|
||||
OrderedNode *Reg = _LoadMem(FPRClass, 8, ST0Location, 1);
|
||||
OrderedNode* Reg = _LoadMem(FPRClass, 8, ST0Location, 1);
|
||||
ST0Location = _Add(OpSize::i64Bit, ST0Location, _Constant(8));
|
||||
OrderedNode *RegHigh = _LoadMem(FPRClass, 2, ST0Location, 1);
|
||||
OrderedNode* RegHigh = _LoadMem(FPRClass, 2, ST0Location, 1);
|
||||
Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh);
|
||||
Reg = _F80CVT(8, Reg); //Convert to double precision
|
||||
Reg = _F80CVT(8, Reg); // Convert to double precision
|
||||
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
|
||||
//FXAM needs change
|
||||
// FXAM needs change
|
||||
void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode *Result = _VExtractToGPR(8, 8, a, 0);
|
||||
OrderedNode* Result = _VExtractToGPR(8, 8, a, 0);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = _Bfe(OpSize::i64Bit, 1, 63, Result);
|
||||
@@ -1051,9 +976,7 @@ void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
// In the case of top being invalid then C3:C2:C0 is 0b101
|
||||
auto C3 = _Select(FEXCore::IR::COND_EQ,
|
||||
TopValid, OneConst,
|
||||
ZeroConst, OneConst);
|
||||
auto C3 = _Select(FEXCore::IR::COND_EQ, TopValid, OneConst, ZeroConst, OneConst);
|
||||
|
||||
auto C2 = TopValid;
|
||||
auto C0 = C3; // Mirror C3 until something other than zero is supported
|
||||
@@ -1063,4 +986,4 @@ void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -39,15 +39,15 @@ X86GeneratedCode::X86GeneratedCode() {
|
||||
// Falling back to this generated code segment still allows a backtrace to work, just might not show
|
||||
// the symbol as VDSO since there is no ELF to parse.
|
||||
constexpr std::array<uint8_t, 9> sigreturn_32_code = {
|
||||
0x58, // pop eax
|
||||
0x58, // pop eax
|
||||
0xb8, 0x77, 0x00, 0x00, 0x00, // mov eax, 0x77
|
||||
0xcd, 0x80, // int 0x80
|
||||
0x90, // nop
|
||||
0xcd, 0x80, // int 0x80
|
||||
0x90, // nop
|
||||
};
|
||||
|
||||
constexpr std::array<uint8_t, 7> rt_sigreturn_32_code = {
|
||||
0xb8, 0xad, 0x00, 0x00, 0x00, // mov eax, 0xad
|
||||
0xcd, 0x80, // int 0x80
|
||||
0xcd, 0x80, // int 0x80
|
||||
};
|
||||
|
||||
CallbackReturn = reinterpret_cast<uint64_t>(CodePtr);
|
||||
@@ -84,10 +84,9 @@ void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
|
||||
// We need to have the sigret handler in the lower 32bits of memory space
|
||||
// Scan top down and try to allocate a location
|
||||
for (size_t Location = 0xFFFF'E000; Location != 0x0; Location -= 0x1000) {
|
||||
void *Ptr = ::mmap(reinterpret_cast<void*>(Location), Size, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
void* Ptr = ::mmap(reinterpret_cast<void*>(Location), Size, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
if (Ptr != MAP_FAILED &&
|
||||
reinterpret_cast<uintptr_t>(Ptr) >= LOCATION_MAX) {
|
||||
if (Ptr != MAP_FAILED && reinterpret_cast<uintptr_t>(Ptr) >= LOCATION_MAX) {
|
||||
// Failed to map in the lower 32bits
|
||||
// Try again
|
||||
// Can happen in the case that host kernel ignores MAP_FIXED_NOREPLACE
|
||||
@@ -108,5 +107,4 @@ void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
|
||||
#endif
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -16,12 +16,12 @@ public:
|
||||
X86GeneratedCode();
|
||||
~X86GeneratedCode();
|
||||
|
||||
uint64_t CallbackReturn{};
|
||||
uint64_t sigreturn_32{};
|
||||
uint64_t rt_sigreturn_32{};
|
||||
uint64_t CallbackReturn {};
|
||||
uint64_t sigreturn_32 {};
|
||||
uint64_t rt_sigreturn_32 {};
|
||||
|
||||
private:
|
||||
void *CodePtr{};
|
||||
void* CodePtr {};
|
||||
void* AllocateGuestCodeSpace(size_t Size);
|
||||
};
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -24,4 +24,4 @@ void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeH0F3ATables(Mode);
|
||||
}
|
||||
|
||||
}
|
||||
} // namespace FEXCore::X86Tables
|
||||
@@ -162,7 +162,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"RDPID", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"RDPID", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_9, PF_66, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_9, PF_66, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
|
||||
@@ -14,9 +14,9 @@ extern "C" {
|
||||
enum jit_actions_t { JIT_NOACTION = 0, JIT_REGISTER_FN, JIT_UNREGISTER_FN };
|
||||
|
||||
struct jit_code_entry {
|
||||
jit_code_entry *next_entry;
|
||||
jit_code_entry *prev_entry;
|
||||
const char *symfile_addr;
|
||||
jit_code_entry* next_entry;
|
||||
jit_code_entry* prev_entry;
|
||||
const char* symfile_addr;
|
||||
uint64_t symfile_size;
|
||||
};
|
||||
|
||||
@@ -25,8 +25,8 @@ struct jit_descriptor {
|
||||
/* This type should be jit_actions_t, but we use uint32_t
|
||||
to be explicit about the bitwidth. */
|
||||
uint32_t action_flag;
|
||||
jit_code_entry *relevant_entry;
|
||||
jit_code_entry *first_entry;
|
||||
jit_code_entry* relevant_entry;
|
||||
jit_code_entry* first_entry;
|
||||
};
|
||||
|
||||
/* Make sure to specify the version statically, because the
|
||||
@@ -42,9 +42,8 @@ void __attribute__((noinline)) __jit_debug_register_code() {
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart,
|
||||
uint64_t GuestRIP, uintptr_t HostEntry,
|
||||
FEXCore::Core::DebugData *DebugData) {
|
||||
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
|
||||
FEXCore::Core::DebugData* DebugData) {
|
||||
auto map = Entry->SourcecodeMap.get();
|
||||
|
||||
if (map) {
|
||||
@@ -52,32 +51,28 @@ void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart,
|
||||
|
||||
auto Sym = map->FindSymbolMapping(FileOffset);
|
||||
|
||||
auto SymName = HLE::SourcecodeSymbolMapping::SymName(
|
||||
Sym, Entry->Filename, HostEntry, FileOffset);
|
||||
auto SymName = HLE::SourcecodeSymbolMapping::SymName(Sym, Entry->Filename, HostEntry, FileOffset);
|
||||
|
||||
fextl::vector<gdb_line_mapping> Lines;
|
||||
for (const auto &GuestOpcode : DebugData->GuestOpcodes) {
|
||||
auto Line = map->FindLineMapping(GuestRIP + GuestOpcode.GuestEntryOffset -
|
||||
VAFileStart);
|
||||
for (const auto& GuestOpcode : DebugData->GuestOpcodes) {
|
||||
auto Line = map->FindLineMapping(GuestRIP + GuestOpcode.GuestEntryOffset - VAFileStart);
|
||||
if (Line) {
|
||||
Lines.push_back(
|
||||
{Line->LineNumber, HostEntry + GuestOpcode.HostEntryOffset});
|
||||
Lines.push_back({Line->LineNumber, HostEntry + GuestOpcode.HostEntryOffset});
|
||||
}
|
||||
}
|
||||
|
||||
size_t size = sizeof(info_t) + 1 * sizeof(blocks_t) +
|
||||
Lines.size() * sizeof(gdb_line_mapping);
|
||||
size_t size = sizeof(info_t) + 1 * sizeof(blocks_t) + Lines.size() * sizeof(gdb_line_mapping);
|
||||
|
||||
auto mem = (uint8_t *)malloc(size);
|
||||
auto mem = (uint8_t*)malloc(size);
|
||||
auto base = mem;
|
||||
info_t *info = (info_t *)mem;
|
||||
info_t* info = (info_t*)mem;
|
||||
mem += sizeof(info_t);
|
||||
|
||||
strncpy(info->filename, map->SourceFile.c_str(), 511);
|
||||
|
||||
info->nblocks = 1;
|
||||
|
||||
auto blocks = (blocks_t *)mem;
|
||||
auto blocks = (blocks_t*)mem;
|
||||
info->blocks_ofs = mem - base;
|
||||
|
||||
mem += info->nblocks * sizeof(blocks_t);
|
||||
@@ -90,7 +85,7 @@ void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart,
|
||||
|
||||
info->nlines = Lines.size();
|
||||
|
||||
auto lines = (gdb_line_mapping *)mem;
|
||||
auto lines = (gdb_line_mapping*)mem;
|
||||
info->lines_ofs = mem - base;
|
||||
mem += info->nlines * sizeof(gdb_line_mapping);
|
||||
|
||||
@@ -98,9 +93,9 @@ void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart,
|
||||
memcpy(lines, &Lines.at(0), info->nlines * sizeof(gdb_line_mapping));
|
||||
}
|
||||
|
||||
auto entry = new jit_code_entry{0, 0, 0, 0};
|
||||
auto entry = new jit_code_entry {0, 0, 0, 0};
|
||||
|
||||
entry->symfile_addr = (const char *)info;
|
||||
entry->symfile_addr = (const char*)info;
|
||||
entry->symfile_size = size;
|
||||
|
||||
if (__jit_debug_descriptor.first_entry) {
|
||||
@@ -118,11 +113,8 @@ void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart,
|
||||
} // namespace FEXCore
|
||||
#else
|
||||
namespace FEXCore {
|
||||
void GDBJITRegister([[maybe_unused]] FEXCore::IR::AOTIRCacheEntry *Entry,
|
||||
[[maybe_unused]] uintptr_t VAFileStart,
|
||||
[[maybe_unused]] uint64_t GuestRIP,
|
||||
[[maybe_unused]] uintptr_t HostEntry,
|
||||
[[maybe_unused]] FEXCore::Core::DebugData *DebugData) {
|
||||
void GDBJITRegister([[maybe_unused]] FEXCore::IR::AOTIRCacheEntry* Entry, [[maybe_unused]] uintptr_t VAFileStart, [[maybe_unused]] uint64_t GuestRIP,
|
||||
[[maybe_unused]] uintptr_t HostEntry, [[maybe_unused]] FEXCore::Core::DebugData* DebugData) {
|
||||
ERROR_AND_DIE_FMT("GDBSymbols support not compiled in");
|
||||
}
|
||||
} // namespace FEXCore
|
||||
|
||||
@@ -4,5 +4,6 @@
|
||||
#include <Interface/IR/AOTIR.h>
|
||||
|
||||
namespace FEXCore {
|
||||
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry, FEXCore::Core::DebugData *DebugData);
|
||||
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
|
||||
FEXCore::Core::DebugData* DebugData);
|
||||
}
|
||||
@@ -6,13 +6,13 @@ tags: glue|thunks
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -39,19 +39,18 @@ extern "C" {
|
||||
#define JEMALLOC_NOTHROW __attribute__((nothrow))
|
||||
// Forward declare jemalloc functions because we can't include the headers from the glibc jemalloc project.
|
||||
// This is because we can't simultaneously set up include paths for both of our internal jemalloc modules.
|
||||
FEX_DEFAULT_VISIBILITY JEMALLOC_NOTHROW extern int glibc_je_is_known_allocation(void *ptr);
|
||||
FEX_DEFAULT_VISIBILITY JEMALLOC_NOTHROW extern int glibc_je_is_known_allocation(void* ptr);
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
static __attribute__((aligned(16), naked, section("HostToGuestTrampolineTemplate"))) void HostToGuestTrampolineTemplate() {
|
||||
#if defined(_M_X86_64)
|
||||
asm(
|
||||
"lea 0f(%rip), %r11 \n"
|
||||
"jmpq *0f(%rip) \n"
|
||||
".align 8 \n"
|
||||
"0: \n"
|
||||
".quad 0, 0, 0, 0 \n" // TrampolineInstanceInfo
|
||||
asm("lea 0f(%rip), %r11 \n"
|
||||
"jmpq *0f(%rip) \n"
|
||||
".align 8 \n"
|
||||
"0: \n"
|
||||
".quad 0, 0, 0, 0 \n" // TrampolineInstanceInfo
|
||||
);
|
||||
#elif defined(_M_ARM_64)
|
||||
asm(
|
||||
@@ -76,441 +75,431 @@ extern char __stop_HostToGuestTrampolineTemplate[];
|
||||
|
||||
namespace FEXCore {
|
||||
#ifndef _WIN32
|
||||
struct LoadlibArgs {
|
||||
const char *Name;
|
||||
struct LoadlibArgs {
|
||||
const char* Name;
|
||||
};
|
||||
|
||||
static thread_local FEXCore::Core::InternalThreadState* Thread = nullptr;
|
||||
|
||||
|
||||
struct ExportEntry {
|
||||
uint8_t* sha256;
|
||||
ThunkedFunction* Fn;
|
||||
};
|
||||
|
||||
struct TrampolineInstanceInfo {
|
||||
void* HostPacker;
|
||||
uintptr_t CallCallback;
|
||||
uintptr_t GuestUnpacker;
|
||||
uintptr_t GuestTarget;
|
||||
};
|
||||
|
||||
// Opaque type pointing to an instance of HostToGuestTrampolineTemplate and its
|
||||
// embedded TrampolineInstanceInfo
|
||||
struct HostToGuestTrampolinePtr;
|
||||
const auto HostToGuestTrampolineSize = __stop_HostToGuestTrampolineTemplate - __start_HostToGuestTrampolineTemplate;
|
||||
|
||||
static TrampolineInstanceInfo& GetInstanceInfo(HostToGuestTrampolinePtr* Trampoline) {
|
||||
const auto Length = __stop_HostToGuestTrampolineTemplate - __start_HostToGuestTrampolineTemplate;
|
||||
const auto InstanceInfoOffset = Length - sizeof(TrampolineInstanceInfo);
|
||||
return *reinterpret_cast<TrampolineInstanceInfo*>(reinterpret_cast<char*>(Trampoline) + InstanceInfoOffset);
|
||||
}
|
||||
|
||||
struct GuestcallInfo {
|
||||
uintptr_t GuestUnpacker;
|
||||
uintptr_t GuestTarget;
|
||||
|
||||
bool operator==(const GuestcallInfo&) const noexcept = default;
|
||||
};
|
||||
|
||||
struct GuestcallInfoHash {
|
||||
size_t operator()(const GuestcallInfo& x) const noexcept {
|
||||
// Hash only the target address, which is generally unique.
|
||||
// For the unlikely case of a hash collision, fextl::unordered_map still picks the correct bucket entry.
|
||||
return std::hash<uintptr_t> {}(x.GuestTarget);
|
||||
}
|
||||
};
|
||||
|
||||
// Bits in a SHA256 sum are already randomly distributed, so truncation yields a suitable hash function
|
||||
struct TruncatingSHA256Hash {
|
||||
size_t operator()(const FEXCore::IR::SHA256Sum& SHA256Sum) const noexcept {
|
||||
return (const size_t&)SHA256Sum;
|
||||
}
|
||||
};
|
||||
|
||||
HostToGuestTrampolinePtr* MakeHostTrampolineForGuestFunction(void* HostPacker, uintptr_t GuestTarget, uintptr_t GuestUnpacker);
|
||||
|
||||
struct ThunkHandler_impl final : public ThunkHandler {
|
||||
std::shared_mutex ThunksMutex;
|
||||
|
||||
fextl::unordered_map<IR::SHA256Sum, ThunkedFunction*, TruncatingSHA256Hash> Thunks = {
|
||||
{// sha256(fex:loadlib)
|
||||
{0x27, 0x7e, 0xb7, 0x69, 0x5b, 0xe9, 0xab, 0x12, 0x6e, 0xf7, 0x85, 0x9d, 0x4b, 0xc9, 0xa2, 0x44,
|
||||
0x46, 0xcf, 0xbd, 0xb5, 0x87, 0x43, 0xef, 0x28, 0xa2, 0x65, 0xba, 0xfc, 0x89, 0x0f, 0x77, 0x80},
|
||||
&LoadLib},
|
||||
{// sha256(fex:is_lib_loaded)
|
||||
{0xee, 0x57, 0xba, 0x0c, 0x5f, 0x6e, 0xef, 0x2a, 0x8c, 0xb5, 0x19, 0x81, 0xc9, 0x23, 0xe6, 0x51,
|
||||
0xae, 0x65, 0x02, 0x8f, 0x2b, 0x5d, 0x59, 0x90, 0x6a, 0x7e, 0xe2, 0xe7, 0x1c, 0x33, 0x8a, 0xff},
|
||||
&IsLibLoaded},
|
||||
{// sha256(fex:is_host_heap_allocation)
|
||||
{0xf5, 0x77, 0x68, 0x43, 0xbb, 0x6b, 0x28, 0x18, 0x40, 0xb0, 0xdb, 0x8a, 0x66, 0xfb, 0x0e, 0x2d,
|
||||
0x98, 0xc2, 0xad, 0xe2, 0x5a, 0x18, 0x5a, 0x37, 0x2e, 0x13, 0xc9, 0xe7, 0xb9, 0x8c, 0xa9, 0x3e},
|
||||
&IsHostHeapAllocation},
|
||||
{// sha256(fex:link_address_to_function)
|
||||
{0xe6, 0xa8, 0xec, 0x1c, 0x7b, 0x74, 0x35, 0x27, 0xe9, 0x4f, 0x5b, 0x6e, 0x2d, 0xc9, 0xa0, 0x27,
|
||||
0xd6, 0x1f, 0x2b, 0x87, 0x8f, 0x2d, 0x35, 0x50, 0xea, 0x16, 0xb8, 0xc4, 0x5e, 0x42, 0xfd, 0x77},
|
||||
&LinkAddressToGuestFunction},
|
||||
{// sha256(fex:allocate_host_trampoline_for_guest_function)
|
||||
{0x9b, 0xb2, 0xf4, 0xb4, 0x83, 0x7d, 0x28, 0x93, 0x40, 0xcb, 0xf4, 0x7a, 0x0b, 0x47, 0x85, 0x87,
|
||||
0xf9, 0xbc, 0xb5, 0x27, 0xca, 0xa6, 0x93, 0xa5, 0xc0, 0x73, 0x27, 0x24, 0xae, 0xc8, 0xb8, 0x5a},
|
||||
&AllocateHostTrampolineForGuestFunction},
|
||||
};
|
||||
|
||||
static thread_local FEXCore::Core::InternalThreadState *Thread = nullptr;
|
||||
// Can't be a string_view. We need to keep a copy of the library name in-case string_view pointer goes away.
|
||||
// Ideally we track when a library has been unloaded and remove it from this set before the memory backing goes away.
|
||||
fextl::set<fextl::string> Libs;
|
||||
|
||||
fextl::unordered_map<GuestcallInfo, HostToGuestTrampolinePtr*, GuestcallInfoHash> GuestcallToHostTrampoline;
|
||||
|
||||
uint8_t* HostTrampolineInstanceDataPtr;
|
||||
size_t HostTrampolineInstanceDataAvailable = 0;
|
||||
|
||||
|
||||
struct ExportEntry { uint8_t *sha256; ThunkedFunction* Fn; };
|
||||
|
||||
struct TrampolineInstanceInfo {
|
||||
void* HostPacker;
|
||||
uintptr_t CallCallback;
|
||||
uintptr_t GuestUnpacker;
|
||||
uintptr_t GuestTarget;
|
||||
};
|
||||
|
||||
// Opaque type pointing to an instance of HostToGuestTrampolineTemplate and its
|
||||
// embedded TrampolineInstanceInfo
|
||||
struct HostToGuestTrampolinePtr;
|
||||
const auto HostToGuestTrampolineSize = __stop_HostToGuestTrampolineTemplate - __start_HostToGuestTrampolineTemplate;
|
||||
|
||||
static TrampolineInstanceInfo& GetInstanceInfo(HostToGuestTrampolinePtr* Trampoline) {
|
||||
const auto Length = __stop_HostToGuestTrampolineTemplate - __start_HostToGuestTrampolineTemplate;
|
||||
const auto InstanceInfoOffset = Length - sizeof(TrampolineInstanceInfo);
|
||||
return *reinterpret_cast<TrampolineInstanceInfo*>(reinterpret_cast<char*>(Trampoline) + InstanceInfoOffset);
|
||||
/*
|
||||
Set arg0/1 to arg regs, use CTX::HandleCallback to handle the callback
|
||||
*/
|
||||
static void CallCallback(void* callback, void* arg0, void* arg1) {
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to invoke guest callback asynchronously");
|
||||
}
|
||||
|
||||
struct GuestcallInfo {
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = (uintptr_t)arg1;
|
||||
} else {
|
||||
if ((reinterpret_cast<uintptr_t>(arg1) >> 32) != 0) {
|
||||
ERROR_AND_DIE_FMT("Tried to call guest function with arguments packed to a 64-bit address");
|
||||
}
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RCX] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDX] = (uintptr_t)arg1;
|
||||
}
|
||||
|
||||
Thread->CTX->HandleCallback(Thread, (uintptr_t)callback);
|
||||
}
|
||||
|
||||
/**
|
||||
* Instructs the Core to redirect calls to functions at the given
|
||||
* address to another function. The original callee address is passed
|
||||
* to the target function through an implicit argument stored in r11.
|
||||
*
|
||||
* For 32-bit the implicit argument is stored in the lower 32-bits of mm0.
|
||||
*
|
||||
* The primary use case of this is ensuring that host function pointers
|
||||
* returned from thunked APIs can safely be called by the guest.
|
||||
*/
|
||||
static void LinkAddressToGuestFunction(void* argsv) {
|
||||
struct args_t {
|
||||
uintptr_t original_callee;
|
||||
uintptr_t target_addr; // Guest function to call when branching to original_callee
|
||||
};
|
||||
|
||||
auto args = reinterpret_cast<args_t*>(argsv);
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(args->original_callee, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_AA_FMT(args->target_addr, "Tried to link address to null pointer guest function");
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
LOGMAN_THROW_AA_FMT((args->original_callee >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_AA_FMT((args->target_addr >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}", args->original_callee, args->target_addr);
|
||||
|
||||
auto Result = CTX->AddCustomIREntrypoint(
|
||||
args->original_callee,
|
||||
[CTX, GuestThunkEntrypoint = args->target_addr](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) {
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0);
|
||||
auto Block = emit->CreateCodeNode();
|
||||
IRHeader.first->Blocks = emit->WrapNode(Block);
|
||||
emit->SetCurrentCodeBlock(Block);
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
if (GPRSize == 8) {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, gregs[X86State::REG_R11]), IR::GPRClass,
|
||||
IR::GPRFixedClass, GPRSize);
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(8, 8, emit->_Constant(Entrypoint)), offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
},
|
||||
CTX->ThunkHandler.get(), (void*)args->target_addr);
|
||||
|
||||
if (!Result) {
|
||||
if (Result.Creator != CTX->ThunkHandler.get()) {
|
||||
ERROR_AND_DIE_FMT("Input address for LinkAddressToGuestFunction is already linked by another module");
|
||||
}
|
||||
if (Result.Data != (void*)args->target_addr) {
|
||||
// NOTE: This may happen in Vulkan thunks if the Vulkan driver resolves two different symbols
|
||||
// to the same function (e.g. vkGetPhysicalDeviceFeatures2/vkGetPhysicalDeviceFeatures2KHR)
|
||||
LogMan::Msg::EFmt("Input address for LinkAddressToGuestFunction is already linked elsewhere");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Guest-side helper to initiate creation of a host trampoline for
|
||||
* calling guest functions. This must be followed by a host-side call
|
||||
* to FinalizeHostTrampolineForGuestFunction to make the trampoline
|
||||
* usable.
|
||||
*
|
||||
* This two-step initialization is equivalent to a host-side call to
|
||||
* MakeHostTrampolineForGuestFunction. The split is needed if the
|
||||
* host doesn't have all information needed to create the trampoline
|
||||
* on its own.
|
||||
*/
|
||||
static void AllocateHostTrampolineForGuestFunction(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
uintptr_t GuestUnpacker;
|
||||
uintptr_t GuestTarget;
|
||||
uintptr_t rv; // Pointer to host trampoline + TrampolineInstanceInfo
|
||||
}* args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
bool operator==(const GuestcallInfo&) const noexcept = default;
|
||||
};
|
||||
args->rv = (uintptr_t)MakeHostTrampolineForGuestFunction(nullptr, args->GuestTarget, args->GuestUnpacker);
|
||||
}
|
||||
|
||||
struct GuestcallInfoHash {
|
||||
size_t operator()(const GuestcallInfo& x) const noexcept {
|
||||
// Hash only the target address, which is generally unique.
|
||||
// For the unlikely case of a hash collision, fextl::unordered_map still picks the correct bucket entry.
|
||||
return std::hash<uintptr_t>{}(x.GuestTarget);
|
||||
}
|
||||
};
|
||||
|
||||
// Bits in a SHA256 sum are already randomly distributed, so truncation yields a suitable hash function
|
||||
struct TruncatingSHA256Hash {
|
||||
size_t operator()(const FEXCore::IR::SHA256Sum& SHA256Sum) const noexcept {
|
||||
return (const size_t&)SHA256Sum;
|
||||
}
|
||||
};
|
||||
|
||||
HostToGuestTrampolinePtr* MakeHostTrampolineForGuestFunction(void* HostPacker, uintptr_t GuestTarget, uintptr_t GuestUnpacker);
|
||||
|
||||
struct ThunkHandler_impl final: public ThunkHandler {
|
||||
std::shared_mutex ThunksMutex;
|
||||
|
||||
fextl::unordered_map<IR::SHA256Sum, ThunkedFunction*, TruncatingSHA256Hash> Thunks = {
|
||||
{
|
||||
// sha256(fex:loadlib)
|
||||
{ 0x27, 0x7e, 0xb7, 0x69, 0x5b, 0xe9, 0xab, 0x12, 0x6e, 0xf7, 0x85, 0x9d, 0x4b, 0xc9, 0xa2, 0x44, 0x46, 0xcf, 0xbd, 0xb5, 0x87, 0x43, 0xef, 0x28, 0xa2, 0x65, 0xba, 0xfc, 0x89, 0x0f, 0x77, 0x80 },
|
||||
&LoadLib
|
||||
},
|
||||
{
|
||||
// sha256(fex:is_lib_loaded)
|
||||
{ 0xee, 0x57, 0xba, 0x0c, 0x5f, 0x6e, 0xef, 0x2a, 0x8c, 0xb5, 0x19, 0x81, 0xc9, 0x23, 0xe6, 0x51, 0xae, 0x65, 0x02, 0x8f, 0x2b, 0x5d, 0x59, 0x90, 0x6a, 0x7e, 0xe2, 0xe7, 0x1c, 0x33, 0x8a, 0xff },
|
||||
&IsLibLoaded
|
||||
},
|
||||
{
|
||||
// sha256(fex:is_host_heap_allocation)
|
||||
{ 0xf5, 0x77, 0x68, 0x43, 0xbb, 0x6b, 0x28, 0x18, 0x40, 0xb0, 0xdb, 0x8a, 0x66, 0xfb, 0x0e, 0x2d, 0x98, 0xc2, 0xad, 0xe2, 0x5a, 0x18, 0x5a, 0x37, 0x2e, 0x13, 0xc9, 0xe7, 0xb9, 0x8c, 0xa9, 0x3e },
|
||||
&IsHostHeapAllocation
|
||||
},
|
||||
{
|
||||
// sha256(fex:link_address_to_function)
|
||||
{ 0xe6, 0xa8, 0xec, 0x1c, 0x7b, 0x74, 0x35, 0x27, 0xe9, 0x4f, 0x5b, 0x6e, 0x2d, 0xc9, 0xa0, 0x27, 0xd6, 0x1f, 0x2b, 0x87, 0x8f, 0x2d, 0x35, 0x50, 0xea, 0x16, 0xb8, 0xc4, 0x5e, 0x42, 0xfd, 0x77 },
|
||||
&LinkAddressToGuestFunction
|
||||
},
|
||||
{
|
||||
// sha256(fex:allocate_host_trampoline_for_guest_function)
|
||||
{ 0x9b, 0xb2, 0xf4, 0xb4, 0x83, 0x7d, 0x28, 0x93, 0x40, 0xcb, 0xf4, 0x7a, 0x0b, 0x47, 0x85, 0x87, 0xf9, 0xbc, 0xb5, 0x27, 0xca, 0xa6, 0x93, 0xa5, 0xc0, 0x73, 0x27, 0x24, 0xae, 0xc8, 0xb8, 0x5a },
|
||||
&AllocateHostTrampolineForGuestFunction
|
||||
},
|
||||
};
|
||||
|
||||
// Can't be a string_view. We need to keep a copy of the library name in-case string_view pointer goes away.
|
||||
// Ideally we track when a library has been unloaded and remove it from this set before the memory backing goes away.
|
||||
fextl::set<fextl::string> Libs;
|
||||
|
||||
fextl::unordered_map<GuestcallInfo, HostToGuestTrampolinePtr*, GuestcallInfoHash> GuestcallToHostTrampoline;
|
||||
|
||||
uint8_t *HostTrampolineInstanceDataPtr;
|
||||
size_t HostTrampolineInstanceDataAvailable = 0;
|
||||
|
||||
|
||||
/*
|
||||
Set arg0/1 to arg regs, use CTX::HandleCallback to handle the callback
|
||||
*/
|
||||
static void CallCallback(void *callback, void *arg0, void* arg1) {
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to invoke guest callback asynchronously");
|
||||
}
|
||||
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = (uintptr_t)arg1;
|
||||
} else {
|
||||
if ((reinterpret_cast<uintptr_t>(arg1) >> 32) != 0) {
|
||||
ERROR_AND_DIE_FMT("Tried to call guest function with arguments packed to a 64-bit address");
|
||||
}
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RCX] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDX] = (uintptr_t)arg1;
|
||||
}
|
||||
|
||||
Thread->CTX->HandleCallback(Thread, (uintptr_t)callback);
|
||||
}
|
||||
|
||||
/**
|
||||
* Instructs the Core to redirect calls to functions at the given
|
||||
* address to another function. The original callee address is passed
|
||||
* to the target function through an implicit argument stored in r11.
|
||||
*
|
||||
* For 32-bit the implicit argument is stored in the lower 32-bits of mm0.
|
||||
*
|
||||
* The primary use case of this is ensuring that host function pointers
|
||||
* returned from thunked APIs can safely be called by the guest.
|
||||
*/
|
||||
static void LinkAddressToGuestFunction(void* argsv) {
|
||||
struct args_t {
|
||||
uintptr_t original_callee;
|
||||
uintptr_t target_addr; // Guest function to call when branching to original_callee
|
||||
};
|
||||
|
||||
auto args = reinterpret_cast<args_t*>(argsv);
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(args->original_callee, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_AA_FMT(args->target_addr, "Tried to link address to null pointer guest function");
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
LOGMAN_THROW_AA_FMT((args->original_callee >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_AA_FMT((args->target_addr >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}",
|
||||
args->original_callee, args->target_addr);
|
||||
|
||||
auto Result = CTX->AddCustomIREntrypoint(
|
||||
args->original_callee,
|
||||
[CTX, GuestThunkEntrypoint = args->target_addr](uintptr_t Entrypoint, FEXCore::IR::IREmitter *emit) {
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0);
|
||||
auto Block = emit->CreateCodeNode();
|
||||
IRHeader.first->Blocks = emit->WrapNode(Block);
|
||||
emit->SetCurrentCodeBlock(Block);
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
if (GPRSize == 8) {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, gregs[X86State::REG_R11]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
}
|
||||
else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(8, 8, emit->_Constant(Entrypoint)), offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
}, CTX->ThunkHandler.get(), (void*)args->target_addr);
|
||||
|
||||
if (!Result) {
|
||||
if (Result.Creator != CTX->ThunkHandler.get()) {
|
||||
ERROR_AND_DIE_FMT("Input address for LinkAddressToGuestFunction is already linked by another module");
|
||||
}
|
||||
if (Result.Data != (void*)args->target_addr) {
|
||||
// NOTE: This may happen in Vulkan thunks if the Vulkan driver resolves two different symbols
|
||||
// to the same function (e.g. vkGetPhysicalDeviceFeatures2/vkGetPhysicalDeviceFeatures2KHR)
|
||||
LogMan::Msg::EFmt("Input address for LinkAddressToGuestFunction is already linked elsewhere");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Guest-side helper to initiate creation of a host trampoline for
|
||||
* calling guest functions. This must be followed by a host-side call
|
||||
* to FinalizeHostTrampolineForGuestFunction to make the trampoline
|
||||
* usable.
|
||||
*
|
||||
* This two-step initialization is equivalent to a host-side call to
|
||||
* MakeHostTrampolineForGuestFunction. The split is needed if the
|
||||
* host doesn't have all information needed to create the trampoline
|
||||
* on its own.
|
||||
*/
|
||||
static void AllocateHostTrampolineForGuestFunction(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
uintptr_t GuestUnpacker;
|
||||
uintptr_t GuestTarget;
|
||||
uintptr_t rv; // Pointer to host trampoline + TrampolineInstanceInfo
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
args->rv = (uintptr_t)MakeHostTrampolineForGuestFunction(nullptr, args->GuestTarget, args->GuestUnpacker);
|
||||
}
|
||||
|
||||
/**
|
||||
* Checks if the given pointer is allocated on the host heap.
|
||||
*
|
||||
* This is useful for thunking APIs that need to work with both guest
|
||||
* and host heap pointers.
|
||||
*/
|
||||
static void IsHostHeapAllocation(void* ArgsRV) {
|
||||
/**
|
||||
* Checks if the given pointer is allocated on the host heap.
|
||||
*
|
||||
* This is useful for thunking APIs that need to work with both guest
|
||||
* and host heap pointers.
|
||||
*/
|
||||
static void IsHostHeapAllocation(void* ArgsRV) {
|
||||
#ifdef ENABLE_JEMALLOC_GLIBC
|
||||
struct ArgsRV_t {
|
||||
void* ptr;
|
||||
bool rv;
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
struct ArgsRV_t {
|
||||
void* ptr;
|
||||
bool rv;
|
||||
}* args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
args->rv = glibc_je_is_known_allocation(args->ptr);
|
||||
args->rv = glibc_je_is_known_allocation(args->ptr);
|
||||
#else
|
||||
// Thunks usage without jemalloc isn't supported
|
||||
ERROR_AND_DIE_FMT("Unsupported: Thunks querying for host heap allocation information");
|
||||
// Thunks usage without jemalloc isn't supported
|
||||
ERROR_AND_DIE_FMT("Unsupported: Thunks querying for host heap allocation information");
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
static void LoadLib(void *ArgsV) {
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
static void LoadLib(void* ArgsV) {
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
|
||||
auto Args = reinterpret_cast<LoadlibArgs*>(ArgsV);
|
||||
auto Args = reinterpret_cast<LoadlibArgs*>(ArgsV);
|
||||
|
||||
std::string_view Name = Args->Name;
|
||||
std::string_view Name = Args->Name;
|
||||
|
||||
auto SOName = (CTX->Config.Is64BitMode() ?
|
||||
CTX->Config.ThunkHostLibsPath() :
|
||||
CTX->Config.ThunkHostLibsPath32())
|
||||
+ "/" + Name.data() + "-host.so";
|
||||
auto SOName = (CTX->Config.Is64BitMode() ? CTX->Config.ThunkHostLibsPath() : CTX->Config.ThunkHostLibsPath32()) + "/" + Name.data() + "-host.so";
|
||||
|
||||
LogMan::Msg::DFmt("LoadLib: {} -> {}", Name, SOName);
|
||||
LogMan::Msg::DFmt("LoadLib: {} -> {}", Name, SOName);
|
||||
|
||||
auto Handle = dlopen(SOName.c_str(), RTLD_LOCAL | RTLD_NOW);
|
||||
if (!Handle) {
|
||||
ERROR_AND_DIE_FMT("LoadLib: Failed to dlopen thunk library {}: {}", SOName, dlerror());
|
||||
}
|
||||
auto Handle = dlopen(SOName.c_str(), RTLD_LOCAL | RTLD_NOW);
|
||||
if (!Handle) {
|
||||
ERROR_AND_DIE_FMT("LoadLib: Failed to dlopen thunk library {}: {}", SOName, dlerror());
|
||||
}
|
||||
|
||||
// Library names often include dashes, which may not be used in C++ identifiers.
|
||||
// They are replaced with underscores hence.
|
||||
auto InitSym = "fexthunks_exports_" + fextl::string { Name };
|
||||
std::replace(InitSym.begin(), InitSym.end(), '-', '_');
|
||||
// Library names often include dashes, which may not be used in C++ identifiers.
|
||||
// They are replaced with underscores hence.
|
||||
auto InitSym = "fexthunks_exports_" + fextl::string {Name};
|
||||
std::replace(InitSym.begin(), InitSym.end(), '-', '_');
|
||||
|
||||
ExportEntry* (*InitFN)();
|
||||
(void*&)InitFN = dlsym(Handle, InitSym.c_str());
|
||||
if (!InitFN) {
|
||||
ERROR_AND_DIE_FMT("LoadLib: Failed to find export {}", InitSym);
|
||||
}
|
||||
ExportEntry* (*InitFN)();
|
||||
(void*&)InitFN = dlsym(Handle, InitSym.c_str());
|
||||
if (!InitFN) {
|
||||
ERROR_AND_DIE_FMT("LoadLib: Failed to find export {}", InitSym);
|
||||
}
|
||||
|
||||
auto Exports = InitFN();
|
||||
if (!Exports) {
|
||||
ERROR_AND_DIE_FMT("LoadLib: Failed to initialize thunk library {}. "
|
||||
"Check if the corresponding host library is installed "
|
||||
"or disable thunking of this library.", Name);
|
||||
}
|
||||
auto Exports = InitFN();
|
||||
if (!Exports) {
|
||||
ERROR_AND_DIE_FMT("LoadLib: Failed to initialize thunk library {}. "
|
||||
"Check if the corresponding host library is installed "
|
||||
"or disable thunking of this library.",
|
||||
Name);
|
||||
}
|
||||
|
||||
auto That = reinterpret_cast<ThunkHandler_impl*>(CTX->ThunkHandler.get());
|
||||
auto That = reinterpret_cast<ThunkHandler_impl*>(CTX->ThunkHandler.get());
|
||||
|
||||
{
|
||||
std::lock_guard lk(That->ThunksMutex);
|
||||
{
|
||||
std::lock_guard lk(That->ThunksMutex);
|
||||
|
||||
That->Libs.insert(fextl::string { Name });
|
||||
That->Libs.insert(fextl::string {Name});
|
||||
|
||||
int i;
|
||||
for (i = 0; Exports[i].sha256; i++) {
|
||||
That->Thunks[*reinterpret_cast<IR::SHA256Sum*>(Exports[i].sha256)] = Exports[i].Fn;
|
||||
}
|
||||
int i;
|
||||
for (i = 0; Exports[i].sha256; i++) {
|
||||
That->Thunks[*reinterpret_cast<IR::SHA256Sum*>(Exports[i].sha256)] = Exports[i].Fn;
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Loaded {} syms", i);
|
||||
}
|
||||
}
|
||||
LogMan::Msg::DFmt("Loaded {} syms", i);
|
||||
}
|
||||
}
|
||||
|
||||
static void IsLibLoaded(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
const char *Name;
|
||||
bool rv;
|
||||
};
|
||||
|
||||
auto &[Name, rv] = *reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
auto That = reinterpret_cast<ThunkHandler_impl*>(CTX->ThunkHandler.get());
|
||||
|
||||
{
|
||||
std::shared_lock lk(That->ThunksMutex);
|
||||
rv = That->Libs.contains(Name);
|
||||
}
|
||||
}
|
||||
|
||||
ThunkedFunction* LookupThunk(const IR::SHA256Sum &sha256) override {
|
||||
|
||||
std::shared_lock lk(ThunksMutex);
|
||||
|
||||
auto it = Thunks.find(sha256);
|
||||
|
||||
if (it != Thunks.end()) {
|
||||
return it->second;
|
||||
} else {
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
void RegisterTLSState(FEXCore::Core::InternalThreadState *_Thread) override {
|
||||
Thread = _Thread;
|
||||
}
|
||||
|
||||
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override {
|
||||
for (auto & Definition : Definitions) {
|
||||
Thunks.emplace(Definition.Sum, Definition.ThunkFunction);
|
||||
}
|
||||
}
|
||||
static void IsLibLoaded(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
const char* Name;
|
||||
bool rv;
|
||||
};
|
||||
|
||||
fextl::unique_ptr<ThunkHandler> ThunkHandler::Create() {
|
||||
return fextl::make_unique<ThunkHandler_impl>();
|
||||
auto& [Name, rv] = *reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
auto That = reinterpret_cast<ThunkHandler_impl*>(CTX->ThunkHandler.get());
|
||||
|
||||
{
|
||||
std::shared_lock lk(That->ThunksMutex);
|
||||
rv = That->Libs.contains(Name);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates a host-callable trampoline to call guest functions via the host ABI.
|
||||
*
|
||||
* This trampoline uses the same calling convention as the given HostPacker. Trampolines
|
||||
* are cached, so it's safe to call this function repeatedly on the same arguments without
|
||||
* leaking memory.
|
||||
*
|
||||
* Invoking the returned trampoline has the effect of:
|
||||
* - packing the arguments (using the HostPacker identified by its SHA256)
|
||||
* - performing a host->guest transition
|
||||
* - unpacking the arguments via GuestUnpacker
|
||||
* - calling the function at GuestTarget
|
||||
*
|
||||
* The primary use case of this is ensuring that guest function pointers ("callbacks")
|
||||
* passed to thunked APIs can safely be called by the native host library.
|
||||
*
|
||||
* Returns a pointer to the generated host trampoline and its TrampolineInstanceInfo.
|
||||
*
|
||||
* If HostPacker is zero, the trampoline will be partially initialized and needs to be
|
||||
* finalized with a call to FinalizeHostTrampolineForGuestFunction. A typical use case
|
||||
* is to allocate the trampoline for a given GuestTarget/GuestUnpacker on the guest-side,
|
||||
* and provide the HostPacker host-side.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY
|
||||
HostToGuestTrampolinePtr* MakeHostTrampolineForGuestFunction(void* HostPacker, uintptr_t GuestTarget, uintptr_t GuestUnpacker) {
|
||||
LOGMAN_THROW_AA_FMT(GuestTarget, "Tried to create host-trampoline to null pointer guest function");
|
||||
ThunkedFunction* LookupThunk(const IR::SHA256Sum& sha256) override {
|
||||
|
||||
const auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
const auto ThunkHandler = reinterpret_cast<ThunkHandler_impl *>(CTX->ThunkHandler.get());
|
||||
std::shared_lock lk(ThunksMutex);
|
||||
|
||||
const GuestcallInfo gci = { GuestUnpacker, GuestTarget };
|
||||
auto it = Thunks.find(sha256);
|
||||
|
||||
// Try first with shared_lock
|
||||
{
|
||||
std::shared_lock lk(ThunkHandler->ThunksMutex);
|
||||
|
||||
auto found = ThunkHandler->GuestcallToHostTrampoline.find(gci);
|
||||
if (found != ThunkHandler->GuestcallToHostTrampoline.end()) {
|
||||
return found->second;
|
||||
}
|
||||
}
|
||||
|
||||
std::lock_guard lk(ThunkHandler->ThunksMutex);
|
||||
|
||||
// Retry lookup with full lock before making a new trampoline to avoid double trampolines
|
||||
{
|
||||
auto found = ThunkHandler->GuestcallToHostTrampoline.find(gci);
|
||||
if (found != ThunkHandler->GuestcallToHostTrampoline.end()) {
|
||||
return found->second;
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding host trampoline for guest function {:#x} via unpacker {:#x}",
|
||||
GuestTarget, GuestUnpacker);
|
||||
|
||||
if (ThunkHandler->HostTrampolineInstanceDataAvailable < HostToGuestTrampolineSize) {
|
||||
const auto allocation_step = 16 * 1024;
|
||||
ThunkHandler->HostTrampolineInstanceDataAvailable = allocation_step;
|
||||
ThunkHandler->HostTrampolineInstanceDataPtr = (uint8_t *)mmap(
|
||||
0, ThunkHandler->HostTrampolineInstanceDataAvailable,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ThunkHandler->HostTrampolineInstanceDataPtr != MAP_FAILED, "Failed to mmap HostTrampolineInstanceDataPtr");
|
||||
}
|
||||
|
||||
auto HostTrampoline = reinterpret_cast<HostToGuestTrampolinePtr* const>(ThunkHandler->HostTrampolineInstanceDataPtr);
|
||||
ThunkHandler->HostTrampolineInstanceDataAvailable -= HostToGuestTrampolineSize;
|
||||
ThunkHandler->HostTrampolineInstanceDataPtr += HostToGuestTrampolineSize;
|
||||
memcpy(HostTrampoline, (void*)&HostToGuestTrampolineTemplate, HostToGuestTrampolineSize);
|
||||
GetInstanceInfo(HostTrampoline) = TrampolineInstanceInfo {
|
||||
.HostPacker = HostPacker,
|
||||
.CallCallback = (uintptr_t)&ThunkHandler_impl::CallCallback,
|
||||
.GuestUnpacker = GuestUnpacker,
|
||||
.GuestTarget = GuestTarget
|
||||
};
|
||||
|
||||
ThunkHandler->GuestcallToHostTrampoline[gci] = HostTrampoline;
|
||||
return HostTrampoline;
|
||||
if (it != Thunks.end()) {
|
||||
return it->second;
|
||||
} else {
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY
|
||||
void FinalizeHostTrampolineForGuestFunction(HostToGuestTrampolinePtr* TrampolineAddress, void* HostPacker) {
|
||||
void RegisterTLSState(FEXCore::Core::InternalThreadState* _Thread) override {
|
||||
Thread = _Thread;
|
||||
}
|
||||
|
||||
if (TrampolineAddress == nullptr) return;
|
||||
|
||||
auto& Trampoline = GetInstanceInfo(TrampolineAddress);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Trampoline.CallCallback == (uintptr_t)&ThunkHandler_impl::CallCallback,
|
||||
"Invalid trampoline at {} passed to {}", fmt::ptr(TrampolineAddress), __FUNCTION__);
|
||||
|
||||
if (!Trampoline.HostPacker) {
|
||||
LogMan::Msg::DFmt("Thunks: Finalizing trampoline at {} with host packer {}", fmt::ptr(TrampolineAddress), fmt::ptr(HostPacker));
|
||||
Trampoline.HostPacker = HostPacker;
|
||||
}
|
||||
void AppendThunkDefinitions(const fextl::vector<FEXCore::IR::ThunkDefinition>& Definitions) override {
|
||||
for (auto& Definition : Definitions) {
|
||||
Thunks.emplace(Definition.Sum, Definition.ThunkFunction);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void* GetGuestStack() {
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to query guest stack pointer asynchronously");
|
||||
}
|
||||
fextl::unique_ptr<ThunkHandler> ThunkHandler::Create() {
|
||||
return fextl::make_unique<ThunkHandler_impl>();
|
||||
}
|
||||
|
||||
return (void*)(uintptr_t)((Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP]));
|
||||
/**
|
||||
* Generates a host-callable trampoline to call guest functions via the host ABI.
|
||||
*
|
||||
* This trampoline uses the same calling convention as the given HostPacker. Trampolines
|
||||
* are cached, so it's safe to call this function repeatedly on the same arguments without
|
||||
* leaking memory.
|
||||
*
|
||||
* Invoking the returned trampoline has the effect of:
|
||||
* - packing the arguments (using the HostPacker identified by its SHA256)
|
||||
* - performing a host->guest transition
|
||||
* - unpacking the arguments via GuestUnpacker
|
||||
* - calling the function at GuestTarget
|
||||
*
|
||||
* The primary use case of this is ensuring that guest function pointers ("callbacks")
|
||||
* passed to thunked APIs can safely be called by the native host library.
|
||||
*
|
||||
* Returns a pointer to the generated host trampoline and its TrampolineInstanceInfo.
|
||||
*
|
||||
* If HostPacker is zero, the trampoline will be partially initialized and needs to be
|
||||
* finalized with a call to FinalizeHostTrampolineForGuestFunction. A typical use case
|
||||
* is to allocate the trampoline for a given GuestTarget/GuestUnpacker on the guest-side,
|
||||
* and provide the HostPacker host-side.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY HostToGuestTrampolinePtr*
|
||||
MakeHostTrampolineForGuestFunction(void* HostPacker, uintptr_t GuestTarget, uintptr_t GuestUnpacker) {
|
||||
LOGMAN_THROW_AA_FMT(GuestTarget, "Tried to create host-trampoline to null pointer guest function");
|
||||
|
||||
const auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
const auto ThunkHandler = reinterpret_cast<ThunkHandler_impl*>(CTX->ThunkHandler.get());
|
||||
|
||||
const GuestcallInfo gci = {GuestUnpacker, GuestTarget};
|
||||
|
||||
// Try first with shared_lock
|
||||
{
|
||||
std::shared_lock lk(ThunkHandler->ThunksMutex);
|
||||
|
||||
auto found = ThunkHandler->GuestcallToHostTrampoline.find(gci);
|
||||
if (found != ThunkHandler->GuestcallToHostTrampoline.end()) {
|
||||
return found->second;
|
||||
}
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void MoveGuestStack(uintptr_t NewAddress) {
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to query guest stack pointer asynchronously");
|
||||
}
|
||||
std::lock_guard lk(ThunkHandler->ThunksMutex);
|
||||
|
||||
if (NewAddress >> 32) {
|
||||
ERROR_AND_DIE_FMT("Tried to set stack pointer for 32-bit guest to a 64-bit address");
|
||||
}
|
||||
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP] = NewAddress;
|
||||
// Retry lookup with full lock before making a new trampoline to avoid double trampolines
|
||||
{
|
||||
auto found = ThunkHandler->GuestcallToHostTrampoline.find(gci);
|
||||
if (found != ThunkHandler->GuestcallToHostTrampoline.end()) {
|
||||
return found->second;
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding host trampoline for guest function {:#x} via unpacker {:#x}", GuestTarget, GuestUnpacker);
|
||||
|
||||
if (ThunkHandler->HostTrampolineInstanceDataAvailable < HostToGuestTrampolineSize) {
|
||||
const auto allocation_step = 16 * 1024;
|
||||
ThunkHandler->HostTrampolineInstanceDataAvailable = allocation_step;
|
||||
ThunkHandler->HostTrampolineInstanceDataPtr = (uint8_t*)mmap(0, ThunkHandler->HostTrampolineInstanceDataAvailable,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ThunkHandler->HostTrampolineInstanceDataPtr != MAP_FAILED, "Failed to mmap HostTrampolineInstanceDataPtr");
|
||||
}
|
||||
|
||||
auto HostTrampoline = reinterpret_cast<HostToGuestTrampolinePtr* const>(ThunkHandler->HostTrampolineInstanceDataPtr);
|
||||
ThunkHandler->HostTrampolineInstanceDataAvailable -= HostToGuestTrampolineSize;
|
||||
ThunkHandler->HostTrampolineInstanceDataPtr += HostToGuestTrampolineSize;
|
||||
memcpy(HostTrampoline, (void*)&HostToGuestTrampolineTemplate, HostToGuestTrampolineSize);
|
||||
GetInstanceInfo(HostTrampoline) = TrampolineInstanceInfo {
|
||||
.HostPacker = HostPacker, .CallCallback = (uintptr_t)&ThunkHandler_impl::CallCallback, .GuestUnpacker = GuestUnpacker, .GuestTarget = GuestTarget};
|
||||
|
||||
ThunkHandler->GuestcallToHostTrampoline[gci] = HostTrampoline;
|
||||
return HostTrampoline;
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void FinalizeHostTrampolineForGuestFunction(HostToGuestTrampolinePtr* TrampolineAddress, void* HostPacker) {
|
||||
|
||||
if (TrampolineAddress == nullptr) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto& Trampoline = GetInstanceInfo(TrampolineAddress);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Trampoline.CallCallback == (uintptr_t)&ThunkHandler_impl::CallCallback, "Invalid trampoline at {} passed to {}",
|
||||
fmt::ptr(TrampolineAddress), __FUNCTION__);
|
||||
|
||||
if (!Trampoline.HostPacker) {
|
||||
LogMan::Msg::DFmt("Thunks: Finalizing trampoline at {} with host packer {}", fmt::ptr(TrampolineAddress), fmt::ptr(HostPacker));
|
||||
Trampoline.HostPacker = HostPacker;
|
||||
}
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void* GetGuestStack() {
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to query guest stack pointer asynchronously");
|
||||
}
|
||||
|
||||
return (void*)(uintptr_t)((Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP]));
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void MoveGuestStack(uintptr_t NewAddress) {
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to query guest stack pointer asynchronously");
|
||||
}
|
||||
|
||||
if (NewAddress >> 32) {
|
||||
ERROR_AND_DIE_FMT("Tried to set stack pointer for 32-bit guest to a 64-bit address");
|
||||
}
|
||||
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP] = NewAddress;
|
||||
}
|
||||
|
||||
#else
|
||||
fextl::unique_ptr<ThunkHandler> ThunkHandler::Create() {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
fextl::unique_ptr<ThunkHandler> ThunkHandler::Create() {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#endif
|
||||
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -7,33 +7,34 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct SHA256Sum;
|
||||
struct SHA256Sum;
|
||||
}
|
||||
|
||||
namespace FEXCore {
|
||||
typedef void ThunkedFunction(void* ArgsRv);
|
||||
typedef void ThunkedFunction(void* ArgsRv);
|
||||
|
||||
class ThunkHandler {
|
||||
public:
|
||||
virtual ThunkedFunction* LookupThunk(const IR::SHA256Sum &sha256) = 0;
|
||||
virtual void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
virtual ~ThunkHandler() { }
|
||||
class ThunkHandler {
|
||||
public:
|
||||
virtual ThunkedFunction* LookupThunk(const IR::SHA256Sum& sha256) = 0;
|
||||
virtual void RegisterTLSState(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
virtual ~ThunkHandler() {}
|
||||
|
||||
static fextl::unique_ptr<ThunkHandler> Create();
|
||||
static fextl::unique_ptr<ThunkHandler> Create();
|
||||
|
||||
virtual void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) = 0;
|
||||
};
|
||||
virtual void AppendThunkDefinitions(const fextl::vector<FEXCore::IR::ThunkDefinition>& Definitions) = 0;
|
||||
};
|
||||
}; // namespace FEXCore
|
||||
@@ -2,9 +2,9 @@
|
||||
#include "FEXHeaderUtils/Filesystem.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
@@ -22,399 +22,399 @@
|
||||
|
||||
|
||||
namespace FEXCore::IR {
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::GetInlineEntry(uint64_t DataOffset) {
|
||||
uintptr_t This = (uintptr_t)this;
|
||||
AOTIRInlineEntry* AOTIRInlineIndex::GetInlineEntry(uint64_t DataOffset) {
|
||||
uintptr_t This = (uintptr_t)this;
|
||||
|
||||
return (AOTIRInlineEntry*)(This + DataBase + DataOffset);
|
||||
}
|
||||
return (AOTIRInlineEntry*)(This + DataBase + DataOffset);
|
||||
}
|
||||
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::Find(uint64_t GuestStart) {
|
||||
ssize_t l = 0;
|
||||
ssize_t r = Count - 1;
|
||||
AOTIRInlineEntry* AOTIRInlineIndex::Find(uint64_t GuestStart) {
|
||||
ssize_t l = 0;
|
||||
ssize_t r = Count - 1;
|
||||
|
||||
while (l <= r) {
|
||||
size_t m = l + (r - l) / 2;
|
||||
while (l <= r) {
|
||||
size_t m = l + (r - l) / 2;
|
||||
|
||||
if (Entries[m].GuestStart == GuestStart)
|
||||
return GetInlineEntry(Entries[m].DataOffset);
|
||||
else if (Entries[m].GuestStart < GuestStart)
|
||||
l = m + 1;
|
||||
else
|
||||
r = m - 1;
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
IR::RegisterAllocationData *AOTIRInlineEntry::GetRAData() {
|
||||
return (IR::RegisterAllocationData *)InlineData;
|
||||
}
|
||||
|
||||
IR::IRListView *AOTIRInlineEntry::GetIRData() {
|
||||
auto RAData = GetRAData();
|
||||
auto Offset = RAData->Size(RAData->MapCount);
|
||||
|
||||
return (IR::IRListView *)&InlineData[Offset];
|
||||
}
|
||||
|
||||
void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
auto Inserted = Index.emplace(GuestRIP, Stream->Offset());
|
||||
|
||||
if (Inserted.second) {
|
||||
//GuestHash
|
||||
Stream->Write((const char*)&Hash, sizeof(Hash));
|
||||
|
||||
//GuestLength
|
||||
Stream->Write((const char*)&Length, sizeof(Length));
|
||||
|
||||
RAData->Serialize(*Stream);
|
||||
|
||||
// IRData (inline)
|
||||
IRList->Serialize(*Stream);
|
||||
if (Entries[m].GuestStart == GuestStart) {
|
||||
return GetInlineEntry(Entries[m].DataOffset);
|
||||
} else if (Entries[m].GuestStart < GuestStart) {
|
||||
l = m + 1;
|
||||
} else {
|
||||
r = m - 1;
|
||||
}
|
||||
}
|
||||
|
||||
static bool readAll(int fd, void *data, size_t size) {
|
||||
int rv = read(fd, data, size);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if (rv != size)
|
||||
return false;
|
||||
else
|
||||
return true;
|
||||
IR::RegisterAllocationData* AOTIRInlineEntry::GetRAData() {
|
||||
return (IR::RegisterAllocationData*)InlineData;
|
||||
}
|
||||
|
||||
IR::IRListView* AOTIRInlineEntry::GetIRData() {
|
||||
auto RAData = GetRAData();
|
||||
auto Offset = RAData->Size(RAData->MapCount);
|
||||
|
||||
return (IR::IRListView*)&InlineData[Offset];
|
||||
}
|
||||
|
||||
void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash,
|
||||
FEXCore::IR::IRListView* IRList, FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
auto Inserted = Index.emplace(GuestRIP, Stream->Offset());
|
||||
|
||||
if (Inserted.second) {
|
||||
// GuestHash
|
||||
Stream->Write((const char*)&Hash, sizeof(Hash));
|
||||
|
||||
// GuestLength
|
||||
Stream->Write((const char*)&Length, sizeof(Length));
|
||||
|
||||
RAData->Serialize(*Stream);
|
||||
|
||||
// IRData (inline)
|
||||
IRList->Serialize(*Stream);
|
||||
}
|
||||
}
|
||||
|
||||
static bool LoadAOTIRCache(AOTIRCacheEntry *Entry, int streamfd) {
|
||||
#ifndef _WIN32
|
||||
uint64_t tag;
|
||||
static bool readAll(int fd, void* data, size_t size) {
|
||||
int rv = read(fd, data, size);
|
||||
|
||||
if (!readAll(streamfd, (char*)&tag, sizeof(tag)) || tag != FEXCore::IR::AOTIR_COOKIE)
|
||||
return false;
|
||||
|
||||
fextl::string Module;
|
||||
uint64_t ModSize;
|
||||
uint64_t IndexSize;
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&ModSize, sizeof(ModSize)))
|
||||
return false;
|
||||
|
||||
Module.resize(ModSize);
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize, SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&Module[0], Module.size()))
|
||||
return false;
|
||||
|
||||
if (Entry->FileId != Module) {
|
||||
return false;
|
||||
}
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize - sizeof(IndexSize), SEEK_END);
|
||||
if (rv != size) {
|
||||
return false;
|
||||
} else {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
if (!readAll(streamfd, (char*)&IndexSize, sizeof(IndexSize)))
|
||||
return false;
|
||||
static bool LoadAOTIRCache(AOTIRCacheEntry* Entry, int streamfd) {
|
||||
#ifndef _WIN32
|
||||
uint64_t tag;
|
||||
|
||||
struct stat fileinfo;
|
||||
if (fstat(streamfd, &fileinfo) < 0)
|
||||
return false;
|
||||
size_t Size = (fileinfo.st_size + 4095) & ~4095;
|
||||
|
||||
size_t IndexOffset = fileinfo.st_size - IndexSize -sizeof(ModSize) - ModSize - sizeof(IndexSize);
|
||||
|
||||
void *FilePtr = FEXCore::Allocator::mmap(nullptr, Size, PROT_READ, MAP_SHARED, streamfd, 0);
|
||||
|
||||
if (FilePtr == MAP_FAILED) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Array = (AOTIRInlineIndex *)((char*)FilePtr + IndexOffset);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr && Entry->FilePtr == nullptr, "Entry must not be initialized here");
|
||||
Entry->Array = Array;
|
||||
Entry->FilePtr = FilePtr;
|
||||
Entry->Size = Size;
|
||||
|
||||
LogMan::Msg::DFmt("AOTIR: Module {} has {} functions", Module, Array->Count);
|
||||
|
||||
return true;
|
||||
#else
|
||||
if (!readAll(streamfd, (char*)&tag, sizeof(tag)) || tag != FEXCore::IR::AOTIR_COOKIE) {
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::FinalizeAOTIRCache() {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
fextl::string Module;
|
||||
uint64_t ModSize;
|
||||
uint64_t IndexSize;
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
lseek(streamfd, -sizeof(ModSize), SEEK_END);
|
||||
|
||||
for (auto& [String, Entry] : AOTIRCaptureCacheMap) {
|
||||
if (!Entry.Stream) {
|
||||
continue;
|
||||
}
|
||||
if (!readAll(streamfd, (char*)&ModSize, sizeof(ModSize))) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto ModSize = String.size();
|
||||
auto &stream = Entry.Stream;
|
||||
Module.resize(ModSize);
|
||||
|
||||
// pad to 32 bytes
|
||||
constexpr char Zero = 0;
|
||||
while(stream->Offset() & 31)
|
||||
stream->Write(&Zero, 1);
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize, SEEK_END);
|
||||
|
||||
// AOTIRInlineIndex
|
||||
const auto FnCount = Entry.Index.size();
|
||||
const size_t DataBase = -stream->Offset();
|
||||
if (!readAll(streamfd, (char*)&Module[0], Module.size())) {
|
||||
return false;
|
||||
}
|
||||
|
||||
stream->Write((const char*)&FnCount, sizeof(FnCount));
|
||||
stream->Write((const char*)&DataBase, sizeof(DataBase));
|
||||
if (Entry->FileId != Module) {
|
||||
return false;
|
||||
}
|
||||
|
||||
for (const auto& [GuestStart, DataOffset] : Entry.Index) {
|
||||
//AOTIRInlineIndexEntry
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize - sizeof(IndexSize), SEEK_END);
|
||||
|
||||
// GuestStart
|
||||
stream->Write((const char*)&GuestStart, sizeof(GuestStart));
|
||||
if (!readAll(streamfd, (char*)&IndexSize, sizeof(IndexSize))) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// DataOffset
|
||||
stream->Write((const char*)&DataOffset, sizeof(DataOffset));
|
||||
}
|
||||
struct stat fileinfo;
|
||||
if (fstat(streamfd, &fileinfo) < 0) {
|
||||
return false;
|
||||
}
|
||||
size_t Size = (fileinfo.st_size + 4095) & ~4095;
|
||||
|
||||
// End of file header
|
||||
const auto IndexSize = FnCount * sizeof(FEXCore::IR::AOTIRInlineIndexEntry) + sizeof(DataBase) + sizeof(FnCount);
|
||||
stream->Write((const char*)&IndexSize, sizeof(IndexSize));
|
||||
stream->Write(String.c_str(), ModSize);
|
||||
stream->Write((const char*)&ModSize, sizeof(ModSize));
|
||||
size_t IndexOffset = fileinfo.st_size - IndexSize - sizeof(ModSize) - ModSize - sizeof(IndexSize);
|
||||
|
||||
// Close the stream
|
||||
stream->Close();
|
||||
void* FilePtr = FEXCore::Allocator::mmap(nullptr, Size, PROT_READ, MAP_SHARED, streamfd, 0);
|
||||
|
||||
// Rename the file to atomically update the cache with the temporary file
|
||||
AOTIRRenamer(String);
|
||||
if (FilePtr == MAP_FAILED) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Array = (AOTIRInlineIndex*)((char*)FilePtr + IndexOffset);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr && Entry->FilePtr == nullptr, "Entry must not be initialized here");
|
||||
Entry->Array = Array;
|
||||
Entry->FilePtr = FilePtr;
|
||||
Entry->Size = Size;
|
||||
|
||||
LogMan::Msg::DFmt("AOTIR: Module {} has {} functions", Module, Array->Count);
|
||||
|
||||
return true;
|
||||
#else
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::FinalizeAOTIRCache() {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
for (auto& [String, Entry] : AOTIRCaptureCacheMap) {
|
||||
if (!Entry.Stream) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto ModSize = String.size();
|
||||
auto& stream = Entry.Stream;
|
||||
|
||||
// pad to 32 bytes
|
||||
constexpr char Zero = 0;
|
||||
while (stream->Offset() & 31) {
|
||||
stream->Write(&Zero, 1);
|
||||
}
|
||||
|
||||
// AOTIRInlineIndex
|
||||
const auto FnCount = Entry.Index.size();
|
||||
const size_t DataBase = -stream->Offset();
|
||||
|
||||
stream->Write((const char*)&FnCount, sizeof(FnCount));
|
||||
stream->Write((const char*)&DataBase, sizeof(DataBase));
|
||||
|
||||
for (const auto& [GuestStart, DataOffset] : Entry.Index) {
|
||||
// AOTIRInlineIndexEntry
|
||||
|
||||
// GuestStart
|
||||
stream->Write((const char*)&GuestStart, sizeof(GuestStart));
|
||||
|
||||
// DataOffset
|
||||
stream->Write((const char*)&DataOffset, sizeof(DataOffset));
|
||||
}
|
||||
|
||||
// End of file header
|
||||
const auto IndexSize = FnCount * sizeof(FEXCore::IR::AOTIRInlineIndexEntry) + sizeof(DataBase) + sizeof(FnCount);
|
||||
stream->Write((const char*)&IndexSize, sizeof(IndexSize));
|
||||
stream->Write(String.c_str(), ModSize);
|
||||
stream->Write((const char*)&ModSize, sizeof(ModSize));
|
||||
|
||||
// Close the stream
|
||||
stream->Close();
|
||||
|
||||
// Rename the file to atomically update the cache with the temporary file
|
||||
AOTIRRenamer(String);
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Flush() {
|
||||
{
|
||||
std::shared_lock lk {AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Flush() {
|
||||
{
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
for (;;) {
|
||||
// This code is tricky to refactor so it doesn't allocate memory through glibc.
|
||||
// The moved std::function object deallocates memory at the end of scope.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
|
||||
AOTIRCaptureCacheWriteoutLock.lock();
|
||||
WriteOutFn fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
bool MaybeEmpty = false;
|
||||
AOTIRCaptureCacheWriteoutQueue.pop();
|
||||
MaybeEmpty = AOTIRCaptureCacheWriteoutQueue.size() == 0;
|
||||
AOTIRCaptureCacheWriteoutLock.unlock();
|
||||
|
||||
fn();
|
||||
if (MaybeEmpty) {
|
||||
std::shared_lock lk {AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
// This code is tricky to refactor so it doesn't allocate memory through glibc.
|
||||
// The moved std::function object deallocates memory at the end of scope.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
|
||||
AOTIRCaptureCacheWriteoutLock.lock();
|
||||
WriteOutFn fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
bool MaybeEmpty = false;
|
||||
AOTIRCaptureCacheWriteoutQueue.pop();
|
||||
MaybeEmpty = AOTIRCaptureCacheWriteoutQueue.size() == 0;
|
||||
AOTIRCaptureCacheWriteoutLock.unlock();
|
||||
|
||||
fn();
|
||||
if (MaybeEmpty) {
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Must never get here");
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn &fn) {
|
||||
bool Flush = false;
|
||||
LOGMAN_MSG_A_FMT("Must never get here");
|
||||
}
|
||||
|
||||
{
|
||||
std::unique_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
AOTIRCaptureCacheWriteoutQueue.push(fn);
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() > 10000) {
|
||||
Flush = true;
|
||||
}
|
||||
}
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn& fn) {
|
||||
bool Flush = false;
|
||||
|
||||
bool test_val = false;
|
||||
if (Flush && AOTIRCaptureCacheWriteoutFlusing.compare_exchange_strong(test_val, true)) {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
{
|
||||
std::unique_lock lk {AOTIRCaptureCacheWriteoutLock};
|
||||
AOTIRCaptureCacheWriteoutQueue.push(fn);
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() > 10000) {
|
||||
Flush = true;
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn &Writer) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for( const auto &Entry: AOTIRCache) {
|
||||
if (Entry.second.ContainsCode) {
|
||||
Writer(Entry.second.FileId, Entry.second.Filename);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult AOTIRCaptureCache::PreGenerateIRFetch(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, FEXCore::IR::IRListView *IRList) {
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
PreGenerateIRFetchResult Result{};
|
||||
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
AOTIRCacheEntry.Entry->ContainsCode = true;
|
||||
|
||||
if (IRList == nullptr && CTX->Config.AOTIRLoad()) {
|
||||
auto Mod = AOTIRCacheEntry.Entry->Array;
|
||||
|
||||
if (Mod != nullptr)
|
||||
{
|
||||
auto AOTEntry = Mod->Find(GuestRIP - AOTIRCacheEntry.VAFileStart);
|
||||
|
||||
if (AOTEntry) {
|
||||
// verify hash
|
||||
auto MappedStart = GuestRIP;
|
||||
auto hash = XXH3_64bits((void*)MappedStart, AOTEntry->GuestLength);
|
||||
if (hash == AOTEntry->GuestHash) {
|
||||
Result.IRList = AOTEntry->GetIRData();
|
||||
//LogMan::Msg::DFmt("using {} + {:x} -> {:x}\n", file->second.fileid, AOTEntry->first, GuestRIP);
|
||||
|
||||
Result.RAData = AOTEntry->GetRAData()->CreateCopy();
|
||||
Result.DebugData = new FEXCore::Core::DebugData();
|
||||
Result.StartAddr = MappedStart;
|
||||
Result.Length = AOTEntry->GuestLength;
|
||||
Result.GeneratedIR = true;
|
||||
} else {
|
||||
LogMan::Msg::IFmt("AOTIR: hash check failed {:x}\n", MappedStart);
|
||||
}
|
||||
} else {
|
||||
//LogMan::Msg::IFmt("AOTIR: Failed to find {:x}, {:x}, {}\n", GuestRIP, GuestRIP - file->second.Start + file->second.Offset, file->second.fileid);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
bool AOTIRCaptureCache::PostCompileCode(
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
void* CodePtr,
|
||||
uint64_t GuestRIP,
|
||||
uint64_t StartAddr,
|
||||
uint64_t Length,
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData,
|
||||
FEXCore::IR::IRListView *IRList,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
bool GeneratedIR) {
|
||||
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (GeneratedIR || CTX->Config.LibraryJITNaming() || CTX->Config.GDBSymbols()) {
|
||||
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
if (DebugData && CTX->Config.LibraryJITNaming()) {
|
||||
CTX->Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, AOTIRCacheEntry.Entry->Filename);
|
||||
}
|
||||
|
||||
if (CTX->Config.GDBSymbols()) {
|
||||
GDBJITRegister(AOTIRCacheEntry.Entry, AOTIRCacheEntry.VAFileStart, GuestRIP, (uintptr_t)CodePtr, DebugData);
|
||||
}
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if (GeneratedIR && RAData &&
|
||||
(CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
auto LocalRIP = GuestRIP - AOTIRCacheEntry.VAFileStart;
|
||||
auto LocalStartAddr = StartAddr - AOTIRCacheEntry.VAFileStart;
|
||||
auto FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
// The underlying pointer and the unique_ptr deleter for RAData must
|
||||
// be marshalled separately to the lambda below. Otherwise, the
|
||||
// lambda can't be used as an std::function due to being non-copyable
|
||||
auto RADataCopy = RAData->CreateCopy();
|
||||
auto RADataCopyDeleter = RADataCopy.get_deleter();
|
||||
auto IRListCopy = IRList->CreateCopy();
|
||||
|
||||
// The lambda is converted to std::function. This is tricky to refactor so it doesn't allocate memory through glibc.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy=RADataCopy.release(), RADataCopyDeleter, FileId]() {
|
||||
|
||||
// It is guaranteed via AOTIRCaptureCacheWriteoutLock and AOTIRCaptureCacheWriteoutFlusing that this will not run concurrently
|
||||
// Memory coherency is guaranteed via AOTIRCaptureCacheWriteoutLock
|
||||
|
||||
auto *AotFile = &AOTIRCaptureCacheMap[FileId];
|
||||
|
||||
if (!AotFile->Stream) {
|
||||
AotFile->Stream = AOTIRWriter(FileId);
|
||||
uint64_t tag = FEXCore::IR::AOTIR_COOKIE;
|
||||
AotFile->Stream->Write(&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy);
|
||||
RADataCopyDeleter(RADataCopy);
|
||||
delete IRListCopy;
|
||||
});
|
||||
|
||||
if (CTX->Config.AOTIRGenerate()) {
|
||||
// cleanup memory and early exit here -- we're not running the application
|
||||
Thread->CPUBackend->ClearCache();
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
if (IRList->IsCopy()) delete IRList;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
AOTIRCacheEntry *AOTIRCaptureCache::LoadAOTIRCacheEntry(const fextl::string &filename) {
|
||||
fextl::string base_filename = FHU::Filesystem::GetFilename(filename);
|
||||
|
||||
if (!base_filename.empty()) {
|
||||
auto filename_hash = XXH3_64bits(filename.c_str(), filename.size());
|
||||
|
||||
auto fileid = fextl::fmt::format("{}-{}-{}{}{}",
|
||||
base_filename,
|
||||
filename_hash,
|
||||
(CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) ? 'S' : 's',
|
||||
CTX->Config.TSOEnabled ? 'T' : 't',
|
||||
CTX->Config.ABILocalFlags ? 'L' : 'l');
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry { .FileId = fileid, .Filename = filename }});
|
||||
auto Entry = &(Inserted.first->second);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr, "Duplicate LoadAOTIRCacheEntry");
|
||||
|
||||
if (CTX->Config.AOTIRLoad && AOTIRLoader) {
|
||||
auto streamfd = AOTIRLoader(fileid);
|
||||
if (streamfd != -1) {
|
||||
FEXCore::IR::LoadAOTIRCache(Entry, streamfd);
|
||||
close(streamfd);
|
||||
}
|
||||
}
|
||||
return Entry;
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::UnloadAOTIRCacheEntry(AOTIRCacheEntry *Entry) {
|
||||
#ifndef _WIN32
|
||||
LOGMAN_THROW_AA_FMT(Entry != nullptr, "Removing not existing entry");
|
||||
|
||||
if (Entry->Array) {
|
||||
FEXCore::Allocator::munmap(Entry->FilePtr, Entry->Size);
|
||||
Entry->Array = nullptr;
|
||||
Entry->FilePtr = nullptr;
|
||||
Entry->Size = 0;
|
||||
}
|
||||
#endif
|
||||
bool test_val = false;
|
||||
if (Flush && AOTIRCaptureCacheWriteoutFlusing.compare_exchange_strong(test_val, true)) {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn& Writer) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for (const auto& Entry : AOTIRCache) {
|
||||
if (Entry.second.ContainsCode) {
|
||||
Writer(Entry.second.FileId, Entry.second.Filename);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult
|
||||
AOTIRCaptureCache::PreGenerateIRFetch(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, FEXCore::IR::IRListView* IRList) {
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
PreGenerateIRFetchResult Result {};
|
||||
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
AOTIRCacheEntry.Entry->ContainsCode = true;
|
||||
|
||||
if (IRList == nullptr && CTX->Config.AOTIRLoad()) {
|
||||
auto Mod = AOTIRCacheEntry.Entry->Array;
|
||||
|
||||
if (Mod != nullptr) {
|
||||
auto AOTEntry = Mod->Find(GuestRIP - AOTIRCacheEntry.VAFileStart);
|
||||
|
||||
if (AOTEntry) {
|
||||
// verify hash
|
||||
auto MappedStart = GuestRIP;
|
||||
auto hash = XXH3_64bits((void*)MappedStart, AOTEntry->GuestLength);
|
||||
if (hash == AOTEntry->GuestHash) {
|
||||
Result.IRList = AOTEntry->GetIRData();
|
||||
// LogMan::Msg::DFmt("using {} + {:x} -> {:x}\n", file->second.fileid, AOTEntry->first, GuestRIP);
|
||||
|
||||
Result.RAData = AOTEntry->GetRAData()->CreateCopy();
|
||||
Result.DebugData = new FEXCore::Core::DebugData();
|
||||
Result.StartAddr = MappedStart;
|
||||
Result.Length = AOTEntry->GuestLength;
|
||||
Result.GeneratedIR = true;
|
||||
} else {
|
||||
LogMan::Msg::IFmt("AOTIR: hash check failed {:x}\n", MappedStart);
|
||||
}
|
||||
} else {
|
||||
// LogMan::Msg::IFmt("AOTIR: Failed to find {:x}, {:x}, {}\n", GuestRIP, GuestRIP - file->second.Start + file->second.Offset, file->second.fileid);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr,
|
||||
uint64_t Length, FEXCore::IR::RegisterAllocationData::UniquePtr RAData,
|
||||
FEXCore::IR::IRListView* IRList, FEXCore::Core::DebugData* DebugData, bool GeneratedIR) {
|
||||
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (GeneratedIR || CTX->Config.LibraryJITNaming() || CTX->Config.GDBSymbols()) {
|
||||
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
if (DebugData && CTX->Config.LibraryJITNaming()) {
|
||||
CTX->Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, AOTIRCacheEntry.Entry->Filename);
|
||||
}
|
||||
|
||||
if (CTX->Config.GDBSymbols()) {
|
||||
GDBJITRegister(AOTIRCacheEntry.Entry, AOTIRCacheEntry.VAFileStart, GuestRIP, (uintptr_t)CodePtr, DebugData);
|
||||
}
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if (GeneratedIR && RAData && (CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
auto LocalRIP = GuestRIP - AOTIRCacheEntry.VAFileStart;
|
||||
auto LocalStartAddr = StartAddr - AOTIRCacheEntry.VAFileStart;
|
||||
auto FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
// The underlying pointer and the unique_ptr deleter for RAData must
|
||||
// be marshalled separately to the lambda below. Otherwise, the
|
||||
// lambda can't be used as an std::function due to being non-copyable
|
||||
auto RADataCopy = RAData->CreateCopy();
|
||||
auto RADataCopyDeleter = RADataCopy.get_deleter();
|
||||
auto IRListCopy = IRList->CreateCopy();
|
||||
|
||||
// The lambda is converted to std::function. This is tricky to refactor so it doesn't allocate memory through glibc.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
AOTIRCaptureCacheWriteoutQueue_Append(
|
||||
[this, LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy = RADataCopy.release(), RADataCopyDeleter, FileId]() {
|
||||
// It is guaranteed via AOTIRCaptureCacheWriteoutLock and AOTIRCaptureCacheWriteoutFlusing that this will not run concurrently
|
||||
// Memory coherency is guaranteed via AOTIRCaptureCacheWriteoutLock
|
||||
|
||||
auto* AotFile = &AOTIRCaptureCacheMap[FileId];
|
||||
|
||||
if (!AotFile->Stream) {
|
||||
AotFile->Stream = AOTIRWriter(FileId);
|
||||
uint64_t tag = FEXCore::IR::AOTIR_COOKIE;
|
||||
AotFile->Stream->Write(&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy);
|
||||
RADataCopyDeleter(RADataCopy);
|
||||
delete IRListCopy;
|
||||
});
|
||||
|
||||
if (CTX->Config.AOTIRGenerate()) {
|
||||
// cleanup memory and early exit here -- we're not running the application
|
||||
Thread->CPUBackend->ClearCache();
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
if (IRList->IsCopy()) {
|
||||
delete IRList;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
AOTIRCacheEntry* AOTIRCaptureCache::LoadAOTIRCacheEntry(const fextl::string& filename) {
|
||||
fextl::string base_filename = FHU::Filesystem::GetFilename(filename);
|
||||
|
||||
if (!base_filename.empty()) {
|
||||
auto filename_hash = XXH3_64bits(filename.c_str(), filename.size());
|
||||
|
||||
auto fileid = fextl::fmt::format("{}-{}-{}{}{}", base_filename, filename_hash,
|
||||
(CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) ? 'S' : 's',
|
||||
CTX->Config.TSOEnabled ? 'T' : 't', CTX->Config.ABILocalFlags ? 'L' : 'l');
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry {.FileId = fileid, .Filename = filename}});
|
||||
auto Entry = &(Inserted.first->second);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr, "Duplicate LoadAOTIRCacheEntry");
|
||||
|
||||
if (CTX->Config.AOTIRLoad && AOTIRLoader) {
|
||||
auto streamfd = AOTIRLoader(fileid);
|
||||
if (streamfd != -1) {
|
||||
FEXCore::IR::LoadAOTIRCache(Entry, streamfd);
|
||||
close(streamfd);
|
||||
}
|
||||
}
|
||||
return Entry;
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::UnloadAOTIRCacheEntry(AOTIRCacheEntry* Entry) {
|
||||
#ifndef _WIN32
|
||||
LOGMAN_THROW_AA_FMT(Entry != nullptr, "Removing not existing entry");
|
||||
|
||||
if (Entry->Array) {
|
||||
FEXCore::Allocator::munmap(Entry->FilePtr, Entry->Size);
|
||||
Entry->Array = nullptr;
|
||||
Entry->FilePtr = nullptr;
|
||||
Entry->Size = 0;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
+115
-117
@@ -1,7 +1,8 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "FEXCore/IR/RegisterAllocationData.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -19,136 +20,133 @@ namespace FEXCore::Core {
|
||||
struct DebugData;
|
||||
}
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
|
||||
constexpr auto COOKIE_VERSION = [](const char CookieText[4], uint32_t Version) {
|
||||
uint64_t Cookie = Version;
|
||||
Cookie <<= 32;
|
||||
constexpr auto COOKIE_VERSION = [](const char CookieText[4], uint32_t Version) {
|
||||
uint64_t Cookie = Version;
|
||||
Cookie <<= 32;
|
||||
|
||||
// Make the cookie text be the lower bits
|
||||
Cookie |= CookieText[3];
|
||||
Cookie <<= 8;
|
||||
Cookie |= CookieText[2];
|
||||
Cookie <<= 8;
|
||||
Cookie |= CookieText[1];
|
||||
Cookie <<= 8;
|
||||
Cookie |= CookieText[0];
|
||||
// Make the cookie text be the lower bits
|
||||
Cookie |= CookieText[3];
|
||||
Cookie <<= 8;
|
||||
Cookie |= CookieText[2];
|
||||
Cookie <<= 8;
|
||||
Cookie |= CookieText[1];
|
||||
Cookie <<= 8;
|
||||
Cookie |= CookieText[0];
|
||||
|
||||
return Cookie;
|
||||
return Cookie;
|
||||
};
|
||||
constexpr static uint32_t AOTIR_VERSION = 0x0000'00004;
|
||||
constexpr static uint64_t AOTIR_COOKIE = COOKIE_VERSION("FEXI", AOTIR_VERSION);
|
||||
|
||||
struct AOTIRInlineEntry {
|
||||
uint64_t GuestHash;
|
||||
uint64_t GuestLength;
|
||||
|
||||
/* RAData followed by IRData */
|
||||
uint8_t InlineData[0];
|
||||
|
||||
IR::RegisterAllocationData* GetRAData();
|
||||
IR::IRListView* GetIRData();
|
||||
};
|
||||
|
||||
struct AOTIRInlineIndexEntry {
|
||||
uint64_t GuestStart;
|
||||
uint64_t DataOffset;
|
||||
};
|
||||
|
||||
struct AOTIRInlineIndex {
|
||||
uint64_t Count;
|
||||
uint64_t DataBase;
|
||||
AOTIRInlineIndexEntry Entries[0];
|
||||
|
||||
AOTIRInlineEntry* Find(uint64_t GuestStart);
|
||||
AOTIRInlineEntry* GetInlineEntry(uint64_t DataOffset);
|
||||
};
|
||||
|
||||
struct AOTIRCaptureCacheEntry {
|
||||
fextl::unique_ptr<FEXCore::Context::AOTIRWriter> Stream;
|
||||
fextl::map<uint64_t, uint64_t> Index;
|
||||
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView* IRList,
|
||||
FEXCore::IR::RegisterAllocationData* RAData);
|
||||
};
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
AOTIRInlineIndex* Array;
|
||||
void* FilePtr;
|
||||
size_t Size;
|
||||
std::unique_ptr<FEXCore::HLE::SourcecodeMap> SourcecodeMap;
|
||||
fextl::string FileId;
|
||||
fextl::string Filename;
|
||||
bool ContainsCode;
|
||||
};
|
||||
|
||||
using AOTCacheType = fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCacheEntry>;
|
||||
|
||||
class AOTIRCaptureCache final {
|
||||
public:
|
||||
using WriteOutFn = std::function<void()>;
|
||||
|
||||
AOTIRCaptureCache(FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx} {}
|
||||
|
||||
void FinalizeAOTIRCache();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn& fn);
|
||||
void WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn& Writer);
|
||||
|
||||
struct PreGenerateIRFetchResult {
|
||||
FEXCore::IR::IRListView* IRList {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
bool GeneratedIR {};
|
||||
};
|
||||
constexpr static uint32_t AOTIR_VERSION = 0x0000'00004;
|
||||
constexpr static uint64_t AOTIR_COOKIE = COOKIE_VERSION("FEXI", AOTIR_VERSION);
|
||||
[[nodiscard]]
|
||||
PreGenerateIRFetchResult PreGenerateIRFetch(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, FEXCore::IR::IRListView* IRList);
|
||||
|
||||
struct AOTIRInlineEntry {
|
||||
uint64_t GuestHash;
|
||||
uint64_t GuestLength;
|
||||
bool PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr, uint64_t Length,
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData, FEXCore::IR::IRListView* IRList,
|
||||
FEXCore::Core::DebugData* DebugData, bool GeneratedIR);
|
||||
|
||||
/* RAData followed by IRData */
|
||||
uint8_t InlineData[0];
|
||||
AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& filename);
|
||||
void UnloadAOTIRCacheEntry(AOTIRCacheEntry* Entry);
|
||||
|
||||
IR::RegisterAllocationData *GetRAData();
|
||||
IR::IRListView *GetIRData();
|
||||
};
|
||||
// Callbacks
|
||||
void SetAOTIRLoader(Context::AOTIRLoaderCBFn CacheReader) {
|
||||
AOTIRLoader = std::move(CacheReader);
|
||||
}
|
||||
|
||||
struct AOTIRInlineIndexEntry {
|
||||
uint64_t GuestStart;
|
||||
uint64_t DataOffset;
|
||||
};
|
||||
void SetAOTIRWriter(Context::AOTIRWriterCBFn CacheWriter) {
|
||||
AOTIRWriter = std::move(CacheWriter);
|
||||
}
|
||||
|
||||
struct AOTIRInlineIndex {
|
||||
uint64_t Count;
|
||||
uint64_t DataBase;
|
||||
AOTIRInlineIndexEntry Entries[0];
|
||||
void SetAOTIRRenamer(Context::AOTIRRenamerCBFn CacheRenamer) {
|
||||
AOTIRRenamer = std::move(CacheRenamer);
|
||||
}
|
||||
|
||||
AOTIRInlineEntry *Find(uint64_t GuestStart);
|
||||
AOTIRInlineEntry *GetInlineEntry(uint64_t DataOffset);
|
||||
};
|
||||
private:
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
struct AOTIRCaptureCacheEntry {
|
||||
fextl::unique_ptr<FEXCore::Context::AOTIRWriter> Stream;
|
||||
fextl::map<uint64_t, uint64_t> Index;
|
||||
std::shared_mutex AOTIRCacheLock;
|
||||
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
|
||||
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
|
||||
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData);
|
||||
};
|
||||
fextl::queue<WriteOutFn> AOTIRCaptureCacheWriteoutQueue;
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
AOTIRInlineIndex *Array;
|
||||
void *FilePtr;
|
||||
size_t Size;
|
||||
std::unique_ptr<FEXCore::HLE::SourcecodeMap> SourcecodeMap;
|
||||
fextl::string FileId;
|
||||
fextl::string Filename;
|
||||
bool ContainsCode;
|
||||
};
|
||||
FEXCore::IR::AOTCacheType AOTIRCache;
|
||||
|
||||
using AOTCacheType = fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCacheEntry>;
|
||||
|
||||
class AOTIRCaptureCache final {
|
||||
public:
|
||||
using WriteOutFn = std::function<void()>;
|
||||
|
||||
AOTIRCaptureCache(FEXCore::Context::ContextImpl *ctx) : CTX {ctx} {}
|
||||
|
||||
void FinalizeAOTIRCache();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn &fn);
|
||||
void WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn &Writer);
|
||||
|
||||
struct PreGenerateIRFetchResult {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
bool GeneratedIR {};
|
||||
};
|
||||
[[nodiscard]] PreGenerateIRFetchResult PreGenerateIRFetch(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, FEXCore::IR::IRListView *IRList);
|
||||
|
||||
bool PostCompileCode(FEXCore::Core::InternalThreadState *Thread,
|
||||
void* CodePtr,
|
||||
uint64_t GuestRIP,
|
||||
uint64_t StartAddr,
|
||||
uint64_t Length,
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData,
|
||||
FEXCore::IR::IRListView *IRList,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
bool GeneratedIR);
|
||||
|
||||
AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string &filename);
|
||||
void UnloadAOTIRCacheEntry(AOTIRCacheEntry *Entry);
|
||||
|
||||
// Callbacks
|
||||
void SetAOTIRLoader(Context::AOTIRLoaderCBFn CacheReader) {
|
||||
AOTIRLoader = std::move(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(Context::AOTIRWriterCBFn CacheWriter) {
|
||||
AOTIRWriter = std::move(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(Context::AOTIRRenamerCBFn CacheRenamer) {
|
||||
AOTIRRenamer = std::move(CacheRenamer);
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
|
||||
std::shared_mutex AOTIRCacheLock;
|
||||
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
|
||||
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
|
||||
|
||||
fextl::queue<WriteOutFn> AOTIRCaptureCacheWriteoutQueue;
|
||||
|
||||
FEXCore::IR::AOTCacheType AOTIRCache;
|
||||
|
||||
Context::AOTIRLoaderCBFn AOTIRLoader;
|
||||
Context::AOTIRWriterCBFn AOTIRWriter;
|
||||
Context::AOTIRRenamerCBFn AOTIRRenamer;
|
||||
fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCaptureCacheEntry> AOTIRCaptureCacheMap;
|
||||
};
|
||||
}
|
||||
Context::AOTIRLoaderCBFn AOTIRLoader;
|
||||
Context::AOTIRWriterCBFn AOTIRWriter;
|
||||
Context::AOTIRRenamerCBFn AOTIRRenamer;
|
||||
fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCaptureCacheEntry> AOTIRCaptureCacheMap;
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
@@ -9,11 +9,701 @@
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class OrderedNode;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
/**
|
||||
* @brief The IROp_Header is an dynamically sized array
|
||||
* At the end it contains a uint8_t for the number of arguments that Op has
|
||||
* Then there is an unsized array of NodeWrapper arguments for the number of arguments this op has
|
||||
* The op structures that are including the header must ensure that they pad themselves correctly to the number of arguments used
|
||||
*/
|
||||
struct IROp_Header;
|
||||
|
||||
/**
|
||||
* @brief Represents the ID of a given IR node.
|
||||
*
|
||||
* Intended to provide strong typing from other integer values
|
||||
* to prevent passing incorrect values to certain API functions.
|
||||
*/
|
||||
struct NodeID final {
|
||||
using value_type = uint32_t;
|
||||
|
||||
constexpr NodeID() noexcept = default;
|
||||
constexpr explicit NodeID(value_type Value_) noexcept
|
||||
: Value {Value_} {}
|
||||
|
||||
constexpr NodeID(const NodeID&) noexcept = default;
|
||||
constexpr NodeID& operator=(const NodeID&) noexcept = default;
|
||||
|
||||
constexpr NodeID(NodeID&&) noexcept = default;
|
||||
constexpr NodeID& operator=(NodeID&&) noexcept = default;
|
||||
|
||||
[[nodiscard]]
|
||||
constexpr bool IsValid() const noexcept {
|
||||
return Value != 0;
|
||||
}
|
||||
[[nodiscard]]
|
||||
constexpr bool IsInvalid() const noexcept {
|
||||
return !IsValid();
|
||||
}
|
||||
constexpr void Invalidate() noexcept {
|
||||
Value = 0;
|
||||
}
|
||||
|
||||
[[nodiscard]] friend constexpr bool operator==(NodeID, NodeID) noexcept = default;
|
||||
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator<(NodeID lhs, NodeID rhs) noexcept {
|
||||
return lhs.Value < rhs.Value;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator>(NodeID lhs, NodeID rhs) noexcept {
|
||||
return operator<(rhs, lhs);
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator<=(NodeID lhs, NodeID rhs) noexcept {
|
||||
return !operator>(lhs, rhs);
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator>=(NodeID lhs, NodeID rhs) noexcept {
|
||||
return !operator<(lhs, rhs);
|
||||
}
|
||||
|
||||
friend std::ostream& operator<<(std::ostream& out, NodeID ID) {
|
||||
out << ID.Value;
|
||||
return out;
|
||||
}
|
||||
friend std::istream& operator>>(std::istream& in, NodeID& ID) {
|
||||
in >> ID.Value;
|
||||
return in;
|
||||
}
|
||||
|
||||
value_type Value {};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is a very simple wrapper for our node pointers
|
||||
* You probably don't want to use this directly
|
||||
* Use OpNodeWrapper and OrderedNodeWrapper types below instead
|
||||
*
|
||||
* This is necessary to allow two things
|
||||
* - Reduce memory usage by having the pointer be an 32bit offset rather than the whole 64bit pointer
|
||||
* - Actually use an offset from a base so we aren't storing pointers for everything
|
||||
* - Makes IR list copying be as cheap as a memcpy
|
||||
* Downsides
|
||||
* - The IR nodes have to be allocated out of a linear array of memory
|
||||
* - We currently only allow a 32bit offset, so *only* 4 million nodes per list
|
||||
* - We have to have the base offset live somewhere else
|
||||
* - Has to be POD and trivially copyable
|
||||
* - Makes every real node access turn in to a [Base + Offset] access
|
||||
* - Can be confusing if you're mixing OpNodeWrapper and OrderedNodeWrapper usage
|
||||
*/
|
||||
template<typename Type>
|
||||
struct NodeWrapperBase final {
|
||||
// On x86-64 using a uint64_t type is more efficient since RIP addressing gives you [<Base> + <Index> + <imm offset>]
|
||||
// On AArch64 using uint32_t is just more memory efficient. 32bit or 64bit offset doesn't matter
|
||||
// We use uint32_t to be more memory efficient (Cuts our node list size in half)
|
||||
using NodeOffsetType = uint32_t;
|
||||
NodeOffsetType NodeOffset;
|
||||
|
||||
explicit NodeWrapperBase() = default;
|
||||
|
||||
[[nodiscard]]
|
||||
static NodeWrapperBase WrapOffset(NodeOffsetType Offset) {
|
||||
NodeWrapperBase Wrapped;
|
||||
Wrapped.NodeOffset = Offset;
|
||||
return Wrapped;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static NodeWrapperBase WrapPtr(uintptr_t Base, uintptr_t Value) {
|
||||
NodeWrapperBase Wrapped;
|
||||
Wrapped.SetOffset(Base, Value);
|
||||
return Wrapped;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static void* UnwrapNode(uintptr_t Base, NodeWrapperBase Node) {
|
||||
return Node.GetNode(Base);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
NodeID ID() const;
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsInvalid() const {
|
||||
return NodeOffset == 0;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
Type* GetNode(uintptr_t Base) {
|
||||
return reinterpret_cast<Type*>(Base + NodeOffset);
|
||||
}
|
||||
[[nodiscard]]
|
||||
const Type* GetNode(uintptr_t Base) const {
|
||||
return reinterpret_cast<const Type*>(Base + NodeOffset);
|
||||
}
|
||||
|
||||
void SetOffset(uintptr_t Base, uintptr_t Value) {
|
||||
NodeOffset = Value - Base;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<NodeWrapperBase<OrderedNode>>);
|
||||
|
||||
static_assert(sizeof(NodeWrapperBase<OrderedNode>) == sizeof(uint32_t));
|
||||
|
||||
using OpNodeWrapper = NodeWrapperBase<IROp_Header>;
|
||||
using OrderedNodeWrapper = NodeWrapperBase<OrderedNode>;
|
||||
|
||||
struct OrderedNodeHeader {
|
||||
OpNodeWrapper Value;
|
||||
OrderedNodeWrapper Next;
|
||||
OrderedNodeWrapper Previous;
|
||||
};
|
||||
|
||||
static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3);
|
||||
|
||||
/**
|
||||
* @brief This is a node in our IR representation
|
||||
* Is a doubly linked list node that lives in a representation of a linearly allocated node list
|
||||
* The links in the nodes can live in a list independent of the data IR data
|
||||
*
|
||||
* ex.
|
||||
* Region1 : ... <-> <OrderedNode> <-> <OrderedNode> <-> ...
|
||||
* | *<Value> |
|
||||
* v v
|
||||
* Region2 : <IROp>..<IROp>..<IROp>..<IROp>
|
||||
*
|
||||
* In this example the OrderedNodes are allocated in one linear memory region (Not necessarily contiguous with one another linking)
|
||||
* The second region is contiguous but they don't have any relationship with one another directly
|
||||
*/
|
||||
class OrderedNode final {
|
||||
friend class NodeWrapperIterator;
|
||||
friend class OrderedList;
|
||||
public:
|
||||
// These three values are laid out very specifically to make it fast to access the NodeWrappers specifically
|
||||
OrderedNodeHeader Header;
|
||||
uint32_t NumUses;
|
||||
|
||||
using value_type = OrderedNodeWrapper;
|
||||
|
||||
OrderedNode() = default;
|
||||
|
||||
/**
|
||||
* @brief Appends a node to this current node
|
||||
*
|
||||
* Before. <Prev> <-> <Current> <-> <Next>
|
||||
* After. <Prev> <-> <Current> <-> <Node> <-> Next
|
||||
*
|
||||
* @return Pointer to the node being added
|
||||
*/
|
||||
value_type append(uintptr_t Base, value_type Node) {
|
||||
// Set Next Node's Previous to incoming node
|
||||
SetPrevious(Base, Header.Next, Node);
|
||||
|
||||
// Set Incoming node's links to this node's links
|
||||
SetPrevious(Base, Node, Wrapped(Base));
|
||||
SetNext(Base, Node, Header.Next);
|
||||
|
||||
// Set this node's next to the incoming node
|
||||
SetNext(Base, Wrapped(Base), Node);
|
||||
|
||||
// Return the node we are appending
|
||||
return Node;
|
||||
}
|
||||
|
||||
OrderedNode* append(uintptr_t Base, OrderedNode* Node) {
|
||||
value_type WNode = Node->Wrapped(Base);
|
||||
// Set Next Node's Previous to incoming node
|
||||
SetPrevious(Base, Header.Next, WNode);
|
||||
|
||||
// Set Incoming node's links to this node's links
|
||||
SetPrevious(Base, WNode, Wrapped(Base));
|
||||
SetNext(Base, WNode, Header.Next);
|
||||
|
||||
// Set this node's next to the incoming node
|
||||
SetNext(Base, Wrapped(Base), WNode);
|
||||
|
||||
// Return the node we are appending
|
||||
return Node;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Prepends a node to the current node
|
||||
* Before. <Prev> <-> <Current> <-> <Next>
|
||||
* After. <Prev> <-> <Node> <-> <Current> <-> Next
|
||||
*
|
||||
* @return Pointer to the node being added
|
||||
*/
|
||||
value_type prepend(uintptr_t Base, value_type Node) {
|
||||
// Set the previous node's next to the incoming node
|
||||
SetNext(Base, Header.Previous, Node);
|
||||
|
||||
// Set the incoming node's links
|
||||
SetPrevious(Base, Node, Header.Previous);
|
||||
SetNext(Base, Node, Wrapped(Base));
|
||||
|
||||
// Set the current node's link
|
||||
SetPrevious(Base, Wrapped(Base), Node);
|
||||
|
||||
// Return the node we are prepending
|
||||
return Node;
|
||||
}
|
||||
|
||||
OrderedNode* prepend(uintptr_t Base, OrderedNode* Node) {
|
||||
value_type WNode = Node->Wrapped(Base);
|
||||
// Set the previous node's next to the incoming node
|
||||
SetNext(Base, Header.Previous, WNode);
|
||||
|
||||
// Set the incoming node's links
|
||||
SetPrevious(Base, WNode, Header.Previous);
|
||||
SetNext(Base, WNode, Wrapped(Base));
|
||||
|
||||
// Set the current node's link
|
||||
SetPrevious(Base, Wrapped(Base), WNode);
|
||||
|
||||
// Return the node we are prepending
|
||||
return Node;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Gets the remaining size of the blocks from this point onward
|
||||
*
|
||||
* Doesn't find the head of the list
|
||||
*
|
||||
*/
|
||||
[[nodiscard]]
|
||||
size_t size(uintptr_t Base) const {
|
||||
size_t Size = 1;
|
||||
// Walk the list forward until we hit a sentinel
|
||||
value_type Current = Header.Next;
|
||||
while (Current.NodeOffset != 0) {
|
||||
++Size;
|
||||
OrderedNode* RealNode = Current.GetNode(Base);
|
||||
Current = RealNode->Header.Next;
|
||||
}
|
||||
return Size;
|
||||
}
|
||||
|
||||
void Unlink(uintptr_t Base) {
|
||||
// This removes the node from the list. Orphaning it
|
||||
// Before: <Previous> <-> <Current> <-> <Next>
|
||||
// After: <Previous <-> <Next>
|
||||
SetNext(Base, Header.Previous, Header.Next);
|
||||
SetPrevious(Base, Header.Next, Header.Previous);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
const IROp_Header* Op(uintptr_t Base) const {
|
||||
return Header.Value.GetNode(Base);
|
||||
}
|
||||
[[nodiscard]]
|
||||
IROp_Header* Op(uintptr_t Base) {
|
||||
return Header.Value.GetNode(Base);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
uint32_t GetUses() const {
|
||||
return NumUses;
|
||||
}
|
||||
|
||||
void AddUse() {
|
||||
++NumUses;
|
||||
}
|
||||
void RemoveUse() {
|
||||
--NumUses;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
value_type Wrapped(uintptr_t Base) const {
|
||||
value_type Tmp;
|
||||
Tmp.SetOffset(Base, reinterpret_cast<uintptr_t>(this));
|
||||
return Tmp;
|
||||
}
|
||||
|
||||
private:
|
||||
[[nodiscard]]
|
||||
value_type WrappedOffset(uint32_t Offset) const {
|
||||
value_type Tmp;
|
||||
Tmp.NodeOffset = Offset;
|
||||
return Tmp;
|
||||
}
|
||||
|
||||
static void SetPrevious(uintptr_t Base, value_type Node, value_type New) {
|
||||
OrderedNode* RealNode = Node.GetNode(Base);
|
||||
RealNode->Header.Previous = New;
|
||||
}
|
||||
|
||||
static void SetNext(uintptr_t Base, value_type Node, value_type New) {
|
||||
OrderedNode* RealNode = Node.GetNode(Base);
|
||||
RealNode->Header.Next = New;
|
||||
}
|
||||
|
||||
void SetUses(uint32_t Uses) {
|
||||
NumUses = Uses;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_copyable_v<OrderedNode>);
|
||||
static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
|
||||
struct RegisterClassType final {
|
||||
using value_type = uint32_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator==(const RegisterClassType&, const RegisterClassType&) = default;
|
||||
};
|
||||
|
||||
struct CondClassType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator==(const CondClassType&, const CondClassType&) = default;
|
||||
};
|
||||
|
||||
struct MemOffsetType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator==(const MemOffsetType&, const MemOffsetType&) = default;
|
||||
};
|
||||
|
||||
struct TypeDefinition final {
|
||||
uint16_t Val;
|
||||
|
||||
[[nodiscard]] constexpr operator uint16_t() const {
|
||||
return Val;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static constexpr TypeDefinition Create(uint8_t Bytes) {
|
||||
TypeDefinition Type {};
|
||||
Type.Val = Bytes << 8;
|
||||
return Type;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static constexpr TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
|
||||
TypeDefinition Type {};
|
||||
Type.Val = (Bytes << 8) | (Elements & 255);
|
||||
return Type;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
constexpr uint8_t Bytes() const {
|
||||
return Val >> 8;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
constexpr uint8_t Elements() const {
|
||||
return Val & 255;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator==(const TypeDefinition&, const TypeDefinition&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<TypeDefinition>);
|
||||
|
||||
struct FenceType final {
|
||||
using value_type = uint8_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator==(const FenceType&, const FenceType&) = default;
|
||||
};
|
||||
|
||||
struct RoundType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator==(const RoundType&, const RoundType&) = default;
|
||||
};
|
||||
|
||||
class NodeIterator;
|
||||
|
||||
/* This iterator can be used to step though nodes.
|
||||
* Due to how our IR is laid out, this can be used to either step
|
||||
* though the CodeBlocks or though the code within a single block.
|
||||
*/
|
||||
class NodeIterator {
|
||||
public:
|
||||
using value_type = std::tuple<OrderedNode*, IROp_Header*>;
|
||||
using size_type = std::size_t;
|
||||
using difference_type = std::ptrdiff_t;
|
||||
using reference = value_type&;
|
||||
using const_reference = const value_type&;
|
||||
using pointer = value_type*;
|
||||
using const_pointer = const value_type*;
|
||||
using iterator = NodeIterator;
|
||||
using const_iterator = const NodeIterator;
|
||||
using reverse_iterator = iterator;
|
||||
using const_reverse_iterator = const_iterator;
|
||||
using iterator_category = std::bidirectional_iterator_tag;
|
||||
|
||||
NodeIterator(uintptr_t Base, uintptr_t IRBase)
|
||||
: BaseList {Base}
|
||||
, IRList {IRBase} {}
|
||||
explicit NodeIterator(uintptr_t Base, uintptr_t IRBase, OrderedNodeWrapper Ptr)
|
||||
: BaseList {Base}
|
||||
, IRList {IRBase}
|
||||
, Node {Ptr} {}
|
||||
|
||||
[[nodiscard]]
|
||||
bool
|
||||
operator==(const NodeIterator& rhs) const {
|
||||
return Node.NodeOffset == rhs.Node.NodeOffset;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool
|
||||
operator!=(const NodeIterator& rhs) const {
|
||||
return !operator==(rhs);
|
||||
}
|
||||
|
||||
NodeIterator operator++() {
|
||||
OrderedNodeHeader* RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
|
||||
Node = RealNode->Next;
|
||||
return *this;
|
||||
}
|
||||
|
||||
NodeIterator operator--() {
|
||||
OrderedNodeHeader* RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
|
||||
Node = RealNode->Previous;
|
||||
return *this;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
value_type
|
||||
operator*() {
|
||||
OrderedNode* RealNode = Node.GetNode(BaseList);
|
||||
return {RealNode, RealNode->Op(IRList)};
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
value_type
|
||||
operator()() {
|
||||
OrderedNode* RealNode = Node.GetNode(BaseList);
|
||||
return {RealNode, RealNode->Op(IRList)};
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
NodeID ID() const {
|
||||
return Node.ID();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static NodeIterator Invalid() {
|
||||
return NodeIterator(0, 0);
|
||||
}
|
||||
|
||||
protected:
|
||||
uintptr_t BaseList {};
|
||||
uintptr_t IRList {};
|
||||
OrderedNodeWrapper Node {};
|
||||
};
|
||||
|
||||
// This must directly match bytes to the named opsize.
|
||||
// Implicit sized IR operations does math to get between sizes.
|
||||
enum OpSize : uint8_t {
|
||||
i8Bit = 1,
|
||||
i16Bit = 2,
|
||||
i32Bit = 4,
|
||||
i64Bit = 8,
|
||||
i128Bit = 16,
|
||||
i256Bit = 32,
|
||||
};
|
||||
|
||||
enum class FloatCompareOp : uint8_t {
|
||||
EQ = 0,
|
||||
LT,
|
||||
LE,
|
||||
UNO,
|
||||
NEQ,
|
||||
ORD,
|
||||
};
|
||||
|
||||
enum class ShiftType : uint8_t {
|
||||
LSL = 0,
|
||||
LSR,
|
||||
ASR,
|
||||
ROR,
|
||||
};
|
||||
|
||||
|
||||
// Converts a size stored as an integer in to an OpSize enum.
|
||||
// This is a nop operation and will be eliminated by the compiler.
|
||||
static inline OpSize SizeToOpSize(uint8_t Size) {
|
||||
switch (Size) {
|
||||
case 1: return OpSize::i8Bit;
|
||||
case 2: return OpSize::i16Bit;
|
||||
case 4: return OpSize::i32Bit;
|
||||
case 8: return OpSize::i64Bit;
|
||||
case 16: return OpSize::i128Bit;
|
||||
case 32: return OpSize::i256Bit;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
#define IROP_ENUM
|
||||
#define IROP_STRUCTS
|
||||
#define IROP_SIZES
|
||||
#define IROP_REG_CLASSES
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
|
||||
/* This iterator can be used to step though every single node in a multi-block in SSA order.
|
||||
*
|
||||
* Iterates in the order of:
|
||||
*
|
||||
* end <-- CodeBlockA <--> BlockAInst1 <--> BlockAInst2 <--> CodeBlockB <--> BlockBInst1 <--> BlockBInst2 --> end
|
||||
*/
|
||||
class AllNodesIterator : public NodeIterator {
|
||||
public:
|
||||
AllNodesIterator(uintptr_t Base, uintptr_t IRBase)
|
||||
: NodeIterator(Base, IRBase) {}
|
||||
explicit AllNodesIterator(uintptr_t Base, uintptr_t IRBase, OrderedNodeWrapper Ptr)
|
||||
: NodeIterator(Base, IRBase, Ptr) {}
|
||||
AllNodesIterator(NodeIterator other)
|
||||
: NodeIterator(other) {} // Allow NodeIterator to be upgraded
|
||||
|
||||
AllNodesIterator operator++() {
|
||||
OrderedNodeHeader* RealNode = reinterpret_cast<OrderedNodeHeader*>(Node.GetNode(BaseList));
|
||||
auto IROp = Node.GetNode(BaseList)->Op(IRList);
|
||||
|
||||
// If this is the last node of a codeblock, we need to continue to the next block
|
||||
if (IROp->Op == OP_ENDBLOCK) {
|
||||
auto EndBlock = IROp->C<IROp_EndBlock>();
|
||||
|
||||
auto CurrentBlock = EndBlock->BlockHeader.GetNode(BaseList);
|
||||
Node = CurrentBlock->Header.Next;
|
||||
} else if (IROp->Op == OP_CODEBLOCK) {
|
||||
auto CodeBlock = IROp->C<IROp_CodeBlock>();
|
||||
|
||||
Node = CodeBlock->Begin;
|
||||
} else {
|
||||
Node = RealNode->Next;
|
||||
}
|
||||
|
||||
return *this;
|
||||
}
|
||||
|
||||
AllNodesIterator operator--() {
|
||||
auto IROp = Node.GetNode(BaseList)->Op(IRList);
|
||||
|
||||
if (IROp->Op == OP_BEGINBLOCK) {
|
||||
auto BeginBlock = IROp->C<IROp_EndBlock>();
|
||||
|
||||
Node = BeginBlock->BlockHeader;
|
||||
} else if (IROp->Op == OP_CODEBLOCK) {
|
||||
auto PrevBlockWrapper = Node.GetNode(BaseList)->Header.Previous;
|
||||
auto PrevCodeBlock = PrevBlockWrapper.GetNode(BaseList)->Op(IRList)->C<IROp_CodeBlock>();
|
||||
|
||||
Node = PrevCodeBlock->Last;
|
||||
} else {
|
||||
Node = Node.GetNode(BaseList)->Header.Previous;
|
||||
}
|
||||
|
||||
return *this;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static AllNodesIterator Invalid() {
|
||||
return AllNodesIterator(0, 0);
|
||||
}
|
||||
};
|
||||
|
||||
class IRListView;
|
||||
class IREmitter;
|
||||
|
||||
void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData);
|
||||
fextl::unique_ptr<IREmitter> Parse(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator, fextl::stringstream &MapsStream);
|
||||
template<typename Type>
|
||||
inline NodeID NodeWrapperBase<Type>::ID() const {
|
||||
return NodeID(NodeOffset / sizeof(IR::OrderedNode));
|
||||
}
|
||||
|
||||
bool IsFragmentExit(FEXCore::IR::IROps Op);
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op);
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocationData* RAData);
|
||||
fextl::unique_ptr<IREmitter> Parse(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, fextl::stringstream& MapsStream);
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
template<>
|
||||
struct std::hash<FEXCore::IR::NodeID> {
|
||||
size_t operator()(const FEXCore::IR::NodeID& ID) const noexcept {
|
||||
return std::hash<FEXCore::IR::NodeID::value_type> {}(ID.Value);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::NodeID::value_type>;
|
||||
|
||||
// Pass-through the underlying value, so IDs can
|
||||
// be formatted like any integral value.
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
|
||||
return Base::format(ID.Value, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
|
||||
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
|
||||
return Base::format(Class.Val, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
|
||||
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
|
||||
return Base::format(Fence.Val, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::OpSize> : fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>> {
|
||||
using Base = fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>>;
|
||||
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::OpSize& OpSize, FormatContext& ctx) const {
|
||||
return Base::format(FEXCore::ToUnderlying(OpSize), ctx);
|
||||
}
|
||||
};
|
||||
@@ -456,6 +456,13 @@
|
||||
"DestSize": "4"
|
||||
},
|
||||
|
||||
"GPR = LoadDF": {
|
||||
"Desc": ["Loads the decimal flag from the context object in -1/1",
|
||||
"representation for easy consumption"
|
||||
],
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
"GPR = LoadFlag u32:$Flag": {
|
||||
"Desc": ["Loads an x86-64 flag from the context object",
|
||||
"Specialized to allow flexible implementation of flag handling"
|
||||
@@ -596,6 +603,15 @@
|
||||
"Ensures the memory operations are globally visible"
|
||||
],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"Prefetch i1:$ForStore, i1:$Stream, i8:$CacheLevel, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a cacheline prefetch operation"
|
||||
],
|
||||
"EmitValidation": [
|
||||
"_CacheLevel > 0 && _CacheLevel < 4"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
}
|
||||
},
|
||||
"Atomic": {
|
||||
@@ -626,7 +642,6 @@
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"NumElements": "2",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
|
||||
@@ -1020,6 +1035,14 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"CondSubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2, CondClass:$Cond, u8:$FalseNZCV": {
|
||||
"Desc": ["If condition is true, set NZCV per difference of GPRs, else force NZCV to a constant."],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AdcWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Adds and set NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"HasSideEffects": true,
|
||||
@@ -1079,6 +1102,11 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"CmpPairZ OpSize:#Size, GPRPair:$Src1, GPRPair:$Src2": {
|
||||
"Desc": ["Compares register pairs and sets Z accordingly, preserving N/Z/V.",
|
||||
"This accelerates cmpxchg."],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"SubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the difference of two GPRs. ",
|
||||
"Carry flag uses arm64 definition, inverted x86.",
|
||||
@@ -1133,6 +1161,13 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = XornShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
|
||||
"Desc": [ "Integer binary exclusive or not with shifted register"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = And OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary and"
|
||||
],
|
||||
@@ -1183,6 +1218,12 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = ShiftFlags OpSize:$Size, GPR:$Result, GPR:$Src1, ShiftType:$Shift, GPR:$Src2, GPR:$PFInput": {
|
||||
"Desc": ["Set NZCV flags for specified variable integer shift with given result.",
|
||||
"Returns updated raw PF."],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
"GPR = Ror OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer rotate right"
|
||||
],
|
||||
@@ -1606,13 +1647,6 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"FPR = VectorZero u8:#RegisterSize": {
|
||||
"Desc": ["Generates a vector zero",
|
||||
"Useful to generate a zero vector without any previous dependencies"
|
||||
],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"FPR = VectorImm u8:#RegisterSize, u8:#ElementSize, u8:$Immediate, u8:$ShiftAmount{0}": {
|
||||
"Desc": ["Generates a vector with each element containg the immediate zexted"
|
||||
],
|
||||
|
||||
@@ -6,9 +6,10 @@ tags: ir|dumper
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
|
||||
#include <algorithm>
|
||||
@@ -16,7 +17,7 @@ $end_info$
|
||||
#include <ostream>
|
||||
#include <stdint.h>
|
||||
#include <string_view>
|
||||
#include <iomanip>
|
||||
#include <iomanip>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
#define IROP_GETNAME_IMPL
|
||||
@@ -28,56 +29,36 @@ namespace FEXCore::IR {
|
||||
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, const SHA256Sum &Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, const SHA256Sum& Arg) {
|
||||
*out << "sha256:";
|
||||
for(auto byte: Arg.data)
|
||||
for (auto byte : Arg.data) {
|
||||
*out << std::hex << std::setfill('0') << std::setw(2) << (unsigned int)byte;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, uint64_t Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, uint64_t Arg) {
|
||||
*out << "#0x" << std::hex << Arg;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, const char* Arg) {
|
||||
*out << Arg;
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, const char* Arg) {
|
||||
*out << Arg;
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, CondClassType Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, CondClassType Arg) {
|
||||
if (Arg == COND_AL) {
|
||||
*out << "ALWAYS";
|
||||
return;
|
||||
}
|
||||
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
"UGE",
|
||||
"ULT",
|
||||
"MI",
|
||||
"PL",
|
||||
"VS",
|
||||
"VC",
|
||||
"UGT",
|
||||
"ULE",
|
||||
"SGE",
|
||||
"SLT",
|
||||
"SGT",
|
||||
"SLE",
|
||||
"ANDZ",
|
||||
"ANDNZ",
|
||||
"FLU",
|
||||
"FGE",
|
||||
"FLEU",
|
||||
"FGT",
|
||||
"FU",
|
||||
"FNU"
|
||||
};
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {"EQ", "NEQ", "UGE", "ULT", "MI", "PL", "VS", "VC",
|
||||
"UGT", "ULE", "SGE", "SLT", "SGT", "SLE", "ANDZ", "ANDNZ",
|
||||
"FLU", "FGE", "FLEU", "FGT", "FU", "FNU"};
|
||||
|
||||
*out << CondNames[Arg];
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, MemOffsetType Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, MemOffsetType Arg) {
|
||||
static constexpr std::array<std::string_view, 3> Names = {
|
||||
"SXTX",
|
||||
"UXTW",
|
||||
@@ -87,22 +68,23 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
*out << Names[Arg];
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, RegisterClassType Arg) {
|
||||
if (Arg == GPRClass.Val)
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, RegisterClassType Arg) {
|
||||
if (Arg == GPRClass.Val) {
|
||||
*out << "GPR";
|
||||
else if (Arg == GPRFixedClass.Val)
|
||||
} else if (Arg == GPRFixedClass.Val) {
|
||||
*out << "GPRFixed";
|
||||
else if (Arg == FPRClass.Val)
|
||||
} else if (Arg == FPRClass.Val) {
|
||||
*out << "FPR";
|
||||
else if (Arg == FPRFixedClass.Val)
|
||||
} else if (Arg == FPRFixedClass.Val) {
|
||||
*out << "FPRFixed";
|
||||
else if (Arg == GPRPairClass.Val)
|
||||
} else if (Arg == GPRPairClass.Val) {
|
||||
*out << "GPRPair";
|
||||
else
|
||||
} else {
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, IRListView const* IR, OrderedNodeWrapper Arg, IR::RegisterAllocationData *RAData) {
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg, IR::RegisterAllocationData* RAData) {
|
||||
auto [CodeNode, IROp] = IR->at(Arg)();
|
||||
const auto ArgID = Arg.ID();
|
||||
|
||||
@@ -114,14 +96,14 @@ static void PrintArg(fextl::stringstream *out, IRListView const* IR, OrderedNode
|
||||
auto PhyReg = RAData->GetNodeRegister(ArgID);
|
||||
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
}
|
||||
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
@@ -148,131 +130,126 @@ static void PrintArg(fextl::stringstream *out, IRListView const* IR, OrderedNode
|
||||
if (NumElements > 1) {
|
||||
*out << "v" << std::dec << NumElements;
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::FenceType Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, FEXCore::IR::FenceType Arg) {
|
||||
if (Arg == IR::Fence_Load) {
|
||||
*out << "Loads";
|
||||
}
|
||||
else if (Arg == IR::Fence_Store) {
|
||||
} else if (Arg == IR::Fence_Store) {
|
||||
*out << "Stores";
|
||||
}
|
||||
else if (Arg == IR::Fence_LoadStore) {
|
||||
} else if (Arg == IR::Fence_LoadStore) {
|
||||
*out << "LoadStores";
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
*out << "<Unknown Fence Type>";
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::RoundType Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, FEXCore::IR::RoundType Arg) {
|
||||
switch (Arg) {
|
||||
case FEXCore::IR::Round_Nearest: *out << "Nearest"; break;
|
||||
case FEXCore::IR::Round_Negative_Infinity: *out << "-Inf"; break;
|
||||
case FEXCore::IR::Round_Positive_Infinity: *out << "+Inf"; break;
|
||||
case FEXCore::IR::Round_Towards_Zero: *out << "Towards Zero"; break;
|
||||
case FEXCore::IR::Round_Host: *out << "Host"; break;
|
||||
default: *out << "<Unknown Round Type>"; break;
|
||||
case FEXCore::IR::Round_Nearest: *out << "Nearest"; break;
|
||||
case FEXCore::IR::Round_Negative_Infinity: *out << "-Inf"; break;
|
||||
case FEXCore::IR::Round_Positive_Infinity: *out << "+Inf"; break;
|
||||
case FEXCore::IR::Round_Towards_Zero: *out << "Towards Zero"; break;
|
||||
case FEXCore::IR::Round_Host: *out << "Host"; break;
|
||||
default: *out << "<Unknown Round Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::SyscallFlags Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, FEXCore::IR::SyscallFlags Arg) {
|
||||
switch (Arg) {
|
||||
case FEXCore::IR::SyscallFlags::DEFAULT: *out << "Default"; break;
|
||||
case FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH: *out << "Optimize Through"; break;
|
||||
case FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY: *out << "No Sync State on Entry"; break;
|
||||
case FEXCore::IR::SyscallFlags::NORETURN: *out << "No Return"; break;
|
||||
case FEXCore::IR::SyscallFlags::NOSIDEEFFECTS: *out << "No Side Effects"; break;
|
||||
default: *out << "<Unknown Round Type>"; break;
|
||||
case FEXCore::IR::SyscallFlags::DEFAULT: *out << "Default"; break;
|
||||
case FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH: *out << "Optimize Through"; break;
|
||||
case FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY: *out << "No Sync State on Entry"; break;
|
||||
case FEXCore::IR::SyscallFlags::NORETURN: *out << "No Return"; break;
|
||||
case FEXCore::IR::SyscallFlags::NOSIDEEFFECTS: *out << "No Side Effects"; break;
|
||||
default: *out << "<Unknown Round Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::NamedVectorConstant Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, FEXCore::IR::NamedVectorConstant Arg) {
|
||||
*out << [Arg] {
|
||||
// clang-format off
|
||||
switch (Arg) {
|
||||
case NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U16_INDEX:
|
||||
return "u16_incremental_index";
|
||||
case NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER:
|
||||
return "u16_incremental_index_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT:
|
||||
return "addsubps_invert";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT_UPPER:
|
||||
return "addsubps_invert_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPD_INVERT:
|
||||
return "addsubpd_invert";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPD_INVERT_UPPER:
|
||||
return "addsubpd_invert_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_MOVMSKPS_SHIFT:
|
||||
return "movmskps_shift";
|
||||
case NamedVectorConstant::NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE:
|
||||
return "aeskeygenassist_swizzle";
|
||||
case NamedVectorConstant::NAMED_VECTOR_ZERO:
|
||||
return "vectorzero";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_ONE:
|
||||
return "x87_1_0";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_LOG2_10:
|
||||
return "x87_log2_10";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_LOG2_E:
|
||||
return "x87_log2_e";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_PI:
|
||||
return "x87_pi";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_LOG10_2:
|
||||
return "x87_log10_2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_LOG_2:
|
||||
return "x87_log2";
|
||||
default:
|
||||
return "<Unknown Named Vector Constant>";
|
||||
}
|
||||
// clang-format on
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, FEXCore::IR::OpSize Arg) {
|
||||
switch (Arg) {
|
||||
case FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U16_INDEX: {
|
||||
*out << "u16_incremental_index";
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER: {
|
||||
*out << "u16_incremental_index_upper";
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT: {
|
||||
*out << "addsubps_invert";
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT_UPPER: {
|
||||
*out << "addsubps_invert_upper";
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_PADDSUBPD_INVERT: {
|
||||
*out << "addsubpd_invert";
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_PADDSUBPD_INVERT_UPPER: {
|
||||
*out << "addsubpd_invert_upper";
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_MOVMSKPS_SHIFT: {
|
||||
*out << "movmskps_shift";
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE: {
|
||||
*out << "aeskeygenassist_swizzle";
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO: {
|
||||
*out << "vectorzero";
|
||||
break;
|
||||
}
|
||||
default: *out << "<Unknown Named Vector Constant>"; break;
|
||||
case OpSize::i8Bit: *out << "i8"; break;
|
||||
case OpSize::i16Bit: *out << "i16"; break;
|
||||
case OpSize::i32Bit: *out << "i32"; break;
|
||||
case OpSize::i64Bit: *out << "i64"; break;
|
||||
case OpSize::i128Bit: *out << "i128"; break;
|
||||
case OpSize::i256Bit: *out << "i256"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::OpSize Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, FEXCore::IR::FloatCompareOp Arg) {
|
||||
switch (Arg) {
|
||||
case OpSize::i8Bit: *out << "i8"; break;
|
||||
case OpSize::i16Bit: *out << "i16"; break;
|
||||
case OpSize::i32Bit: *out << "i32"; break;
|
||||
case OpSize::i64Bit: *out << "i64"; break;
|
||||
case OpSize::i128Bit: *out << "i128"; break;
|
||||
case OpSize::i256Bit: *out << "i256"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
case FloatCompareOp::EQ: *out << "FEQ"; break;
|
||||
case FloatCompareOp::LT: *out << "FLT"; break;
|
||||
case FloatCompareOp::LE: *out << "FLE"; break;
|
||||
case FloatCompareOp::UNO: *out << "UNO"; break;
|
||||
case FloatCompareOp::NEQ: *out << "NEQ"; break;
|
||||
case FloatCompareOp::ORD: *out << "ORD"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::FloatCompareOp Arg) {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: *out << "FEQ"; break;
|
||||
case FloatCompareOp::LT: *out << "FLT"; break;
|
||||
case FloatCompareOp::LE: *out << "FLE"; break;
|
||||
case FloatCompareOp::UNO: *out << "UNO"; break;
|
||||
case FloatCompareOp::NEQ: *out << "NEQ"; break;
|
||||
case FloatCompareOp::ORD: *out << "ORD"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::BreakDefinition Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, FEXCore::IR::BreakDefinition Arg) {
|
||||
*out << "{" << Arg.ErrorRegister << ".";
|
||||
*out << static_cast<uint32_t>(Arg.Signal) << ".";
|
||||
*out << static_cast<uint32_t>(Arg.TrapNumber) << ".";
|
||||
*out << static_cast<uint32_t>(Arg.si_code) << "}";
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::ShiftType Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, FEXCore::IR::ShiftType Arg) {
|
||||
switch (Arg) {
|
||||
case ShiftType::LSL: *out << "LSL"; break;
|
||||
case ShiftType::LSR: *out << "LSR"; break;
|
||||
case ShiftType::ASR: *out << "ASR"; break;
|
||||
case ShiftType::ROR: *out << "ROR"; break;
|
||||
default: *out << "<Unknown Shift Type>"; break;
|
||||
case ShiftType::LSL: *out << "LSL"; break;
|
||||
case ShiftType::LSR: *out << "LSR"; break;
|
||||
case ShiftType::ASR: *out << "ASR"; break;
|
||||
case ShiftType::ROR: *out << "ROR"; break;
|
||||
default: *out << "<Unknown Shift Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData) {
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocationData* RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
int8_t CurrentIndent = 0;
|
||||
@@ -284,7 +261,8 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
|
||||
++CurrentIndent;
|
||||
AddIndent();
|
||||
*out << "(%0) " << "IRHeader ";
|
||||
*out << "(%0) "
|
||||
<< "IRHeader ";
|
||||
*out << "%" << HeaderOp->Blocks.ID() << ", ";
|
||||
*out << "#" << std::dec << HeaderOp->OriginalRIP << ", ";
|
||||
*out << "#" << std::dec << HeaderOp->BlockCount << ", ";
|
||||
@@ -295,7 +273,8 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
auto BlockIROp = BlockHeader->C<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
AddIndent();
|
||||
*out << "(%" << IR->GetID(BlockNode) << ") " << "CodeBlock ";
|
||||
*out << "(%" << IR->GetID(BlockNode) << ") "
|
||||
<< "CodeBlock ";
|
||||
|
||||
*out << "%" << BlockIROp->Begin.ID() << ", ";
|
||||
*out << "%" << BlockIROp->Last.ID() << std::endl;
|
||||
@@ -325,14 +304,14 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
}
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
|
||||
@@ -348,8 +327,7 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
}
|
||||
|
||||
*out << " = ";
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
|
||||
uint32_t ElementSize = IROp->ElementSize;
|
||||
if (!IROp->ElementSize) {
|
||||
@@ -369,19 +347,18 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
}
|
||||
*out << Name;
|
||||
|
||||
#define IROP_ARGPRINTER_HELPER
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
default: *out << "<Unknown Args>"; break;
|
||||
}
|
||||
|
||||
//*out << " (" << std::dec << CodeNode->GetUses() << ")";
|
||||
|
||||
*out << "\n";
|
||||
#define IROP_ARGPRINTER_HELPER
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
default: *out << "<Unknown Args>"; break;
|
||||
}
|
||||
|
||||
//*out << " (" << std::dec << CodeNode->GetUses() << ")";
|
||||
|
||||
*out << "\n";
|
||||
}
|
||||
|
||||
CurrentIndent = std::max(0, CurrentIndent - 1);
|
||||
}
|
||||
}
|
||||
|
||||
CurrentIndent = std::max(0, CurrentIndent - 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -9,7 +9,6 @@ $end_info$
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
@@ -21,77 +20,69 @@ namespace FEXCore::IR {
|
||||
|
||||
bool IsFragmentExit(FEXCore::IR::IROps Op) {
|
||||
switch (Op) {
|
||||
case OP_EXITFUNCTION:
|
||||
case OP_BREAK:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
case OP_EXITFUNCTION:
|
||||
case OP_BREAK: return true;
|
||||
default: return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op) {
|
||||
switch(Op) {
|
||||
case OP_JUMP:
|
||||
case OP_CONDJUMP:
|
||||
return true;
|
||||
default:
|
||||
return IsFragmentExit(Op);
|
||||
switch (Op) {
|
||||
case OP_JUMP:
|
||||
case OP_CONDJUMP: return true;
|
||||
default: return IsFragmentExit(Op);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(OrderedNode *Node) {
|
||||
FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(OrderedNode* Node) {
|
||||
auto Class = GetOpRegClass(Node);
|
||||
switch (Class) {
|
||||
case GPRClass:
|
||||
case GPRPairClass:
|
||||
case FPRClass:
|
||||
case GPRFixedClass:
|
||||
case FPRFixedClass:
|
||||
case InvalidClass:
|
||||
return Class;
|
||||
default: break;
|
||||
case GPRClass:
|
||||
case GPRPairClass:
|
||||
case FPRClass:
|
||||
case GPRFixedClass:
|
||||
case FPRFixedClass:
|
||||
case InvalidClass: return Class;
|
||||
default: break;
|
||||
}
|
||||
|
||||
// Complex case, needs to be handled on an op by op basis
|
||||
uintptr_t DataBegin = DualListData.DataBegin();
|
||||
|
||||
FEXCore::IR::IROp_Header *IROp = Node->Op(DataBegin);
|
||||
FEXCore::IR::IROp_Header* IROp = Node->Op(DataBegin);
|
||||
|
||||
switch (IROp->Op) {
|
||||
case IROps::OP_LOADREGISTER: {
|
||||
auto Op = IROp->C<IROp_LoadRegister>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_LOADCONTEXT: {
|
||||
auto Op = IROp->C<IROp_LoadContext>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_LOADCONTEXTINDEXED: {
|
||||
auto Op = IROp->C<IROp_LoadContextIndexed>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_FILLREGISTER: {
|
||||
auto Op = IROp->C<IROp_FillRegister>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_LOADMEM: {
|
||||
auto Op = IROp->C<IROp_LoadMem>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_LOADMEMTSO: {
|
||||
auto Op = IROp->C<IROp_LoadMemTSO>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled op type: {} {} in argument class validation",
|
||||
ToUnderlying(IROp->Op), GetOpName(Node));
|
||||
break;
|
||||
case IROps::OP_LOADREGISTER: {
|
||||
auto Op = IROp->C<IROp_LoadRegister>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_LOADCONTEXT: {
|
||||
auto Op = IROp->C<IROp_LoadContext>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_LOADCONTEXTINDEXED: {
|
||||
auto Op = IROp->C<IROp_LoadContextIndexed>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_FILLREGISTER: {
|
||||
auto Op = IROp->C<IROp_FillRegister>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_LOADMEM: {
|
||||
auto Op = IROp->C<IROp_LoadMem>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
case IROps::OP_LOADMEMTSO: {
|
||||
auto Op = IROp->C<IROp_LoadMemTSO>();
|
||||
return Op->Class;
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled op type: {} {} in argument class validation", ToUnderlying(IROp->Op), GetOpName(Node)); break;
|
||||
}
|
||||
return InvalidClass;
|
||||
}
|
||||
@@ -106,7 +97,7 @@ void IREmitter::ResetWorkingList() {
|
||||
CurrentCodeBlock = nullptr;
|
||||
}
|
||||
|
||||
void IREmitter::ReplaceAllUsesWithRange(OrderedNode *Node, OrderedNode *NewNode, AllNodesIterator Begin, AllNodesIterator End) {
|
||||
void IREmitter::ReplaceAllUsesWithRange(OrderedNode* Node, OrderedNode* NewNode, AllNodesIterator Begin, AllNodesIterator End) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
auto NodeId = Node->Wrapped(ListBegin).ID();
|
||||
|
||||
@@ -131,23 +122,23 @@ void IREmitter::ReplaceAllUsesWithRange(OrderedNode *Node, OrderedNode *NewNode,
|
||||
}
|
||||
}
|
||||
|
||||
void IREmitter::ReplaceNodeArgument(OrderedNode *Node, uint8_t Arg, OrderedNode *NewArg) {
|
||||
void IREmitter::ReplaceNodeArgument(OrderedNode* Node, uint8_t Arg, OrderedNode* NewArg) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
uintptr_t DataBegin = DualListData.DataBegin();
|
||||
|
||||
FEXCore::IR::IROp_Header *IROp = Node->Op(DataBegin);
|
||||
FEXCore::IR::IROp_Header* IROp = Node->Op(DataBegin);
|
||||
OrderedNodeWrapper OldArgWrapper = IROp->Args[Arg];
|
||||
OrderedNode *OldArg = OldArgWrapper.GetNode(ListBegin);
|
||||
OrderedNode* OldArg = OldArgWrapper.GetNode(ListBegin);
|
||||
OldArg->RemoveUse();
|
||||
NewArg->AddUse();
|
||||
IROp->Args[Arg].NodeOffset = NewArg->Wrapped(ListBegin).NodeOffset;
|
||||
}
|
||||
|
||||
void IREmitter::RemoveArgUses(OrderedNode *Node) {
|
||||
void IREmitter::RemoveArgUses(OrderedNode* Node) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
uintptr_t DataBegin = DualListData.DataBegin();
|
||||
|
||||
FEXCore::IR::IROp_Header *IROp = Node->Op(DataBegin);
|
||||
FEXCore::IR::IROp_Header* IROp = Node->Op(DataBegin);
|
||||
|
||||
const uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
@@ -156,7 +147,7 @@ void IREmitter::RemoveArgUses(OrderedNode *Node) {
|
||||
}
|
||||
}
|
||||
|
||||
void IREmitter::Remove(OrderedNode *Node) {
|
||||
void IREmitter::Remove(OrderedNode* Node) {
|
||||
RemoveArgUses(Node);
|
||||
|
||||
Node->Unlink(DualListData.ListBegin());
|
||||
@@ -175,8 +166,9 @@ IREmitter::IRPair<IROp_CodeBlock> IREmitter::CreateNewCodeBlockAfter(OrderedNode
|
||||
// Find last block
|
||||
auto LastBlock = CurrentCodeBlock;
|
||||
|
||||
while (LastBlock->Header.Next.GetNode(DualListData.ListBegin()) != InvalidNode)
|
||||
while (LastBlock->Header.Next.GetNode(DualListData.ListBegin()) != InvalidNode) {
|
||||
LastBlock = LastBlock->Header.Next.GetNode(DualListData.ListBegin());
|
||||
}
|
||||
|
||||
// Append it after the last block
|
||||
LinkCodeBlocks(LastBlock, CodeNode);
|
||||
@@ -187,34 +179,34 @@ IREmitter::IRPair<IROp_CodeBlock> IREmitter::CreateNewCodeBlockAfter(OrderedNode
|
||||
return CodeNode;
|
||||
}
|
||||
|
||||
void IREmitter::SetCurrentCodeBlock(OrderedNode *Node) {
|
||||
void IREmitter::SetCurrentCodeBlock(OrderedNode* Node) {
|
||||
CurrentCodeBlock = Node;
|
||||
LOGMAN_THROW_A_FMT(Node->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Node wasn't codeblock. It was '{}'", IR::GetName(Node->Op(DualListData.DataBegin())->Op));
|
||||
LOGMAN_THROW_A_FMT(Node->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Node wasn't codeblock. It was '{}'",
|
||||
IR::GetName(Node->Op(DualListData.DataBegin())->Op));
|
||||
SetWriteCursor(Node->Op(DualListData.DataBegin())->CW<IROp_CodeBlock>()->Begin.GetNode(DualListData.ListBegin()));
|
||||
}
|
||||
|
||||
void IREmitter::ReplaceWithConstant(OrderedNode *Node, uint64_t Value) {
|
||||
auto Header = Node->Op(DualListData.DataBegin());
|
||||
void IREmitter::ReplaceWithConstant(OrderedNode* Node, uint64_t Value) {
|
||||
auto Header = Node->Op(DualListData.DataBegin());
|
||||
|
||||
if (IRSizes[Header->Op] >= sizeof(IROp_Constant)) {
|
||||
// Unlink any arguments the node currently has
|
||||
RemoveArgUses(Node);
|
||||
if (IRSizes[Header->Op] >= sizeof(IROp_Constant)) {
|
||||
// Unlink any arguments the node currently has
|
||||
RemoveArgUses(Node);
|
||||
|
||||
// Overwrite data with the new constant op
|
||||
Header->Op = OP_CONSTANT;
|
||||
auto Const = Header->CW<IROp_Constant>();
|
||||
Const->Constant = Value;
|
||||
} else {
|
||||
// Fallback path for when the node to overwrite is too small
|
||||
auto cursor = GetWriteCursor();
|
||||
SetWriteCursor(Node);
|
||||
// Overwrite data with the new constant op
|
||||
Header->Op = OP_CONSTANT;
|
||||
auto Const = Header->CW<IROp_Constant>();
|
||||
Const->Constant = Value;
|
||||
} else {
|
||||
// Fallback path for when the node to overwrite is too small
|
||||
auto cursor = GetWriteCursor();
|
||||
SetWriteCursor(Node);
|
||||
|
||||
auto NewNode = _Constant(Value);
|
||||
ReplaceAllUsesWith(Node, NewNode);
|
||||
auto NewNode = _Constant(Value);
|
||||
ReplaceAllUsesWith(Node, NewNode);
|
||||
|
||||
SetWriteCursor(cursor);
|
||||
}
|
||||
SetWriteCursor(cursor);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,7 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -17,36 +20,40 @@ class Pass;
|
||||
class PassManager;
|
||||
|
||||
class IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
|
||||
public:
|
||||
IREmitter(FEXCore::Utils::IntrusivePooledAllocator &ThreadAllocator)
|
||||
: DualListData {ThreadAllocator, 8 * 1024 * 1024} {
|
||||
ReownOrClaimBuffer();
|
||||
ResetWorkingList();
|
||||
}
|
||||
public:
|
||||
IREmitter(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator)
|
||||
: DualListData {ThreadAllocator, 8 * 1024 * 1024} {
|
||||
ReownOrClaimBuffer();
|
||||
ResetWorkingList();
|
||||
}
|
||||
|
||||
virtual ~IREmitter() = default;
|
||||
virtual ~IREmitter() = default;
|
||||
|
||||
void ReownOrClaimBuffer() {
|
||||
DualListData.ReownOrClaimBuffer();
|
||||
}
|
||||
void ReownOrClaimBuffer() {
|
||||
DualListData.ReownOrClaimBuffer();
|
||||
}
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
DualListData.DelayedDisownBuffer();
|
||||
}
|
||||
void DelayedDisownBuffer() {
|
||||
DualListData.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
IRListView ViewIR() { return IRListView(&DualListData, false); }
|
||||
IRListView *CreateIRCopy() { return new IRListView(&DualListData, true); }
|
||||
void ResetWorkingList();
|
||||
IRListView ViewIR() {
|
||||
return IRListView(&DualListData, false);
|
||||
}
|
||||
IRListView* CreateIRCopy() {
|
||||
return new IRListView(&DualListData, true);
|
||||
}
|
||||
void ResetWorkingList();
|
||||
|
||||
/**
|
||||
* @name IR allocation routines
|
||||
*
|
||||
* @{ */
|
||||
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNode *Node);
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNode* Node);
|
||||
|
||||
// These handlers add cost to the constructor and destructor
|
||||
// If it becomes an issue then blow them away
|
||||
@@ -66,119 +73,112 @@ friend class FEXCore::IR::PassManager;
|
||||
IRPair<IROp_Jump> _Jump() {
|
||||
return _Jump(InvalidNode);
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(OrderedNode *ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> _CondJump(OrderedNode* ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
return _CondJump(ssa0, _Constant(0), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(OrderedNode *ssa0, OrderedNode *ssa1, OrderedNode *ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> _CondJump(OrderedNode* ssa0, OrderedNode* ssa1, OrderedNode* ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
|
||||
}
|
||||
// TODO: Work to remove this implicit sized Select implementation.
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, OrderedNode *ssa0, OrderedNode *ssa1, OrderedNode *ssa2, OrderedNode *ssa3, uint8_t CompareSize = 0) {
|
||||
if (CompareSize == 0)
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, OrderedNode* ssa0, OrderedNode* ssa1, OrderedNode* ssa2, OrderedNode* ssa3, uint8_t CompareSize = 0) {
|
||||
if (CompareSize == 0) {
|
||||
CompareSize = std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa0), GetOpSize(ssa1)));
|
||||
}
|
||||
|
||||
return _Select(IR::SizeToOpSize(std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa2), GetOpSize(ssa3)))), IR::SizeToOpSize(CompareSize), CondClassType{Cond}, ssa0, ssa1, ssa2, ssa3);
|
||||
return _Select(IR::SizeToOpSize(std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa2), GetOpSize(ssa3)))),
|
||||
IR::SizeToOpSize(CompareSize), CondClassType {Cond}, ssa0, ssa1, ssa2, ssa3);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* ssa0, uint8_t Align = 1) {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
|
||||
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* ssa0, uint8_t Align = 1) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* Addr, OrderedNode* Value, uint8_t Align = 1) {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMemTSO> _StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
|
||||
IRPair<IROp_StoreMemTSO>
|
||||
_StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* Addr, OrderedNode* Value, uint8_t Align = 1) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
OrderedNode *Invalid() {
|
||||
OrderedNode* Invalid() {
|
||||
return InvalidNode;
|
||||
}
|
||||
|
||||
void SetJumpTarget(IR::IROp_Jump *Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
void SetJumpTarget(IR::IROp_Jump* Op, OrderedNode* Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
Op->Header.Args[0].NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetTrueJumpTarget(IR::IROp_CondJump *Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
void SetTrueJumpTarget(IR::IROp_CondJump* Op, OrderedNode* Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
Op->TrueBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetFalseJumpTarget(IR::IROp_CondJump *Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
void SetFalseJumpTarget(IR::IROp_CondJump* Op, OrderedNode* Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
Op->FalseBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
|
||||
void SetJumpTarget(IRPair<IROp_Jump> Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
void SetJumpTarget(IRPair<IROp_Jump> Op, OrderedNode* Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
Op.first->Header.Args[0].NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetTrueJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
void SetTrueJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode* Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
Op.first->TrueBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetFalseJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
void SetFalseJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode* Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
Op.first->FalseBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
|
||||
/** @} */
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNodeWrapper ssa) {
|
||||
OrderedNode *RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return WalkFindRegClass(RealNode);
|
||||
OrderedNode* RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return WalkFindRegClass(RealNode);
|
||||
}
|
||||
|
||||
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t *Constant = nullptr) {
|
||||
OrderedNode *RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
FEXCore::IR::IROp_Header *IROp = RealNode->Op(DualListData.DataBegin());
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
if (Constant) *Constant = Op->Constant;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t* Constant = nullptr) {
|
||||
OrderedNode* RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
if (Constant) {
|
||||
*Constant = Op->Constant;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool IsValueInlineConstant(OrderedNodeWrapper ssa) {
|
||||
OrderedNode *RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
FEXCore::IR::IROp_Header *IROp = RealNode->Op(DualListData.DataBegin());
|
||||
if (IROp->Op == OP_INLINECONSTANT) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
OrderedNode* RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
if (IROp->Op == OP_INLINECONSTANT) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
FEXCore::IR::IROp_Header *GetOpHeader(OrderedNodeWrapper ssa) {
|
||||
OrderedNode *RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
FEXCore::IR::IROp_Header* GetOpHeader(OrderedNodeWrapper ssa) {
|
||||
OrderedNode* RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return RealNode->Op(DualListData.DataBegin());
|
||||
}
|
||||
|
||||
OrderedNode *UnwrapNode(OrderedNodeWrapper ssa) {
|
||||
OrderedNode* UnwrapNode(OrderedNodeWrapper ssa) {
|
||||
return ssa.GetNode(DualListData.ListBegin());
|
||||
}
|
||||
|
||||
OrderedNodeWrapper WrapNode(OrderedNode *node) {
|
||||
OrderedNodeWrapper WrapNode(OrderedNode* node) {
|
||||
return node->Wrapped(DualListData.ListBegin());
|
||||
}
|
||||
|
||||
@@ -189,23 +189,23 @@ friend class FEXCore::IR::PassManager;
|
||||
// Overwrite a node with a constant
|
||||
// Depending on what node has been overwritten, there might be some unallocated space around the node
|
||||
// Because we are overwriting the node, we don't have to worry about update all the arguments which use it
|
||||
void ReplaceWithConstant(OrderedNode *Node, uint64_t Value);
|
||||
void ReplaceWithConstant(OrderedNode* Node, uint64_t Value);
|
||||
|
||||
void ReplaceAllUsesWithRange(OrderedNode *Node, OrderedNode *NewNode, AllNodesIterator Begin, AllNodesIterator End);
|
||||
void ReplaceAllUsesWithRange(OrderedNode* Node, OrderedNode* NewNode, AllNodesIterator Begin, AllNodesIterator End);
|
||||
|
||||
void ReplaceUsesWithAfter(OrderedNode *Node, OrderedNode *NewNode, AllNodesIterator After) {
|
||||
void ReplaceUsesWithAfter(OrderedNode* Node, OrderedNode* NewNode, AllNodesIterator After) {
|
||||
++After;
|
||||
ReplaceAllUsesWithRange(Node, NewNode, After, AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin()));
|
||||
}
|
||||
|
||||
void ReplaceUsesWithAfter(OrderedNode *Node, OrderedNode *NewNode, OrderedNode *After) {
|
||||
void ReplaceUsesWithAfter(OrderedNode* Node, OrderedNode* NewNode, OrderedNode* After) {
|
||||
auto Wrapped = After->Wrapped(DualListData.ListBegin());
|
||||
AllNodesIterator It = AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin(), Wrapped);
|
||||
|
||||
ReplaceUsesWithAfter(Node, NewNode, It);
|
||||
}
|
||||
|
||||
void ReplaceAllUsesWith(OrderedNode *Node, OrderedNode *NewNode) {
|
||||
void ReplaceAllUsesWith(OrderedNode* Node, OrderedNode* NewNode) {
|
||||
auto Start = AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin(), Node->Wrapped(DualListData.ListBegin()));
|
||||
|
||||
ReplaceAllUsesWithRange(Node, NewNode, Start, AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin()));
|
||||
@@ -220,34 +220,45 @@ friend class FEXCore::IR::PassManager;
|
||||
}
|
||||
}
|
||||
|
||||
void ReplaceNodeArgument(OrderedNode *Node, uint8_t Arg, OrderedNode *NewArg);
|
||||
void ReplaceNodeArgument(OrderedNode* Node, uint8_t Arg, OrderedNode* NewArg);
|
||||
|
||||
void Remove(OrderedNode *Node);
|
||||
void Remove(OrderedNode* Node);
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, OrderedNode *Src);
|
||||
OrderedNode *GetPackedRFLAG(bool Lower8);
|
||||
void SetPackedRFLAG(bool Lower8, OrderedNode* Src);
|
||||
OrderedNode* GetPackedRFLAG(bool Lower8);
|
||||
|
||||
void CopyData(IREmitter const &rhs) {
|
||||
LOGMAN_THROW_A_FMT(rhs.DualListData.DataBackingSize() <= DualListData.DataBackingSize(), "Trying to take ownership of data that is too large");
|
||||
LOGMAN_THROW_A_FMT(rhs.DualListData.ListBackingSize() <= DualListData.ListBackingSize(), "Trying to take ownership of data that is too large");
|
||||
void CopyData(const IREmitter& rhs) {
|
||||
LOGMAN_THROW_A_FMT(rhs.DualListData.DataBackingSize() <= DualListData.DataBackingSize(), "Trying to take ownership of data that is too "
|
||||
"large");
|
||||
LOGMAN_THROW_A_FMT(rhs.DualListData.ListBackingSize() <= DualListData.ListBackingSize(), "Trying to take ownership of data that is too "
|
||||
"large");
|
||||
DualListData.CopyData(rhs.DualListData);
|
||||
InvalidNode = rhs.InvalidNode->Wrapped(rhs.DualListData.ListBegin()).GetNode(DualListData.ListBegin());
|
||||
CurrentWriteCursor = rhs.CurrentWriteCursor;
|
||||
CodeBlocks = rhs.CodeBlocks;
|
||||
for (auto& CodeBlock: CodeBlocks) {
|
||||
for (auto& CodeBlock : CodeBlocks) {
|
||||
CodeBlock = CodeBlock->Wrapped(rhs.DualListData.ListBegin()).GetNode(DualListData.ListBegin());
|
||||
}
|
||||
}
|
||||
|
||||
void SetWriteCursor(OrderedNode *Node) {
|
||||
void SetWriteCursor(OrderedNode* Node) {
|
||||
CurrentWriteCursor = Node;
|
||||
}
|
||||
|
||||
OrderedNode *GetWriteCursor() {
|
||||
// Set cursor to write before Node
|
||||
void SetWriteCursorBefore(OrderedNode* Node) {
|
||||
auto IR = ViewIR();
|
||||
auto Before = IR.at(Node);
|
||||
--Before;
|
||||
|
||||
SetWriteCursor(std::get<0>(*Before));
|
||||
}
|
||||
|
||||
OrderedNode* GetWriteCursor() {
|
||||
return CurrentWriteCursor;
|
||||
}
|
||||
|
||||
OrderedNode *GetCurrentBlock() {
|
||||
OrderedNode* GetCurrentBlock() {
|
||||
return CurrentCodeBlock;
|
||||
}
|
||||
|
||||
@@ -267,7 +278,7 @@ friend class FEXCore::IR::PassManager;
|
||||
|
||||
CodeBlocks.emplace_back(CodeNode);
|
||||
|
||||
SetWriteCursor(nullptr);// Orphan from any future nodes
|
||||
SetWriteCursor(nullptr); // Orphan from any future nodes
|
||||
|
||||
auto Begin = _BeginBlock(CodeNode);
|
||||
CodeNode.first->Begin = Begin.Node->Wrapped(DualListData.ListBegin());
|
||||
@@ -289,64 +300,66 @@ friend class FEXCore::IR::PassManager;
|
||||
*
|
||||
* @{ */
|
||||
/** @} */
|
||||
void LinkCodeBlocks(OrderedNode *CodeNode, OrderedNode *Next) {
|
||||
void LinkCodeBlocks(OrderedNode* CodeNode, OrderedNode* Next) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
FEXCore::IR::IROp_CodeBlock *CurrentIROp =
|
||||
FEXCore::IR::IROp_CodeBlock* CurrentIROp =
|
||||
#endif
|
||||
CodeNode->Op(DualListData.DataBegin())->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
CodeNode->Op(DualListData.DataBegin())->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
LOGMAN_THROW_A_FMT(CurrentIROp->Header.Op == IROps::OP_CODEBLOCK, "Invalid");
|
||||
|
||||
CodeNode->append(DualListData.ListBegin(), Next);
|
||||
}
|
||||
|
||||
IRPair<IROp_CodeBlock> CreateNewCodeBlockAtEnd() { return CreateNewCodeBlockAfter(nullptr); }
|
||||
IRPair<IROp_CodeBlock> CreateNewCodeBlockAtEnd() {
|
||||
return CreateNewCodeBlockAfter(nullptr);
|
||||
}
|
||||
IRPair<IROp_CodeBlock> CreateNewCodeBlockAfter(OrderedNode* insertAfter);
|
||||
void SetCurrentCodeBlock(OrderedNode *Node);
|
||||
void SetCurrentCodeBlock(OrderedNode* Node);
|
||||
|
||||
protected:
|
||||
void RemoveArgUses(OrderedNode *Node);
|
||||
protected:
|
||||
void RemoveArgUses(OrderedNode* Node);
|
||||
|
||||
OrderedNode *CreateNode(IROp_Header *Op) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
size_t Size = sizeof(OrderedNode);
|
||||
void *Ptr = DualListData.ListAllocate(Size);
|
||||
OrderedNode *Node = new (Ptr) OrderedNode();
|
||||
Node->Header.Value.SetOffset(DualListData.DataBegin(), reinterpret_cast<uintptr_t>(Op));
|
||||
OrderedNode* CreateNode(IROp_Header* Op) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
size_t Size = sizeof(OrderedNode);
|
||||
void* Ptr = DualListData.ListAllocate(Size);
|
||||
OrderedNode* Node = new (Ptr) OrderedNode();
|
||||
Node->Header.Value.SetOffset(DualListData.DataBegin(), reinterpret_cast<uintptr_t>(Op));
|
||||
|
||||
if (CurrentWriteCursor) {
|
||||
CurrentWriteCursor->append(ListBegin, Node);
|
||||
}
|
||||
CurrentWriteCursor = Node;
|
||||
return Node;
|
||||
if (CurrentWriteCursor) {
|
||||
CurrentWriteCursor->append(ListBegin, Node);
|
||||
}
|
||||
CurrentWriteCursor = Node;
|
||||
return Node;
|
||||
}
|
||||
|
||||
OrderedNode *GetNode(uint32_t SSANode) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
OrderedNode *Node = reinterpret_cast<OrderedNode *>(ListBegin + SSANode * sizeof(OrderedNode));
|
||||
return Node;
|
||||
}
|
||||
OrderedNode* GetNode(uint32_t SSANode) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
OrderedNode* Node = reinterpret_cast<OrderedNode*>(ListBegin + SSANode * sizeof(OrderedNode));
|
||||
return Node;
|
||||
}
|
||||
|
||||
OrderedNode *EmplaceOrphanedNode(OrderedNode *OldNode) {
|
||||
size_t Size = sizeof(OrderedNode);
|
||||
OrderedNode *Ptr = reinterpret_cast<OrderedNode*>(DualListData.ListAllocate(Size));
|
||||
memcpy(Ptr, OldNode, Size);
|
||||
return Ptr;
|
||||
}
|
||||
OrderedNode* EmplaceOrphanedNode(OrderedNode* OldNode) {
|
||||
size_t Size = sizeof(OrderedNode);
|
||||
OrderedNode* Ptr = reinterpret_cast<OrderedNode*>(DualListData.ListAllocate(Size));
|
||||
memcpy(Ptr, OldNode, Size);
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
virtual void SaveNZCV(IROps Op) {
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
}
|
||||
virtual void SaveNZCV(IROps Op) {
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
}
|
||||
|
||||
OrderedNode *CurrentWriteCursor = nullptr;
|
||||
OrderedNode* CurrentWriteCursor = nullptr;
|
||||
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
DualIntrusiveAllocatorThreadPool DualListData;
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
DualIntrusiveAllocatorThreadPool DualListData;
|
||||
|
||||
OrderedNode *InvalidNode;
|
||||
OrderedNode *CurrentCodeBlock{};
|
||||
fextl::vector<OrderedNode*> CodeBlocks;
|
||||
uint64_t Entry;
|
||||
OrderedNode* InvalidNode;
|
||||
OrderedNode* CurrentCodeBlock {};
|
||||
fextl::vector<OrderedNode*> CodeBlocks;
|
||||
uint64_t Entry;
|
||||
};
|
||||
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,485 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstring>
|
||||
#include <tuple>
|
||||
#include <istream>
|
||||
#include <ostream>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
/**
|
||||
* @brief This is purely an intrusive allocator
|
||||
* This doesn't support any form of ordering at all
|
||||
* Just provides a chunk of memory for allocating IR nodes from
|
||||
*
|
||||
* Can potentially support reallocation if we are smart and make sure to invalidate anything holding a true pointer
|
||||
*/
|
||||
class DualIntrusiveAllocator {
|
||||
public:
|
||||
[[nodiscard]]
|
||||
bool DataCheckSize(size_t Size) const {
|
||||
size_t NewOffset = DataCurrentOffset + Size;
|
||||
return NewOffset <= MemorySize;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool ListCheckSize(size_t Size) const {
|
||||
size_t NewOffset = ListCurrentOffset + Size;
|
||||
return NewOffset <= MemorySize;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
void* DataAllocate(size_t Size) {
|
||||
LOGMAN_THROW_A_FMT(DataCheckSize(Size), "Ran out of space in DualIntrusiveAllocator during allocation");
|
||||
size_t NewOffset = DataCurrentOffset + Size;
|
||||
uintptr_t NewPointer = Data + DataCurrentOffset;
|
||||
DataCurrentOffset = NewOffset;
|
||||
return reinterpret_cast<void*>(NewPointer);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
void* ListAllocate(size_t Size) {
|
||||
LOGMAN_THROW_A_FMT(ListCheckSize(Size), "Ran out of space in DualIntrusiveAllocator during allocation");
|
||||
size_t NewOffset = ListCurrentOffset + Size;
|
||||
uintptr_t NewPointer = List + ListCurrentOffset;
|
||||
ListCurrentOffset = NewOffset;
|
||||
return reinterpret_cast<void*>(NewPointer);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
size_t DataSize() const {
|
||||
return DataCurrentOffset;
|
||||
}
|
||||
[[nodiscard]]
|
||||
size_t DataBackingSize() const {
|
||||
return MemorySize;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
size_t ListSize() const {
|
||||
return ListCurrentOffset;
|
||||
}
|
||||
[[nodiscard]]
|
||||
size_t ListBackingSize() const {
|
||||
return MemorySize;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
uintptr_t DataBegin() const {
|
||||
return Data;
|
||||
}
|
||||
[[nodiscard]]
|
||||
uintptr_t ListBegin() const {
|
||||
return List;
|
||||
}
|
||||
|
||||
void Reset() {
|
||||
DataCurrentOffset = 0;
|
||||
ListCurrentOffset = 0;
|
||||
}
|
||||
|
||||
void CopyData(const DualIntrusiveAllocator& rhs) {
|
||||
DataCurrentOffset = rhs.DataCurrentOffset;
|
||||
ListCurrentOffset = rhs.ListCurrentOffset;
|
||||
memcpy(reinterpret_cast<void*>(Data), reinterpret_cast<void*>(rhs.Data), DataCurrentOffset);
|
||||
memcpy(reinterpret_cast<void*>(List), reinterpret_cast<void*>(rhs.List), ListCurrentOffset);
|
||||
}
|
||||
|
||||
protected:
|
||||
DualIntrusiveAllocator(size_t Size)
|
||||
: MemorySize {Size} {}
|
||||
|
||||
uintptr_t Data;
|
||||
uintptr_t List;
|
||||
size_t DataCurrentOffset {0};
|
||||
size_t ListCurrentOffset {0};
|
||||
size_t MemorySize;
|
||||
};
|
||||
|
||||
class DualIntrusiveAllocatorMalloc final : public DualIntrusiveAllocator {
|
||||
public:
|
||||
DualIntrusiveAllocatorMalloc(size_t Size)
|
||||
: DualIntrusiveAllocator {Size} {
|
||||
Data = reinterpret_cast<uintptr_t>(FEXCore::Allocator::malloc(Size * 2));
|
||||
List = reinterpret_cast<uintptr_t>(Data + Size);
|
||||
}
|
||||
|
||||
~DualIntrusiveAllocatorMalloc() {
|
||||
FEXCore::Allocator::free(reinterpret_cast<void*>(Data));
|
||||
}
|
||||
};
|
||||
|
||||
class DualIntrusiveAllocatorThreadPool final : public DualIntrusiveAllocator {
|
||||
public:
|
||||
DualIntrusiveAllocatorThreadPool(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, size_t Size)
|
||||
: DualIntrusiveAllocator {Size}
|
||||
, PoolObject {ThreadAllocator, Size * 2} {
|
||||
// Claim a buffer on allocation
|
||||
PoolObject.ReownOrClaimBuffer();
|
||||
}
|
||||
|
||||
~DualIntrusiveAllocatorThreadPool() {
|
||||
PoolObject.UnclaimBuffer();
|
||||
}
|
||||
|
||||
void ReownOrClaimBuffer() {
|
||||
Data = PoolObject.ReownOrClaimBuffer();
|
||||
List = Data + MemorySize;
|
||||
}
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
PoolObject.DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
private:
|
||||
Utils::FixedSizePooledAllocation<uintptr_t, 5000, 500> PoolObject;
|
||||
};
|
||||
|
||||
class IRListView final : public FEXCore::Allocator::FEXAllocOperators {
|
||||
enum Flags {
|
||||
FLAG_IsCopy = 1,
|
||||
FLAG_Shared = 2,
|
||||
};
|
||||
|
||||
public:
|
||||
IRListView() = delete;
|
||||
IRListView(IRListView&&) = delete;
|
||||
|
||||
IRListView(DualIntrusiveAllocator* Data, bool _IsCopy) {
|
||||
SetCopy(_IsCopy);
|
||||
DataSize = Data->DataSize();
|
||||
ListSize = Data->ListSize();
|
||||
|
||||
if (_IsCopy) {
|
||||
IRDataInternal = FEXCore::Allocator::malloc(DataSize + ListSize);
|
||||
ListDataInternal = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(IRDataInternal) + DataSize);
|
||||
memcpy(IRDataInternal, reinterpret_cast<void*>(Data->DataBegin()), DataSize);
|
||||
memcpy(ListDataInternal, reinterpret_cast<void*>(Data->ListBegin()), ListSize);
|
||||
} else {
|
||||
// We are just pointing to the data
|
||||
IRDataInternal = reinterpret_cast<void*>(Data->DataBegin());
|
||||
ListDataInternal = reinterpret_cast<void*>(Data->ListBegin());
|
||||
}
|
||||
}
|
||||
|
||||
IRListView(IRListView* Old, bool _IsCopy) {
|
||||
SetCopy(_IsCopy);
|
||||
DataSize = Old->DataSize;
|
||||
ListSize = Old->ListSize;
|
||||
if (_IsCopy) {
|
||||
IRDataInternal = FEXCore::Allocator::malloc(DataSize + ListSize);
|
||||
ListDataInternal = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(IRDataInternal) + DataSize);
|
||||
memcpy(IRDataInternal, Old->IRDataInternal, DataSize);
|
||||
memcpy(ListDataInternal, Old->ListDataInternal, ListSize);
|
||||
} else {
|
||||
IRDataInternal = Old->IRDataInternal;
|
||||
ListDataInternal = Old->ListDataInternal;
|
||||
}
|
||||
}
|
||||
|
||||
~IRListView() {
|
||||
if (IsCopy()) {
|
||||
FEXCore::Allocator::free(IRDataInternal);
|
||||
// ListData is just offset from IRData
|
||||
}
|
||||
}
|
||||
|
||||
void Serialize(FEXCore::Context::AOTIRWriter& stream) const {
|
||||
void* nul = nullptr;
|
||||
// void *IRDataInternal;
|
||||
stream.Write((const char*)&nul, sizeof(nul));
|
||||
// void *ListDataInternal;
|
||||
stream.Write((const char*)&nul, sizeof(nul));
|
||||
// size_t DataSize;
|
||||
stream.Write((const char*)&DataSize, sizeof(DataSize));
|
||||
// size_t ListSize;
|
||||
stream.Write((const char*)&ListSize, sizeof(ListSize));
|
||||
// uint64_t Flags;
|
||||
uint64_t WrittenFlags = FLAG_Shared; // on disk format always has the Shared flag
|
||||
stream.Write((const char*)&WrittenFlags, sizeof(WrittenFlags));
|
||||
|
||||
// inline data
|
||||
stream.Write((const char*)GetData(), DataSize);
|
||||
stream.Write((const char*)GetListData(), ListSize);
|
||||
}
|
||||
|
||||
void Serialize(uint8_t* ptr) const {
|
||||
void* nul = nullptr;
|
||||
// void *IRDataInternal;
|
||||
memcpy(ptr, &nul, sizeof(nul));
|
||||
ptr += sizeof(nul);
|
||||
// void *ListDataInternal;
|
||||
memcpy(ptr, &nul, sizeof(nul));
|
||||
ptr += sizeof(nul);
|
||||
// size_t DataSize;
|
||||
memcpy(ptr, &DataSize, sizeof(DataSize));
|
||||
ptr += sizeof(DataSize);
|
||||
// size_t ListSize;
|
||||
memcpy(ptr, &ListSize, sizeof(ListSize));
|
||||
ptr += sizeof(ListSize);
|
||||
// uint64_t Flags;
|
||||
uint64_t WrittenFlags = FLAG_Shared; // on disk format always has the Shared flag
|
||||
memcpy(ptr, &WrittenFlags, sizeof(WrittenFlags));
|
||||
ptr += sizeof(WrittenFlags);
|
||||
|
||||
// inline data
|
||||
memcpy(ptr, (const void*)GetData(), DataSize);
|
||||
ptr += DataSize;
|
||||
memcpy(ptr, (const void*)GetListData(), ListSize);
|
||||
ptr += ListSize;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
size_t GetInlineSize() const {
|
||||
static_assert(sizeof(*this) == 40);
|
||||
return sizeof(*this) + DataSize + ListSize;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
IRListView* CreateCopy() {
|
||||
return new IRListView(this, true);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
size_t GetDataSize() const {
|
||||
return DataSize;
|
||||
}
|
||||
[[nodiscard]]
|
||||
size_t GetListSize() const {
|
||||
return ListSize;
|
||||
}
|
||||
[[nodiscard]]
|
||||
size_t GetSSACount() const {
|
||||
return ListSize / sizeof(OrderedNode);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsCopy() const {
|
||||
return (Flags & FLAG_IsCopy) != 0;
|
||||
}
|
||||
void SetCopy(bool Set) {
|
||||
if (Set) {
|
||||
Flags |= FLAG_IsCopy;
|
||||
} else {
|
||||
Flags &= ~FLAG_IsCopy;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsShared() const {
|
||||
return (Flags & FLAG_Shared) != 0;
|
||||
}
|
||||
void SetShared(bool Set) {
|
||||
if (Set) {
|
||||
Flags |= FLAG_Shared;
|
||||
} else {
|
||||
Flags &= ~FLAG_Shared;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
NodeID GetID(const OrderedNode* Node) const {
|
||||
return Node->Wrapped(GetListData()).ID();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
OrderedNode* GetHeaderNode() const {
|
||||
OrderedNodeWrapper Wrapped;
|
||||
Wrapped.NodeOffset = sizeof(OrderedNode);
|
||||
return Wrapped.GetNode(GetListData());
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
IROp_IRHeader* GetHeader() const {
|
||||
return GetOp<IROp_IRHeader>(GetHeaderNode());
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
[[nodiscard]]
|
||||
T* GetOp(OrderedNode* Node) const {
|
||||
auto OpHeader = Node->Op(GetData());
|
||||
auto Op = OpHeader->template CW<T>();
|
||||
|
||||
// If we are casting to something narrower than just the header, check the opcode.
|
||||
if constexpr (!std::is_same<T, IROp_Header>::value) {
|
||||
LOGMAN_THROW_A_FMT(Op->OPCODE == Op->Header.Op, "Expected Node to be '{}'. Found '{}' instead", GetName(Op->OPCODE),
|
||||
GetName(Op->Header.Op));
|
||||
}
|
||||
|
||||
return Op;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
[[nodiscard]]
|
||||
T* GetOp(OrderedNodeWrapper Wrapper) const {
|
||||
auto Node = Wrapper.GetNode(GetListData());
|
||||
return GetOp<T>(Node);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
OrderedNode* GetNode(OrderedNodeWrapper Wrapper) const {
|
||||
return Wrapper.GetNode(GetListData());
|
||||
}
|
||||
|
||||
///< Gets an OrderedNode from the IRListView as an OrderedNodeWrapper.
|
||||
[[nodiscard]]
|
||||
OrderedNodeWrapper WrapNode(OrderedNode* Node) const {
|
||||
return Node->Wrapped(GetListData());
|
||||
}
|
||||
|
||||
private:
|
||||
struct BlockRange {
|
||||
using iterator = NodeIterator;
|
||||
const IRListView* View;
|
||||
|
||||
BlockRange(const IRListView* parent)
|
||||
: View(parent) {};
|
||||
|
||||
[[nodiscard]]
|
||||
iterator begin() const noexcept {
|
||||
auto Header = View->GetHeader();
|
||||
return iterator(View->GetListData(), View->GetData(), Header->Blocks);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
iterator end() const noexcept {
|
||||
return iterator(View->GetListData(), View->GetData());
|
||||
}
|
||||
};
|
||||
|
||||
struct CodeRange {
|
||||
using iterator = NodeIterator;
|
||||
const IRListView* View;
|
||||
const OrderedNodeWrapper BlockWrapper;
|
||||
|
||||
CodeRange(const IRListView* parent, OrderedNodeWrapper block)
|
||||
: View(parent)
|
||||
, BlockWrapper(block) {};
|
||||
|
||||
[[nodiscard]]
|
||||
iterator begin() const noexcept {
|
||||
auto Block = View->GetOp<IROp_CodeBlock>(BlockWrapper);
|
||||
return iterator(View->GetListData(), View->GetData(), Block->Begin);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
iterator end() const noexcept {
|
||||
return iterator(View->GetListData(), View->GetData());
|
||||
}
|
||||
};
|
||||
|
||||
struct AllCodeRange {
|
||||
using iterator = AllNodesIterator; // Diffrent Iterator
|
||||
const IRListView* View;
|
||||
|
||||
AllCodeRange(const IRListView* parent)
|
||||
: View(parent) {};
|
||||
|
||||
[[nodiscard]]
|
||||
iterator begin() const noexcept {
|
||||
auto Header = View->GetHeader();
|
||||
return iterator(View->GetListData(), View->GetData(), Header->Blocks);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
iterator end() const noexcept {
|
||||
return iterator(View->GetListData(), View->GetData());
|
||||
}
|
||||
};
|
||||
|
||||
public:
|
||||
using iterator = NodeIterator;
|
||||
|
||||
[[nodiscard]]
|
||||
BlockRange GetBlocks() const {
|
||||
return BlockRange(this);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
CodeRange GetCode(const OrderedNode* block) const {
|
||||
return CodeRange(this, block->Wrapped(GetListData()));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
AllCodeRange GetAllCode() const {
|
||||
return AllCodeRange(this);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
iterator begin() const noexcept {
|
||||
OrderedNodeWrapper Wrapped;
|
||||
Wrapped.NodeOffset = sizeof(OrderedNode);
|
||||
return iterator(GetListData(), GetData(), Wrapped);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief This is not an iterator that you can reverse iterator through!
|
||||
*
|
||||
* @return Our iterator sentinel to ensure ending correctly
|
||||
*/
|
||||
[[nodiscard]]
|
||||
iterator end() const noexcept {
|
||||
OrderedNodeWrapper Wrapped;
|
||||
Wrapped.NodeOffset = 0;
|
||||
return iterator(GetListData(), GetData(), Wrapped);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Convert a OrderedNodeWrapper to an interator that we can iterate over
|
||||
* @return Iterator for this op
|
||||
*/
|
||||
[[nodiscard]]
|
||||
iterator at(OrderedNodeWrapper Wrapped) const noexcept {
|
||||
return iterator(GetListData(), GetData(), Wrapped);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
iterator at(NodeID ID) const noexcept {
|
||||
OrderedNodeWrapper Wrapped;
|
||||
Wrapped.NodeOffset = ID.Value * sizeof(OrderedNode);
|
||||
return iterator(GetListData(), GetData(), Wrapped);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
iterator at(const OrderedNode* Node) const noexcept {
|
||||
const auto ListData = GetListData();
|
||||
auto Wrapped = Node->Wrapped(ListData);
|
||||
return iterator(ListData, GetData(), Wrapped);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
uintptr_t GetData() const {
|
||||
return reinterpret_cast<uintptr_t>(IRDataInternal ? IRDataInternal : InlineData);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
uintptr_t GetListData() const {
|
||||
return reinterpret_cast<uintptr_t>(ListDataInternal ? ListDataInternal : &InlineData[DataSize]);
|
||||
}
|
||||
|
||||
private:
|
||||
void* IRDataInternal;
|
||||
void* ListDataInternal;
|
||||
size_t DataSize;
|
||||
size_t ListSize;
|
||||
uint64_t Flags {0};
|
||||
uint8_t InlineData[0];
|
||||
};
|
||||
|
||||
struct IRListViewDeleter {
|
||||
void operator()(IRListView* r) {
|
||||
if (!r->IsShared()) {
|
||||
delete r;
|
||||
}
|
||||
}
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
@@ -66,7 +66,7 @@ void PassManager::Finalize() {
|
||||
}
|
||||
}
|
||||
|
||||
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants) {
|
||||
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx, bool InlineConstants) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
@@ -80,7 +80,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool Inli
|
||||
|
||||
InsertPass(CreateDeadStoreElimination(ctx->HostFeatures.SupportsAVX));
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
InsertPass(CreateConstProp(InlineConstants, ctx->HostFeatures.SupportsTSOImm9));
|
||||
InsertPass(CreateConstProp(InlineConstants, ctx->HostFeatures.SupportsTSOImm9, Is64BitMode()));
|
||||
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
|
||||
@@ -105,21 +105,20 @@ void PassManager::InsertRegisterAllocationPass(bool SupportsAVX) {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(GetPass("Compaction"), SupportsAVX), "RA");
|
||||
}
|
||||
|
||||
bool PassManager::Run(IREmitter *IREmit) {
|
||||
bool PassManager::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::Run");
|
||||
|
||||
bool Changed = false;
|
||||
for (auto const &Pass : Passes) {
|
||||
for (const auto& Pass : Passes) {
|
||||
Changed |= Pass->Run(IREmit);
|
||||
}
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
for (auto const &Pass : ValidationPasses) {
|
||||
for (const auto& Pass : ValidationPasses) {
|
||||
Changed |= Pass->Run(IREmit);
|
||||
}
|
||||
#endif
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
Loaded 100 of 546 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user