mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 11:00:17 +02:00
Compare commits
413
Commits
FEX-2310
...
FEX-2312_1
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
faed139c34 | ||
|
|
80927bf0e1 | ||
|
|
64276dbd0c | ||
|
|
b619f381b3 | ||
|
|
3b0aff5fb9 | ||
|
|
a8ab8bbe8e | ||
|
|
3e2ba6d835 | ||
|
|
c6497fe32b | ||
|
|
c8ef77c15f | ||
|
|
b35fadf7e3 | ||
|
|
250ffb6d23 | ||
|
|
f6b1434d63 | ||
|
|
6bbae69c75 | ||
|
|
d0f54bcb23 | ||
|
|
470615b896 | ||
|
|
b02ab8ee19 | ||
|
|
5f6046be4c | ||
|
|
e923e83efb | ||
|
|
e836e4212d | ||
|
|
9417c93110 | ||
|
|
068599b1ec | ||
|
|
7216415bfc | ||
|
|
3020626506 | ||
|
|
0a79fa8d5d | ||
|
|
f8380b9adb | ||
|
|
d898028bc3 | ||
|
|
f090700184 | ||
|
|
01d29dffb9 | ||
|
|
14ba64a22d | ||
|
|
6716077cb6 | ||
|
|
8892580c41 | ||
|
|
1b41304fc1 | ||
|
|
7de66ac3a4 | ||
|
|
85a1c1ff25 | ||
|
|
8e892ece59 | ||
|
|
aa1344aadd | ||
|
|
3f02d7c665 | ||
|
|
f328fca880 | ||
|
|
47d79978ef | ||
|
|
2e24f34a3f | ||
|
|
6e8af295c5 | ||
|
|
bba156a3c1 | ||
|
|
8015ce2099 | ||
|
|
1153c1a538 | ||
|
|
a47b3cccb8 | ||
|
|
2070056d16 | ||
|
|
e227f1343f | ||
|
|
3c7335713d | ||
|
|
bdf4089264 | ||
|
|
b027113998 | ||
|
|
389c6b11dd | ||
|
|
fa5d9dc3b7 | ||
|
|
cb56728e57 | ||
|
|
b89c3a4573 | ||
|
|
0cc11108ba | ||
|
|
a7caf83022 | ||
|
|
053452c40c | ||
|
|
d4361c87ae | ||
|
|
6469eb7a0e | ||
|
|
13fbd0e802 | ||
|
|
e555a8f817 | ||
|
|
f9fb61cf1a | ||
|
|
43cf2e4e2c | ||
|
|
98f9a65202 | ||
|
|
8726c8fb73 | ||
|
|
27f3cb336f | ||
|
|
aa3bacd938 | ||
|
|
c71492ef32 | ||
|
|
70191f2d28 | ||
|
|
93db8b7ca7 | ||
|
|
d33b0cb9e3 | ||
|
|
0806d4ec25 | ||
|
|
9b646746b2 | ||
|
|
11993daec4 | ||
|
|
a78ffeeaba | ||
|
|
05b78339f6 | ||
|
|
365c221029 | ||
|
|
f60608a9c0 | ||
|
|
153d871be2 | ||
|
|
b69f2d7773 | ||
|
|
09ffe7ef6b | ||
|
|
3dfb94b524 | ||
|
|
d1e43d94e9 | ||
|
|
1b490e0e53 | ||
|
|
094146d630 | ||
|
|
23c2a53683 | ||
|
|
82b7689ca4 | ||
|
|
149f3e6f6d | ||
|
|
cea551c2ac | ||
|
|
2dcae23776 | ||
|
|
92e4e75217 | ||
|
|
5ca35bf77c | ||
|
|
c956b82d27 | ||
|
|
1c115096c4 | ||
|
|
c1d5fae018 | ||
|
|
17d49fc00f | ||
|
|
0e1e4c16b1 | ||
|
|
bec8e27b4f | ||
|
|
56841f0e50 | ||
|
|
723146050b | ||
|
|
85b1aa4c2d | ||
|
|
0506369519 | ||
|
|
ba1632974e | ||
|
|
c69082b1a4 | ||
|
|
74b2548982 | ||
|
|
f31656ec65 | ||
|
|
e91420c405 | ||
|
|
25df59a65d | ||
|
|
83fdd5720f | ||
|
|
db63241fd4 | ||
|
|
89b00c89aa | ||
|
|
d38917b5f0 | ||
|
|
910e0242c1 | ||
|
|
651b7bb75d | ||
|
|
c9f13ae1dd | ||
|
|
862e575100 | ||
|
|
4669c4541c | ||
|
|
769a8c41c4 | ||
|
|
bec9dba2b1 | ||
|
|
205ba2ea13 | ||
|
|
2073f6d287 | ||
|
|
0f25a960ee | ||
|
|
282ed3e309 | ||
|
|
cd031a7d38 | ||
|
|
e1885ed0bd | ||
|
|
4a31b619fa | ||
|
|
bd4464bd5e | ||
|
|
d907a7dc9f | ||
|
|
732070f750 | ||
|
|
48442b6b03 | ||
|
|
7fdbe547a3 | ||
|
|
59565b828d | ||
|
|
1e2d059890 | ||
|
|
0aa41908a2 | ||
|
|
b27ce3f79c | ||
|
|
238e52f74a | ||
|
|
9398b931fb | ||
|
|
b2a9785959 | ||
|
|
224a1f19a3 | ||
|
|
109c53f22b | ||
|
|
e25849b2cb | ||
|
|
57978accc1 | ||
|
|
ff37177f4d | ||
|
|
472d143021 | ||
|
|
157f95b08f | ||
|
|
6bf7ab0778 | ||
|
|
0de958be2a | ||
|
|
ef544fecf2 | ||
|
|
5eea68d6c6 | ||
|
|
5471367db1 | ||
|
|
b8265b1067 | ||
|
|
099c683a5a | ||
|
|
0357bb23e8 | ||
|
|
c7193b52fb | ||
|
|
ec14a65e23 | ||
|
|
0d70c6a0d0 | ||
|
|
bfa069c4d5 | ||
|
|
61bdf64e15 | ||
|
|
482b35c283 | ||
|
|
b74d886017 | ||
|
|
1667abad7e | ||
|
|
b187a853e7 | ||
|
|
228c7d142e | ||
|
|
041199644c | ||
|
|
3767f3633d | ||
|
|
af3253947e | ||
|
|
efc5eb2933 | ||
|
|
b4eeb96375 | ||
|
|
c9832e3d34 | ||
|
|
da3e3fc7a3 | ||
|
|
b5c83f0628 | ||
|
|
03087a55ba | ||
|
|
1ce3c16b30 | ||
|
|
bf702850a9 | ||
|
|
584c4cc05e | ||
|
|
279afd88bb | ||
|
|
c0a6d82025 | ||
|
|
3a03e1c93c | ||
|
|
bdaa70405f | ||
|
|
1281145982 | ||
|
|
87cac09477 | ||
|
|
5336129b58 | ||
|
|
72fc2b522d | ||
|
|
b6f6c84790 | ||
|
|
11e9be13b1 | ||
|
|
afdb8753ba | ||
|
|
f6a2e6739d | ||
|
|
d6569d510d | ||
|
|
04e4993d9b | ||
|
|
783e09d67d | ||
|
|
314f478225 | ||
|
|
c1dbc28aa2 | ||
|
|
8f7e393ffb | ||
|
|
b3055523b4 | ||
|
|
cf6b21564c | ||
|
|
996a4c023c | ||
|
|
0dcbdcc0e2 | ||
|
|
5bdd422db6 | ||
|
|
bf147f47b5 | ||
|
|
3f1f7faf34 | ||
|
|
1fc6725826 | ||
|
|
fa8c35feba | ||
|
|
73958b9163 | ||
|
|
9b81a83894 | ||
|
|
65b7d4007e | ||
|
|
2c0444e846 | ||
|
|
b92e716d0c | ||
|
|
a7a1365cf7 | ||
|
|
8ee5b5cf50 | ||
|
|
b45023bedf | ||
|
|
3702e513f5 | ||
|
|
829384e488 | ||
|
|
ed23fbe932 | ||
|
|
0af0427efd | ||
|
|
a499272d81 | ||
|
|
5dee921300 | ||
|
|
c0dcf8925a | ||
|
|
e91c5ff906 | ||
|
|
190f7c27e0 | ||
|
|
b15f0b5d36 | ||
|
|
e2c65189ff | ||
|
|
03f63f99a8 | ||
|
|
e305a9a0d5 | ||
|
|
3a90dbbb35 | ||
|
|
5103f2d92b | ||
|
|
d4a6b031ea | ||
|
|
319cf4bf3d | ||
|
|
d75c0f2c50 | ||
|
|
a586d3823d | ||
|
|
5522c6db9c | ||
|
|
367e1658ad | ||
|
|
bbad06f81a | ||
|
|
9612b2fe4b | ||
|
|
ef5503f0b7 | ||
|
|
26bf67ca76 | ||
|
|
e5df636efd | ||
|
|
f34f4a0227 | ||
|
|
f45722d2cd | ||
|
|
db7ef0e4b0 | ||
|
|
de10cbad98 | ||
|
|
e4d9c264d8 | ||
|
|
97b3efa90a | ||
|
|
18065199e3 | ||
|
|
77d92872bc | ||
|
|
460f13be71 | ||
|
|
47c9463217 | ||
|
|
15c825f362 | ||
|
|
bbd20b47ba | ||
|
|
5b70209728 | ||
|
|
a287f2a189 | ||
|
|
3a240e3b61 | ||
|
|
de1e593ec2 | ||
|
|
13cd8b33a2 | ||
|
|
0f26bc20a3 | ||
|
|
eacab3cc22 | ||
|
|
09e3371a0d | ||
|
|
ff3f7345b6 | ||
|
|
8181e53727 | ||
|
|
5431aa5a28 | ||
|
|
1a293cc542 | ||
|
|
b1e78934ad | ||
|
|
61f22911c7 | ||
|
|
fe8778bb96 | ||
|
|
14e5ea1e22 | ||
|
|
74f1205f33 | ||
|
|
4045bfd187 | ||
|
|
9db93a43dd | ||
|
|
a379d50729 | ||
|
|
f5822f83b0 | ||
|
|
5028434292 | ||
|
|
8f0461cac8 | ||
|
|
0b4cb23411 | ||
|
|
bab96b9441 | ||
|
|
11e2f14185 | ||
|
|
6177290e9d | ||
|
|
20a54913bd | ||
|
|
dd9ed89a7a | ||
|
|
a4e1e0a1fb | ||
|
|
5e9f69001d | ||
|
|
39c5ab1c81 | ||
|
|
5bf790324c | ||
|
|
adcdb32d49 | ||
|
|
f264578f12 | ||
|
|
7149da387a | ||
|
|
0f3d14e7c0 | ||
|
|
aad5080224 | ||
|
|
c77a3d673c | ||
|
|
0ff2e6e1e3 | ||
|
|
8538f5bac4 | ||
|
|
a305baf6e5 | ||
|
|
807619aa02 | ||
|
|
6db2125b41 | ||
|
|
9f6d80fe5d | ||
|
|
423ce12001 | ||
|
|
4edd72fc33 | ||
|
|
978f607dd9 | ||
|
|
9ba78c9771 | ||
|
|
e2144345c0 | ||
|
|
63e4c3682d | ||
|
|
2956e84ead | ||
|
|
4466c50c2b | ||
|
|
dd5ca1d349 | ||
|
|
95c756b466 | ||
|
|
99465faf63 | ||
|
|
e018917f76 | ||
|
|
a261d9909e | ||
|
|
6de8bc6848 | ||
|
|
d87155e4ee | ||
|
|
7484cacaf9 | ||
|
|
42259974c4 | ||
|
|
e455996dbd | ||
|
|
b5dd1d05e9 | ||
|
|
bbaf70da15 | ||
|
|
65e8d094ef | ||
|
|
4c801d594a | ||
|
|
d4403edea9 | ||
|
|
06ef012fb2 | ||
|
|
8f8f37684a | ||
|
|
7140b8d901 | ||
|
|
d5beba9423 | ||
|
|
826e15aea9 | ||
|
|
14e80ce228 | ||
|
|
165d3d3d4d | ||
|
|
887200e571 | ||
|
|
2c0bc0654d | ||
|
|
b3d76bd2f1 | ||
|
|
2e694412f4 | ||
|
|
d84577c36c | ||
|
|
cf9c2aa72c | ||
|
|
1cb8e4891c | ||
|
|
24f2796141 | ||
|
|
cb215b5f21 | ||
|
|
0cf2695772 | ||
|
|
6a6886305e | ||
|
|
5ef7537e61 | ||
|
|
167fe85cc3 | ||
|
|
cf65747667 | ||
|
|
27bb28b47f | ||
|
|
a00da800e7 | ||
|
|
bf835e80ac | ||
|
|
8f246b206b | ||
|
|
3c5c23bf36 | ||
|
|
5bcfaf4b9f | ||
|
|
1f6c6345d9 | ||
|
|
93792577eb | ||
|
|
3d23cd5765 | ||
|
|
fcc239552c | ||
|
|
8238de024f | ||
|
|
39e658f02a | ||
|
|
e89dd27f2a | ||
|
|
f85fae0041 | ||
|
|
65eec673fc | ||
|
|
5c93a085d2 | ||
|
|
4b356a7c2c | ||
|
|
2b67f87054 | ||
|
|
1ea40ae676 | ||
|
|
e0ef32e0bf | ||
|
|
a2b53c8eb0 | ||
|
|
21b6cccb4e | ||
|
|
d539829251 | ||
|
|
ef321e4bf8 | ||
|
|
47a0f14537 | ||
|
|
6d39f369b0 | ||
|
|
2304cfc530 | ||
|
|
1a39de4509 | ||
|
|
efb479f88f | ||
|
|
b27bf43901 | ||
|
|
c612fa8f2f | ||
|
|
4ccc40f697 | ||
|
|
cb53a704ba | ||
|
|
483423674a | ||
|
|
180d16af7a | ||
|
|
1acc038826 | ||
|
|
cc558fd5dc | ||
|
|
8f04223193 | ||
|
|
4be649c44e | ||
|
|
6253f4f708 | ||
|
|
cc2eef619c | ||
|
|
3bff42e6a7 | ||
|
|
2671246fef | ||
|
|
8cb8f090dd | ||
|
|
a37d89a7d5 | ||
|
|
252d7712ea | ||
|
|
c548625fbe | ||
|
|
f036a0b84f | ||
|
|
2e1389b25e | ||
|
|
a5f82a57fa | ||
|
|
cd83d3eb24 | ||
|
|
93ab8ab23c | ||
|
|
462fff2c67 | ||
|
|
8dab35cbf8 | ||
|
|
a1a479e69f | ||
|
|
6403290019 | ||
|
|
580bd50a00 | ||
|
|
b2a8b0ca12 | ||
|
|
f78bdf0852 | ||
|
|
5652eb4c5d | ||
|
|
a52bb47551 | ||
|
|
22590dde77 | ||
|
|
6543a80ff9 | ||
|
|
9c36d1061b | ||
|
|
5a3cc7b469 | ||
|
|
4cff3e5f1f | ||
|
|
a1eb571630 | ||
|
|
559cf6491a | ||
|
|
4bdda1eeb5 | ||
|
|
fc70fc3506 | ||
|
|
26ee63cc24 | ||
|
|
0092ea7c0b | ||
|
|
439a3b9c3a | ||
|
|
b4ddf36582 | ||
|
|
12c44f26e5 | ||
|
|
5b7ba06d5c |
No files matched your search
@@ -56,6 +56,16 @@ jobs:
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Set vixl_sim x86
|
||||
if: matrix.arch[1] == 'x64'
|
||||
run: |
|
||||
echo "VIXL_SIM_ENABLED=True" >> $GITHUB_ENV
|
||||
|
||||
- name: Set vixl_sim Arm64
|
||||
if: matrix.arch[1] == 'ARM64'
|
||||
run: |
|
||||
echo "VIXL_SIM_ENABLED=False" >> $GITHUB_ENV
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
@@ -64,7 +74,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=False -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
+6
-5
@@ -293,10 +293,11 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
set(FEX_TUNE_COMPILE_FLAGS)
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
add_compile_options("-march=${TUNE_ARCH}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
endif()
|
||||
@@ -309,7 +310,7 @@ if (TUNE_CPU STREQUAL "native")
|
||||
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
|
||||
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=native")
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
@@ -323,19 +324,19 @@ if (TUNE_CPU STREQUAL "native")
|
||||
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${AARCH64_CPU}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${AARCH64_CPU}")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
add_compile_options("-march=native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=native")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${TUNE_CPU}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${TUNE_CPU}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
|
||||
endif()
|
||||
|
||||
Vendored
-13
@@ -1,13 +0,0 @@
|
||||
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
|
||||
Version 2, December 2004
|
||||
|
||||
Copyright (C) 2018 Ryan Houdek <Sonicadvance1@gmail.com>
|
||||
|
||||
Everyone is permitted to copy and distribute verbatim or modified
|
||||
copies of this license document, and changing it is allowed as long
|
||||
as the name is changed.
|
||||
|
||||
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
|
||||
TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION
|
||||
|
||||
0. You just DO WHAT THE FUCK YOU WANT TO.
|
||||
Vendored
+1
-1
Submodule External/fmt updated: e57ca2e368...f5e54359df.
@@ -38,7 +38,12 @@ check_cxx_source_compiles(
|
||||
HAS_CLANG_PRESERVE_ALL)
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
if (HAS_CLANG_PRESERVE_ALL)
|
||||
message(STATUS "Has clang::preserve_all")
|
||||
if (MINGW_BUILD)
|
||||
message(STATUS "Ignoring broken clang::preserve_all support")
|
||||
set(HAS_CLANG_PRESERVE_ALL FALSE)
|
||||
else()
|
||||
message(STATUS "Has clang::preserve_all")
|
||||
endif()
|
||||
endif ()
|
||||
|
||||
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
|
||||
|
||||
@@ -46,11 +46,15 @@ class OpDefinition:
|
||||
NumElements: str
|
||||
OpClass: str
|
||||
HasSideEffects: bool
|
||||
ImplicitFlagClobber: bool
|
||||
RAOverride: int
|
||||
SwitchGen: bool
|
||||
ArgPrinter: bool
|
||||
SSAArgNum: int
|
||||
NonSSAArgNum: int
|
||||
DynamicDispatch: bool
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
@@ -64,11 +68,15 @@ class OpDefinition:
|
||||
self.OpClass = None
|
||||
self.OpSize = 0
|
||||
self.HasSideEffects = False
|
||||
self.ImplicitFlagClobber = False
|
||||
self.RAOverride = -1
|
||||
self.SwitchGen = True
|
||||
self.ArgPrinter = True
|
||||
self.SSAArgNum = 0
|
||||
self.NonSSAArgNum = 0
|
||||
self.DynamicDispatch = False
|
||||
self.JITDispatch = True
|
||||
self.JITDispatchOverride = None
|
||||
self.Arguments = []
|
||||
self.EmitValidation = []
|
||||
self.Desc = []
|
||||
@@ -213,6 +221,9 @@ def parse_ops(ops):
|
||||
if "HasSideEffects" in op_val:
|
||||
OpDef.HasSideEffects = bool(op_val["HasSideEffects"])
|
||||
|
||||
if "ImplicitFlagClobber" in op_val:
|
||||
OpDef.ImplicitFlagClobber = bool(op_val["ImplicitFlagClobber"])
|
||||
|
||||
if "ArgPrinter" in op_val:
|
||||
OpDef.ArgPrinter = bool(op_val["ArgPrinter"])
|
||||
|
||||
@@ -228,6 +239,15 @@ def parse_ops(ops):
|
||||
if "Desc" in op_val:
|
||||
OpDef.Desc = op_val["Desc"]
|
||||
|
||||
if "DynamicDispatch" in op_val:
|
||||
OpDef.DynamicDispatch = bool(op_val["DynamicDispatch"])
|
||||
|
||||
if "JITDispatch" in op_val:
|
||||
OpDef.JITDispatch = bool(op_val["JITDispatch"])
|
||||
|
||||
if "JITDispatchOverride" in op_val:
|
||||
OpDef.JITDispatchOverride = op_val["JITDispatchOverride"]
|
||||
|
||||
# Do some fixups of the data here
|
||||
if len(OpDef.EmitValidation) != 0:
|
||||
for i in range(len(OpDef.EmitValidation)):
|
||||
@@ -357,6 +377,7 @@ def print_ir_sizes():
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool ImplicitFlagClobber(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
|
||||
|
||||
output_file.write("#undef IROP_SIZES\n")
|
||||
@@ -450,15 +471,17 @@ def print_ir_getraargs():
|
||||
def print_ir_hassideeffects():
|
||||
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> SideEffects = {\n")
|
||||
for op in IROps:
|
||||
output_file.write("\t{},\n".format(("true" if op.HasSideEffects else "false")))
|
||||
for array, prop in [("SideEffects", "HasSideEffects"),
|
||||
("ImplicitFlagClobbers", "ImplicitFlagClobber")]:
|
||||
output_file.write(f"constexpr std::array<uint8_t, OP_LAST + 1> {array} = {{\n")
|
||||
for op in IROps:
|
||||
output_file.write("\t{},\n".format(("true" if getattr(op, prop) else "false")))
|
||||
|
||||
output_file.write("};\n\n")
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("bool HasSideEffects(IROps Op) {\n")
|
||||
output_file.write(" return SideEffects[Op];\n")
|
||||
output_file.write("}\n")
|
||||
output_file.write(f"bool {prop}(IROps Op) {{\n")
|
||||
output_file.write(f" return {array}[Op];\n")
|
||||
output_file.write("}\n")
|
||||
|
||||
output_file.write("#undef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -627,6 +650,10 @@ def print_ir_allocator_helpers():
|
||||
|
||||
output_file.write(") {\n")
|
||||
|
||||
# Save NZCV if needed before clobbering NZCV
|
||||
if op.ImplicitFlagClobber:
|
||||
output_file.write("\t\tSaveNZCV(IROps::OP_{});".format(op.Name.upper()))
|
||||
|
||||
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
@@ -675,7 +702,8 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"\");\n".format(Validation))
|
||||
Sanitized = Validation.replace("\"", "\\\"")
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
|
||||
output_file.write("\t\treturn Op;\n")
|
||||
@@ -730,10 +758,38 @@ def print_ir_parser_switch_helper():
|
||||
output_file.write("#undef IROP_PARSER_SWITCH_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
|
||||
if (len(sys.argv) < 3):
|
||||
def print_ir_dispatcher_defs():
|
||||
output_dispatch_file.write("#ifdef IROP_DISPATCH_DEFS\n")
|
||||
for op in IROps:
|
||||
if op.Name != "Last" and op.SwitchGen and op.JITDispatch and op.JITDispatchOverride == None:
|
||||
output_dispatch_file.write("DEF_OP({});\n".format(op.Name))
|
||||
|
||||
output_dispatch_file.write("#undef IROP_DISPATCH_DEFS\n")
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
def print_ir_dispatcher_dispatch():
|
||||
output_dispatch_file.write("#ifdef IROP_DISPATCH_DISPATCH\n")
|
||||
for op in IROps:
|
||||
if op.Name != "Last" and op.JITDispatch:
|
||||
DispatchName = op.Name
|
||||
if op.JITDispatchOverride != None:
|
||||
DispatchName = op.JITDispatchOverride
|
||||
|
||||
if (op.DynamicDispatch):
|
||||
output_dispatch_file.write("REGISTER_OP_RT({}, {});\n".format(op.Name.upper(), DispatchName))
|
||||
else:
|
||||
output_dispatch_file.write("REGISTER_OP({}, {});\n".format(op.Name.upper(), DispatchName))
|
||||
|
||||
output_dispatch_file.write("#undef IROP_DISPATCH_DISPATCH\n")
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
|
||||
if (len(sys.argv) < 4):
|
||||
ExitError()
|
||||
|
||||
output_filename = sys.argv[2]
|
||||
output_dispatcher_filename = sys.argv[3]
|
||||
|
||||
json_file = open(sys.argv[1], "r")
|
||||
json_text = json_file.read()
|
||||
json_file.close()
|
||||
@@ -763,3 +819,10 @@ print_ir_allocator_helpers()
|
||||
print_ir_parser_switch_helper()
|
||||
|
||||
output_file.close()
|
||||
|
||||
output_dispatch_file = open(output_dispatcher_filename, "w")
|
||||
print_ir_dispatcher_defs()
|
||||
print_ir_dispatcher_dispatch()
|
||||
|
||||
output_dispatch_file.close()
|
||||
|
||||
@@ -90,7 +90,6 @@ set (SRCS
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/GdbServer.cpp
|
||||
Interface/Core/HostFeatures.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
@@ -101,9 +100,7 @@ set (SRCS
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/SignalDelegator.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
Interface/Core/X86DebugInfo.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
@@ -221,15 +218,16 @@ configure_file(
|
||||
# Generate IR include file
|
||||
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
|
||||
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
|
||||
set(OUTPUT_DISPATCHER_NAME "${OUTPUT_IR_FOLDER}/IRDefines_Dispatch.inc")
|
||||
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
|
||||
|
||||
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_NAME}"
|
||||
OUTPUT "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
|
||||
)
|
||||
|
||||
set_source_files_properties(${OUTPUT_NAME} PROPERTIES
|
||||
@@ -361,6 +359,7 @@ function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
@@ -369,6 +368,7 @@ endfunction()
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
if (MINGW_BUILD)
|
||||
# Mingw build isn't building a linux shared library, so it can't have a SONAME.
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -8,6 +7,7 @@
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
@@ -321,16 +321,6 @@ namespace DefaultValues {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
{
|
||||
// Always fix up the number of threads and create the configuration
|
||||
// Otherwise the application could receive zero as the number of threads
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
if (Cores == 0) {
|
||||
// When the number of emulated CPU cores is zero then auto detect
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THREADS, fextl::fmt::format("{}", FEXCore::CPUInfo::CalculateNumberOfCPUs()));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
|
||||
@@ -31,15 +31,6 @@
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"Threads": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"ShortArg": "T",
|
||||
"Desc": [
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
"0 will auto detect."
|
||||
]
|
||||
},
|
||||
"CacheObjectCodeCompilation": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
|
||||
@@ -82,7 +73,11 @@
|
||||
"ENABLEFLAGM": "enableflagm",
|
||||
"DISABLEFLAGM": "disableflagm",
|
||||
"ENABLEFLAGM2": "enableflagm2",
|
||||
"DISABLEFLAGM2": "disableflagm2"
|
||||
"DISABLEFLAGM2": "disableflagm2",
|
||||
"ENABLECRYPTO": "enablecrypto",
|
||||
"DISABLECRYPTO": "disablecrypto",
|
||||
"ENABLERPRES": "enablerpres",
|
||||
"DISABLERPRES": "disablerpres"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -100,7 +95,9 @@
|
||||
"\t{enable,disable}atomics: Will force enable or disable ARMv8.1 LSE atomics even if the host doesn't support it",
|
||||
"\t{enable,disable}fcma: Will force enable or disable fcma even if the host doesn't support it",
|
||||
"\t{enable,disable}flagm: Will force enable or disable flagm even if the host doesn't support it",
|
||||
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it"
|
||||
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it",
|
||||
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
|
||||
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it"
|
||||
]
|
||||
}
|
||||
},
|
||||
|
||||
@@ -26,12 +26,6 @@ namespace FEXCore::Context {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>();
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::InitializeContext() {
|
||||
// This should be used for generating things that are shared between threads
|
||||
CPUID.Init(this);
|
||||
return true;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
CustomExitHandler = std::move(handler);
|
||||
}
|
||||
@@ -48,22 +42,14 @@ namespace FEXCore::Context {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason FEXCore::Context::ContextImpl::GetExitReason() {
|
||||
return ParentThread->ExitReason;
|
||||
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::IsDone() const {
|
||||
return IsPaused();
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::GetCPUState(FEXCore::Core::CPUState *State) const {
|
||||
memcpy(State, ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCPUState(const FEXCore::Core::CPUState *State) {
|
||||
memcpy(ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
@@ -14,8 +14,8 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -37,7 +37,6 @@
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
class ThunkHandler;
|
||||
class GdbServer;
|
||||
|
||||
namespace CodeSerialize {
|
||||
class CodeObjectSerializeService;
|
||||
@@ -73,8 +72,6 @@ namespace FEXCore::Context {
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitializeContext() override;
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer) override;
|
||||
|
||||
void SetExitHandler(ExitHandler handler) override;
|
||||
@@ -90,16 +87,10 @@ namespace FEXCore::Context {
|
||||
void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
|
||||
|
||||
int GetProgramStatus() const override;
|
||||
|
||||
ExitReason GetExitReason() override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
bool IsDone() const override;
|
||||
|
||||
void GetCPUState(FEXCore::Core::CPUState *State) const override;
|
||||
void SetCPUState(const FEXCore::Core::CPUState *State) override;
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
@@ -107,31 +98,37 @@ namespace FEXCore::Context {
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) override;
|
||||
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The initial thread state to setup for our state, if inheriting.
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
*
|
||||
* @return The InternalThreadState object that tracks all of the emulated thread's state
|
||||
*
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - CTX->RunUntilExit(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - InitializeThread(Thread);
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(CopyOfThreadState, PPID);
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(NewState, PPID);
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
@@ -215,6 +212,9 @@ namespace FEXCore::Context {
|
||||
// this is for internal use
|
||||
bool ValidateIRarser { false };
|
||||
|
||||
// Used if the JIT needs to have its interrupt fault code emitted.
|
||||
bool NeedsPendingInterruptFaultCheck { false };
|
||||
|
||||
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
|
||||
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
@@ -281,22 +281,18 @@ namespace FEXCore::Context {
|
||||
~ContextImpl();
|
||||
|
||||
bool IsPaused() const { return !Running; }
|
||||
void WaitForThreadsToRun();
|
||||
void WaitForThreadsToRun() override;
|
||||
void Stop(bool IgnoreCurrentThread);
|
||||
void WaitForIdle();
|
||||
void WaitForIdle() override;
|
||||
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
|
||||
|
||||
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
|
||||
void StartGdbServer();
|
||||
void StopGdbServer();
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
@@ -307,7 +303,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
@@ -322,7 +318,7 @@ namespace FEXCore::Context {
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
@@ -333,8 +329,8 @@ namespace FEXCore::Context {
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
@@ -349,8 +345,6 @@ namespace FEXCore::Context {
|
||||
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
@@ -398,6 +392,13 @@ namespace FEXCore::Context {
|
||||
|
||||
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
|
||||
|
||||
ThreadsState GetThreads() override {
|
||||
return ThreadsState {
|
||||
.ParentThread = ParentThread,
|
||||
.Threads = &Threads,
|
||||
};
|
||||
}
|
||||
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
|
||||
protected:
|
||||
@@ -415,15 +416,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Does some final thread initialization
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*
|
||||
* InitCore and CreateThread both call this to finish up thread object initialization
|
||||
*/
|
||||
void InitializeThreadData(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
@@ -441,7 +433,6 @@ namespace FEXCore::Context {
|
||||
|
||||
// Entry Cache
|
||||
std::mutex ExitMutex;
|
||||
fextl::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
@@ -28,7 +29,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
namespace x64 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
@@ -36,23 +37,23 @@ namespace x64 {
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29,
|
||||
// PF/AF must be last.
|
||||
REG_PF, REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
@@ -175,19 +176,20 @@ namespace x64 {
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 10> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
// PF/AF must be last.
|
||||
REG_PF, REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 15> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
@@ -199,11 +201,10 @@ namespace x32 {
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 7> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
@@ -335,8 +336,8 @@ namespace x32 {
|
||||
}
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
: Emitter(size ? (uint8_t*)FEXCore::Allocator::VirtualAlloc(size, true) : nullptr, size)
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr, size_t size)
|
||||
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
|
||||
, EmitterCTX {ctx}
|
||||
#ifdef VIXL_SIMULATOR
|
||||
, Simulator {&SimDecoder}
|
||||
@@ -368,7 +369,7 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
}
|
||||
else {
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
@@ -379,13 +380,6 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
}
|
||||
}
|
||||
|
||||
Arm64Emitter::~Arm64Emitter() {
|
||||
auto BufferSize = GetBufferSize();
|
||||
if (BufferSize) {
|
||||
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
@@ -406,6 +400,15 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
Segments = 2;
|
||||
}
|
||||
|
||||
if (!Is64Bit && ((~Constant) & 0xFFFF0000) == 0) {
|
||||
movn(s, Reg.W(), (~Constant) & 0xFFFF);
|
||||
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
// Count the number of move segments
|
||||
@@ -581,10 +584,42 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Disable AFP features when spilling registers.
|
||||
//
|
||||
// Disable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
// Also interacts with RPRES to change reciprocal/rsqrt precision from 8-bit mantissa to 12-bit.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
|
||||
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
|
||||
// is always static and almost certainly clobbered by the subsequent code.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
|
||||
GPRSpillMask &= ~PFAFSpillMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
@@ -600,6 +635,14 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (PFAFSpillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
|
||||
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
@@ -645,6 +688,46 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
[[maybe_unused]] bool FoundRegister{};
|
||||
for (auto Reg : StaticRegisters) {
|
||||
if (((1U << Reg.Idx()) & GPRFillMask)) {
|
||||
TmpReg = Reg;
|
||||
FoundRegister = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(FoundRegister, "Didn't have an SRA register to use as a temporary while spilling!");
|
||||
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Enable AFP features when filling JIT state.
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
|
||||
// Enable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
|
||||
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
|
||||
// is always static and was almost certainly clobbered.
|
||||
//
|
||||
// TODO: Can we prove that NZCV is not used across a call in some cases and
|
||||
// omit this? Might help x87 perf? Future idea.
|
||||
ldr(TmpReg.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
|
||||
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
@@ -655,11 +738,11 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
// since all that matters is we restore them on a fill.
|
||||
// It's not a concern if they get trounced by something else.
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE) {
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
@@ -672,8 +755,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
@@ -704,6 +785,11 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
|
||||
GPRFillMask &= ~PFAFMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
@@ -718,6 +804,14 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1]));
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (PFAFFillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
|
||||
|
||||
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
|
||||
|
||||
@@ -53,12 +53,15 @@ constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
|
||||
|
||||
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
|
||||
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
|
||||
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
|
||||
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size);
|
||||
~Arm64Emitter();
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
@@ -129,21 +132,21 @@ protected:
|
||||
|
||||
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
|
||||
if (SupportsPreserveAllABI) {
|
||||
SpillForPreserveAllABICall(TMP1, true);
|
||||
SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
}
|
||||
else {
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TmpReg, FPRs);
|
||||
PushDynamicRegsAndLR(TmpReg);
|
||||
}
|
||||
}
|
||||
|
||||
void FillForABICall(bool SupportsPreserveAllABI, bool FPRs = true) {
|
||||
if (SupportsPreserveAllABI) {
|
||||
FillForPreserveAllABICall(true);
|
||||
FillForPreserveAllABICall(FPRs);
|
||||
}
|
||||
else {
|
||||
PopDynamicRegsAndLR();
|
||||
FillStaticRegs();
|
||||
FillStaticRegs(FPRs);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -766,6 +766,21 @@ public:
|
||||
EvaluateIntoFlags(Op, 1, rn);
|
||||
}
|
||||
|
||||
void cfinv() {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0001'1111;
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
void axflag() {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0101'1111;
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
void xaflag() {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0011'1111;
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
// Conditional compare - register
|
||||
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0011'1010'010 << 21;
|
||||
|
||||
@@ -60,7 +60,7 @@ public:
|
||||
}
|
||||
void sha256su1(FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn, FEXCore::ARMEmitter::VRegister rm) {
|
||||
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
|
||||
Crypto3RegSHA(Op, 0b100, rd, rn, rm);
|
||||
Crypto3RegSHA(Op, 0b110, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Cryptographic two-register SHA
|
||||
|
||||
@@ -1384,12 +1384,10 @@ public:
|
||||
}
|
||||
|
||||
// SVE predicate initialize
|
||||
template <SubRegSize size>
|
||||
void ptrue(PRegister pd, PredicatePattern pattern) {
|
||||
void ptrue(SubRegSize size, PRegister pd, PredicatePattern pattern) {
|
||||
SVEPredicateMisc(0b1000, 0b10000, FEXCore::ToUnderlying(pattern), size, pd);
|
||||
}
|
||||
template <SubRegSize size>
|
||||
void ptrues(PRegister pd, PredicatePattern pattern) {
|
||||
void ptrues(SubRegSize size, PRegister pd, PredicatePattern pattern) {
|
||||
SVEPredicateMisc(0b1001, 0b10000, FEXCore::ToUnderlying(pattern), size, pd);
|
||||
}
|
||||
|
||||
|
||||
@@ -798,6 +798,52 @@ public:
|
||||
// XXX:
|
||||
//
|
||||
// Floating-point data-processing (1 source)
|
||||
void fmov(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000000, rd, rn);
|
||||
}
|
||||
void fabs(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000001, rd, rn);
|
||||
}
|
||||
void fneg(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000010, rd, rn);
|
||||
}
|
||||
void fsqrt(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000011, rd, rn);
|
||||
}
|
||||
void frintn(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001000, rd, rn);
|
||||
}
|
||||
void frintp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001001, rd, rn);
|
||||
}
|
||||
void frintm(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001010, rd, rn);
|
||||
}
|
||||
void frintz(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001011, rd, rn);
|
||||
}
|
||||
void frinta(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001100, rd, rn);
|
||||
}
|
||||
void frintx(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001110, rd, rn);
|
||||
}
|
||||
void frinti(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001111, rd, rn);
|
||||
}
|
||||
void frint32z(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010000, rd, rn);
|
||||
}
|
||||
void frint32x(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010001, rd, rn);
|
||||
}
|
||||
void frint64z(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010010, rd, rn);
|
||||
}
|
||||
void frint64x(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010011, rd, rn);
|
||||
}
|
||||
|
||||
void fmov(SRegister rd, SRegister rn) {
|
||||
Float1Source(0, 0, 0b00, 0b000000, rd.V(), rn.V());
|
||||
}
|
||||
@@ -1065,6 +1111,34 @@ public:
|
||||
}
|
||||
|
||||
// Floating-point data-processing (2 source)
|
||||
void fmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0000, rd, rn, rm);
|
||||
}
|
||||
void fdiv(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0001, rd, rn, rm);
|
||||
}
|
||||
void fadd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0010, rd, rn, rm);
|
||||
}
|
||||
void fsub(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0011, rd, rn, rm);
|
||||
}
|
||||
void fmax(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0100, rd, rn, rm);
|
||||
}
|
||||
void fmin(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0101, rd, rn, rm);
|
||||
}
|
||||
void fmaxnm(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0110, rd, rn, rm);
|
||||
}
|
||||
void fminnm(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0111, rd, rn, rm);
|
||||
}
|
||||
void fnmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b1000, rd, rn, rm);
|
||||
}
|
||||
|
||||
void fmul(SRegister rd, SRegister rn, SRegister rm) {
|
||||
Float2Source(0, 0, 0b00, 0b0000, rd.V(), rn.V(), rm.V());
|
||||
}
|
||||
@@ -1150,6 +1224,16 @@ public:
|
||||
}
|
||||
|
||||
// Floating-point conditional select
|
||||
void fcsel(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
FloatConditionalSelect(0, 0, ConvertedSize, rd, rn, rm, Cond);
|
||||
}
|
||||
|
||||
void fcsel(SRegister rd, SRegister rn, SRegister rm, Condition Cond) {
|
||||
FloatConditionalSelect(0, 0, 0b00, rd.V(), rn.V(), rm.V(), Cond);
|
||||
}
|
||||
@@ -1305,6 +1389,16 @@ private:
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
void Float1Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float1Source(M, S, ConvertedSize, opcode, rd, rn);
|
||||
}
|
||||
|
||||
// Floating-point compare
|
||||
void FloatCompare(uint32_t M, uint32_t S, uint32_t ftype, uint32_t op, uint32_t opcode2, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0010'0000'0000'0000;
|
||||
@@ -1337,6 +1431,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
// Floating-point data-processing (2 source)
|
||||
|
||||
void Float2Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1000'0000'0000;
|
||||
|
||||
@@ -1351,6 +1446,16 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void Float2Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float2Source(M, S, ConvertedSize, opcode, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Floating-point conditional select
|
||||
void FloatConditionalSelect(uint32_t M, uint32_t S, uint32_t ptype, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1100'0000'0000;
|
||||
|
||||
@@ -17,6 +17,12 @@ constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant:
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
|
||||
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
|
||||
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
|
||||
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
|
||||
{0x0706'0504'0302'0100ULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0111B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1001B
|
||||
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
|
||||
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {
|
||||
@@ -176,6 +182,96 @@ constexpr static auto SHUFPS_LUT {
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto DPPS_MASK {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint32_t Val[4];
|
||||
};
|
||||
|
||||
std::array<LUTType, 16> TotalLUT{};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1U;
|
||||
}
|
||||
return 0U;
|
||||
};
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
LUT.Val[2] = GetLUT(i, 2);
|
||||
LUT.Val[3] = GetLUT(i, 3);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto DPPD_MASK {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
|
||||
std::array<LUTType, 4> TotalLUT{};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1ULL;
|
||||
}
|
||||
return 0ULL;
|
||||
};
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto PBLENDW_LUT {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint16_t Val[8];
|
||||
};
|
||||
// 16-bit words in [127:112], [111:96], [95:80], [79:64], [63:48], [47:32], [31:16], [15:0] are selected using 8-bit swizzle.
|
||||
// Expectation for this LUT is to simulate PBLENDW with ARM's TBX (one register) instruction.
|
||||
// PBLENDW behaviour:
|
||||
// 16-bit words from the source is moved in to the destination based on the bit in the swizzle.
|
||||
// Dest[15:0] = Swizzle[0] ? Src[15:0] : Dest[15:0]
|
||||
// Dest[31:16] = Swizzle[1] ? Src[31:16] : Dest[31:16]
|
||||
// Dest[47:32] = Swizzle[2] ? Src[47:32] : Dest[47:32]
|
||||
// Dest[63:48] = Swizzle[3] ? Src[63:48] : Dest[63:48]
|
||||
// Dest[79:64] = Swizzle[4] ? Src[79:64] : Dest[79:64]
|
||||
// Dest[95:80] = Swizzle[5] ? Src[95:80] : Dest[95:80]
|
||||
// Dest[111:96] = Swizzle[6] ? Src[111:96] : Dest[111:96]
|
||||
// Dest[127:112] = Swizzle[7] ? Src[127:112] : Dest[127:112]
|
||||
|
||||
std::array<LUTType, 256> TotalLUT{};
|
||||
const uint16_t WordSelectionSrc[8] = {
|
||||
0x01'00,
|
||||
0x03'02,
|
||||
0x05'04,
|
||||
0x07'06,
|
||||
0x09'08,
|
||||
0x0B'0A,
|
||||
0x0D'0C,
|
||||
0x0F'0E,
|
||||
};
|
||||
|
||||
constexpr uint16_t OriginalDest = 0xFF'FF;
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
for (size_t j = 0; j < 8; ++j) {
|
||||
LUT.Val[j] = ((i >> j) & 1) ? WordSelectionSrc[j] : OriginalDest;
|
||||
}
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {
|
||||
|
||||
@@ -194,6 +290,9 @@ CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t I
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] = reinterpret_cast<uint64_t>(DPPS_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] = reinterpret_cast<uint64_t>(DPPD_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] = reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
|
||||
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
// Fill in telemetry values
|
||||
|
||||
@@ -33,14 +33,19 @@ namespace ProductNames {
|
||||
static const char ARM_A76[] = "Cortex-A76";
|
||||
static const char ARM_A76AE[] = "Cortex-A76AE";
|
||||
static const char ARM_V1[] = "Neoverse V1";
|
||||
static const char ARM_V2[] = "Neoverse V2";
|
||||
static const char ARM_A77[] = "Cortex-A77";
|
||||
static const char ARM_A78[] = "Cortex-A78";
|
||||
static const char ARM_A78AE[] = "Cortex-A78AE";
|
||||
static const char ARM_A78C[] = "Cortex-A78C";
|
||||
static const char ARM_A710[] = "Cortex-A710";
|
||||
static const char ARM_A715[] = "Cortex-A715";
|
||||
static const char ARM_A720[] = "Cortex-A720";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_X3[] = "Cortex-X3";
|
||||
static const char ARM_X4[] = "Cortex-X4";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
static const char ARM_E1[] = "Neoverse E1";
|
||||
@@ -49,6 +54,7 @@ namespace ProductNames {
|
||||
static const char ARM_A55[] = "Cortex-A55";
|
||||
static const char ARM_A65[] = "Cortex-A65";
|
||||
static const char ARM_A510[] = "Cortex-A510";
|
||||
static const char ARM_A520[] = "Cortex-A520";
|
||||
|
||||
static const char ARM_Kryo200[] = "Kryo 2xx";
|
||||
static const char ARM_Kryo300[] = "Kryo 3xx";
|
||||
@@ -100,11 +106,10 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
size_t CPUs = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
PerCPUData.resize(Cores);
|
||||
|
||||
uint64_t MIDR{};
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
std::error_code ec{};
|
||||
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
|
||||
@@ -138,10 +143,15 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 36> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 42> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
@@ -173,6 +183,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
// Typically Little CPU cores
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm}, // Apple M1 Icestorm
|
||||
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
{0x41, 0xd05, 0, ProductNames::ARM_A55}, // A55
|
||||
@@ -206,7 +217,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
fextl::vector<const CPUMIDR*> LittleCores;
|
||||
|
||||
// Separate CPU cores out to big or little selected
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
if (MIDROption) {
|
||||
@@ -322,7 +333,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
}
|
||||
else {
|
||||
// If we aren't hybrid then just claim everything is big
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
|
||||
@@ -368,7 +379,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
|
||||
// Processor Info and Features bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
|
||||
// Hypervisor bit is normally set but some applications have issues with it.
|
||||
uint32_t Hypervisor = HideHypervisorBit() ? 0 : 1;
|
||||
@@ -377,7 +387,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(CoreCount << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(0 << 24); // Local APIC ID
|
||||
|
||||
Res.ecx =
|
||||
@@ -484,7 +494,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
|
||||
if (Leaf == 0) {
|
||||
// Report L1D
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Data | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
@@ -508,7 +518,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
}
|
||||
else if (Leaf == 1) {
|
||||
// Report L1I
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Instruction | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
@@ -532,7 +542,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
// Report L2
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b010 << 5) | // Cache level
|
||||
@@ -556,7 +566,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
}
|
||||
else if (Leaf == 3) {
|
||||
// Report L3
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b011 << 5) | // Cache level
|
||||
@@ -1058,7 +1068,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) con
|
||||
(0 << 1) | // IRPerf: Instructions retired count support
|
||||
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
|
||||
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
Res.ecx =
|
||||
(0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
@@ -1156,7 +1166,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_001Dh(uint32_t Leaf) con
|
||||
}
|
||||
else if (Leaf == 3) {
|
||||
// Report L3
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b011 << 5) | // Cache level
|
||||
@@ -1197,6 +1207,7 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
|
||||
void CPUIDEmu::Init(FEXCore::Context::ContextImpl *ctx) {
|
||||
CTX = ctx;
|
||||
Cores = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
@@ -113,7 +113,7 @@ public:
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
bool Hybrid{};
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
uint32_t Cores{};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
|
||||
@@ -9,12 +9,11 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "FEXCore/Utils/DeferredSignalMutex.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
@@ -46,6 +45,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include "FEXCore/Utils/SignalScopeGuards.h"
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
@@ -76,64 +76,6 @@ $end_info$
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct ThreadLocalData {
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
};
|
||||
|
||||
constexpr std::array<std::string_view const, 22> FlagNames = {
|
||||
"CF",
|
||||
"",
|
||||
"PF",
|
||||
"",
|
||||
"AF",
|
||||
"",
|
||||
"ZF",
|
||||
"SF",
|
||||
"TF",
|
||||
"IF",
|
||||
"DF",
|
||||
"OF",
|
||||
"IOPL",
|
||||
"",
|
||||
"NT",
|
||||
"",
|
||||
"RF",
|
||||
"VM",
|
||||
"AC",
|
||||
"VIF",
|
||||
"VIP",
|
||||
"ID",
|
||||
};
|
||||
|
||||
std::string_view const& GetFlagName(unsigned Flag) {
|
||||
return FlagNames[Flag];
|
||||
}
|
||||
|
||||
constexpr std::array<std::string_view const, 16> RegNames = {
|
||||
"rax",
|
||||
"rbx",
|
||||
"rcx",
|
||||
"rdx",
|
||||
"rsi",
|
||||
"rdi",
|
||||
"rbp",
|
||||
"rsp",
|
||||
"r8",
|
||||
"r9",
|
||||
"r10",
|
||||
"r11",
|
||||
"r12",
|
||||
"r13",
|
||||
"r14",
|
||||
"r15",
|
||||
};
|
||||
|
||||
std::string_view const& GetGRegName(unsigned Reg) {
|
||||
return RegNames[Reg];
|
||||
}
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl()
|
||||
: IRCaptureCache {this} {
|
||||
@@ -157,6 +99,8 @@ namespace FEXCore::Context {
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
CPUID.Init(this);
|
||||
}
|
||||
|
||||
ContextImpl::~ContextImpl() {
|
||||
@@ -221,45 +165,66 @@ namespace FEXCore::Context {
|
||||
return Frame->State.rip;
|
||||
}
|
||||
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) {
|
||||
uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
uint32_t EFLAGS{};
|
||||
|
||||
// Currently these flags just map 1:1 inside of the resulting value.
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
if (i == X86State::RFLAG_PF_LOC || i == X86State::RFLAG_AF_LOC) {
|
||||
// Intentionally do nothing.
|
||||
// These contain multiple bits which can corrupt other members when compacted.
|
||||
continue;
|
||||
switch (i) {
|
||||
case X86State::RFLAG_CF_RAW_LOC:
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
case X86State::RFLAG_AF_RAW_LOC:
|
||||
case X86State::RFLAG_ZF_RAW_LOC:
|
||||
case X86State::RFLAG_SF_RAW_LOC:
|
||||
case X86State::RFLAG_OF_RAW_LOC:
|
||||
// Intentionally do nothing.
|
||||
// These contain multiple bits which can corrupt other members when compacted.
|
||||
break;
|
||||
default:
|
||||
EFLAGS |= uint32_t{Frame->State.flags[i]} << i;
|
||||
break;
|
||||
}
|
||||
|
||||
EFLAGS |= uint32_t{Frame->State.flags[i]} << i;
|
||||
}
|
||||
|
||||
// SF/ZF/CF/OF are packed in a 32-bit value in RFLAG_NZCV_LOC.
|
||||
uint32_t Packed_NZCV{};
|
||||
memcpy(&Packed_NZCV, &Frame->State.flags[X86State::RFLAG_NZCV_LOC], sizeof(Packed_NZCV));
|
||||
uint32_t OF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_LOC)) & 1;
|
||||
uint32_t CF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_LOC)) & 1;
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_LOC)) & 1;
|
||||
uint32_t SF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_LOC)) & 1;
|
||||
if (WasInJIT) {
|
||||
// If we were in the JIT then NZCV is in the CPU's PSTATE object.
|
||||
// Packed in to the same bit locations as RFLAG_NZCV_LOC.
|
||||
Packed_NZCV = PSTATE;
|
||||
|
||||
// If we were in the JIT then PF and AF are in registers.
|
||||
// Move them to the CPUState frame now.
|
||||
Frame->State.pf_raw = HostGPRs[CPU::REG_PF.Idx()];
|
||||
Frame->State.af_raw = HostGPRs[CPU::REG_AF.Idx()];
|
||||
}
|
||||
else {
|
||||
// If we were not in the JIT then the NZCV state is stored in the CPUState RFLAG_NZCV_LOC.
|
||||
// SF/ZF/CF/OF are packed in a 32-bit value in RFLAG_NZCV_LOC.
|
||||
memcpy(&Packed_NZCV, &Frame->State.flags[X86State::RFLAG_NZCV_LOC], sizeof(Packed_NZCV));
|
||||
}
|
||||
|
||||
uint32_t OF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC)) & 1;
|
||||
uint32_t CF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC)) & 1;
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC)) & 1;
|
||||
uint32_t SF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC)) & 1;
|
||||
|
||||
// Pack in to EFLAGS
|
||||
EFLAGS |= OF << X86State::RFLAG_OF_LOC;
|
||||
EFLAGS |= CF << X86State::RFLAG_CF_LOC;
|
||||
EFLAGS |= ZF << X86State::RFLAG_ZF_LOC;
|
||||
EFLAGS |= SF << X86State::RFLAG_SF_LOC;
|
||||
EFLAGS |= OF << X86State::RFLAG_OF_RAW_LOC;
|
||||
EFLAGS |= CF << X86State::RFLAG_CF_RAW_LOC;
|
||||
EFLAGS |= ZF << X86State::RFLAG_ZF_RAW_LOC;
|
||||
EFLAGS |= SF << X86State::RFLAG_SF_RAW_LOC;
|
||||
|
||||
// PF calculation is deferred, calculate it now.
|
||||
// Popcount the 8-bit flag and then extract the lower bit.
|
||||
uint32_t PFByte = Frame->State.flags[X86State::RFLAG_PF_LOC];
|
||||
uint32_t PFByte = Frame->State.pf_raw & 0xff;
|
||||
uint32_t PF = std::popcount(PFByte ^ 1) & 1;
|
||||
EFLAGS |= PF << X86State::RFLAG_PF_LOC;
|
||||
EFLAGS |= PF << X86State::RFLAG_PF_RAW_LOC;
|
||||
|
||||
// AF calculation is deferred, calculate it now.
|
||||
// XOR with PF byte and extract bit 4.
|
||||
uint32_t AF = ((Frame->State.flags[X86State::RFLAG_AF_LOC] ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_LOC;
|
||||
uint32_t AF = ((Frame->State.af_raw ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
|
||||
|
||||
return EFLAGS;
|
||||
}
|
||||
@@ -268,21 +233,21 @@ namespace FEXCore::Context {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
switch (i) {
|
||||
case X86State::RFLAG_OF_LOC:
|
||||
case X86State::RFLAG_CF_LOC:
|
||||
case X86State::RFLAG_ZF_LOC:
|
||||
case X86State::RFLAG_SF_LOC:
|
||||
case X86State::RFLAG_OF_RAW_LOC:
|
||||
case X86State::RFLAG_CF_RAW_LOC:
|
||||
case X86State::RFLAG_ZF_RAW_LOC:
|
||||
case X86State::RFLAG_SF_RAW_LOC:
|
||||
// Intentionally do nothing.
|
||||
break;
|
||||
case X86State::RFLAG_AF_LOC:
|
||||
case X86State::RFLAG_AF_RAW_LOC:
|
||||
// AF stored in bit 4 in our internal representation. It is also
|
||||
// XORed with byte 4 of the PF byte, but we write that as zero here so
|
||||
// we don't need any special handling for that.
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
|
||||
Frame->State.af_raw = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
|
||||
break;
|
||||
case X86State::RFLAG_PF_LOC:
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
// PF is inverted in our internal representation.
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 0 : 1;
|
||||
Frame->State.pf_raw = (EFLAGS & (1U << i)) ? 0 : 1;
|
||||
break;
|
||||
default:
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 1 : 0;
|
||||
@@ -292,10 +257,10 @@ namespace FEXCore::Context {
|
||||
|
||||
// Calculate packed NZCV
|
||||
uint32_t Packed_NZCV{};
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_OF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_ZF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_SF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_OF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_ZF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_SF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC) : 0;
|
||||
memcpy(&Frame->State.flags[X86State::RFLAG_NZCV_LOC], &Packed_NZCV, sizeof(Packed_NZCV));
|
||||
|
||||
// Reserved, Read-As-1, Write-as-1
|
||||
@@ -352,47 +317,28 @@ namespace FEXCore::Context {
|
||||
// Give this configuration to the SignalDelegator.
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
if (Config.GdbServer) {
|
||||
StartGdbServer();
|
||||
}
|
||||
else {
|
||||
StopGdbServer();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#else
|
||||
// WIN32 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
|
||||
using namespace FEXCore::Core;
|
||||
if (Config.GdbServer) {
|
||||
// If gdbserver is enabled then this needs to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
// FEX needs to start paused when gdb is enabled.
|
||||
StartPaused = true;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(nullptr, 0);
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(InitialRIP, StackPointer, nullptr, 0);
|
||||
|
||||
// We are the parent thread
|
||||
ParentThread = Thread;
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
InitializeThreadData(Thread);
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::StartGdbServer() {
|
||||
#ifndef _WIN32
|
||||
if (!DebugServer) {
|
||||
DebugServer = fextl::make_unique<GdbServer>(this);
|
||||
StartPaused = true;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::StopGdbServer() {
|
||||
#ifndef _WIN32
|
||||
DebugServer.reset();
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
@@ -565,14 +511,6 @@ namespace FEXCore::Context {
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
}
|
||||
|
||||
int ContextImpl::GetProgramStatus() const {
|
||||
return ParentThread->StatusCode;
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeThreadData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CPUBackend->Initialize();
|
||||
}
|
||||
|
||||
struct ExecutionThreadHandler {
|
||||
ContextImpl *This;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
@@ -671,20 +609,22 @@ namespace FEXCore::Context {
|
||||
Thread->PassManager->Finalize();
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState* ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{};
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
if (NewThreadState) {
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(&Thread->CurrentFrame->State, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
// Set up the thread manager state
|
||||
Thread->ThreadManager.parent_tid = ParentTID;
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
InitializeCompiler(Thread);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
|
||||
@@ -826,7 +766,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
ContextImpl::GenerateIRResult ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
ContextImpl::GenerateIRResult ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
@@ -851,7 +791,7 @@ namespace FEXCore::Context {
|
||||
|
||||
bool HadDispatchError {false};
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, MaxInst, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Thread, Start, Length);
|
||||
}
|
||||
@@ -1005,7 +945,7 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
@@ -1056,7 +996,7 @@ namespace FEXCore::Context {
|
||||
|
||||
if (IRList == nullptr) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols());
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
@@ -1077,7 +1017,7 @@ namespace FEXCore::Context {
|
||||
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
|
||||
// In the future with code caching getting wired up, we will pass the rest of the data forward.
|
||||
// TODO: Pass the data forward when code caching is wired up to this.
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get(), GetGdbServerStatus()).BlockEntry,
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get()).BlockEntry,
|
||||
.IRData = IRList,
|
||||
.DebugData = DebugData,
|
||||
.RAData = std::move(RAData),
|
||||
@@ -1098,12 +1038,12 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
@@ -1118,7 +1058,7 @@ namespace FEXCore::Context {
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {}, Length {};
|
||||
|
||||
auto [Code, IR, Data, RAData, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP);
|
||||
auto [Code, IR, Data, RAData, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
CodePtr = Code;
|
||||
IRList = IR;
|
||||
DebugData = Data;
|
||||
@@ -1286,7 +1226,7 @@ namespace FEXCore::Context {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSectionWithFallback(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
}
|
||||
@@ -1295,7 +1235,7 @@ namespace FEXCore::Context {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSectionWithFallback(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
@@ -1323,7 +1263,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
@@ -1371,17 +1311,11 @@ namespace FEXCore::Context {
|
||||
|
||||
IR::AOTIRCacheEntry *ContextImpl::LoadAOTIRCacheEntry(const fextl::string &filename) {
|
||||
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
}
|
||||
return rv;
|
||||
}
|
||||
|
||||
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry *Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
if (DebugServer) {
|
||||
DebugServer->AlertLibrariesChanged();
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
|
||||
@@ -53,15 +53,22 @@ uint64_t Dispatcher::GetCompileBlockPtr() {
|
||||
return CompileBlockPtr.Data;
|
||||
}
|
||||
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
|
||||
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
|
||||
: Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
|
||||
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
|
||||
, CTX {ctx}
|
||||
, config {config} {
|
||||
EmitDispatcher();
|
||||
}
|
||||
|
||||
Dispatcher::~Dispatcher() {
|
||||
auto BufferSize = GetBufferSize();
|
||||
if (BufferSize) {
|
||||
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
|
||||
}
|
||||
}
|
||||
|
||||
void Dispatcher::EmitDispatcher() {
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
@@ -121,8 +128,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
|
||||
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, RipReg.R());
|
||||
b(ARMEmitter::Condition::CC_NE, &FullLookup);
|
||||
sub(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &FullLookup);
|
||||
|
||||
br(ARMEmitter::Reg::r3);
|
||||
|
||||
@@ -167,8 +174,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
cmp(ARMEmitter::XReg::x1, RipReg);
|
||||
b(ARMEmitter::Condition::CC_NE, &NoBlock);
|
||||
sub(ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
|
||||
@@ -225,7 +232,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
@@ -262,7 +269,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
@@ -393,7 +400,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->X86CodeGen.CallbackReturn);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, CTX->Config.Is64BitMode ? 16 : 12);
|
||||
str(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
@@ -549,40 +556,6 @@ void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_
|
||||
|
||||
#endif
|
||||
|
||||
size_t Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
FEXCore::ARMEmitter::Emitter emit{CodeBuffer, MaxGDBPauseCheckSize};
|
||||
|
||||
ARMEmitter::ForwardLabel RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(FEXCore::Context::ContextImpl::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Thread));
|
||||
emit.ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, offsetof(FEXCore::Core::InternalThreadState, CTX)); // Get Context
|
||||
emit.ldr(ARMEmitter::WReg::w0, ARMEmitter::Reg::r0, offsetof(FEXCore::Context::ContextImpl, Config.RunningMode));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cbz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, &RunBlock);
|
||||
{
|
||||
ARMEmitter::ForwardLabel l_GuestRIP;
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.ldr(ARMEmitter::XReg::x0, &l_GuestRIP);
|
||||
emit.str(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Stop the thread
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.br(ARMEmitter::Reg::r0);
|
||||
emit.Bind(&l_GuestRIP);
|
||||
emit.dc64(GuestRIP);
|
||||
}
|
||||
emit.Bind(&RunBlock);
|
||||
|
||||
auto UsedBytes = emit.GetCursorOffset();
|
||||
emit.ClearICache(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
|
||||
@@ -43,7 +43,7 @@ public:
|
||||
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
|
||||
|
||||
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config);
|
||||
~Dispatcher() = default;
|
||||
~Dispatcher();
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
@@ -71,11 +71,6 @@ public:
|
||||
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
// These are across all arches for now
|
||||
static constexpr size_t MaxGDBPauseCheckSize = 128;
|
||||
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) ;
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
@@ -90,7 +85,9 @@ public:
|
||||
#endif
|
||||
|
||||
uint16_t GetSRAGPRCount() const {
|
||||
return StaticRegisters.size();
|
||||
// PF/AF are the final two SRA registers.
|
||||
// Only return the SRA for GPRs.
|
||||
return StaticRegisters.size() - 2;
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const {
|
||||
@@ -98,7 +95,7 @@ public:
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const {
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
for (size_t i = 0; i < StaticRegisters.size() - 2; ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1095,7 +1095,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, uint64_t MaxInst, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
@@ -1133,6 +1133,10 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
if (MaxInst == 0) {
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1195,9 +1199,9 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
CanContinue = true;
|
||||
}
|
||||
|
||||
bool FinalInstruction = DecodedSize >= CTX->Config.MaxInstPerBlock ||
|
||||
bool FinalInstruction = DecodedSize >= MaxInst ||
|
||||
DecodedSize >= DefaultDecodedBufferSize ||
|
||||
TotalInstructions >= CTX->Config.MaxInstPerBlock;
|
||||
TotalInstructions >= MaxInst;
|
||||
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP) {
|
||||
// If we have multiblock enabled
|
||||
|
||||
@@ -34,7 +34,7 @@ public:
|
||||
|
||||
Decoder(FEXCore::Context::ContextImpl *ctx);
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, uint64_t MaxInst, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
|
||||
DecodedBlockInformation const *GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
|
||||
@@ -69,61 +69,29 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
return;
|
||||
}
|
||||
|
||||
const bool DisableAVX = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEAVX;
|
||||
const bool EnableAVX = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEAVX;
|
||||
LogMan::Throw::AFmt(!(DisableAVX && EnableAVX), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
#define ENABLE_DISABLE_OPTION(name, enum_name) \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
|
||||
|
||||
const bool DisableAVX2 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEAVX2;
|
||||
const bool EnableAVX2 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEAVX2;
|
||||
LogMan::Throw::AFmt(!(DisableAVX2 && EnableAVX2), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
ENABLE_DISABLE_OPTION(AVX, AVX);
|
||||
ENABLE_DISABLE_OPTION(AVX2, AVX2);
|
||||
ENABLE_DISABLE_OPTION(SVE, SVE);
|
||||
ENABLE_DISABLE_OPTION(AFP, AFP);
|
||||
ENABLE_DISABLE_OPTION(LRCPC, LRCPC);
|
||||
ENABLE_DISABLE_OPTION(LRCPC2, LRCPC2);
|
||||
ENABLE_DISABLE_OPTION(CSSC, CSSC);
|
||||
ENABLE_DISABLE_OPTION(PMULL128, PMULL128);
|
||||
ENABLE_DISABLE_OPTION(RNG, RNG);
|
||||
ENABLE_DISABLE_OPTION(CLZERO, CLZERO);
|
||||
ENABLE_DISABLE_OPTION(Atomics, ATOMICS);
|
||||
ENABLE_DISABLE_OPTION(FCMA, FCMA);
|
||||
ENABLE_DISABLE_OPTION(FlagM, FLAGM);
|
||||
ENABLE_DISABLE_OPTION(FlagM2, FLAGM2);
|
||||
ENABLE_DISABLE_OPTION(Crypto, CRYPTO);
|
||||
ENABLE_DISABLE_OPTION(RPRES, RPRES);
|
||||
|
||||
const bool DisableSVE = HostFeatures() & FEXCore::Config::HostFeatures::DISABLESVE;
|
||||
const bool EnableSVE = HostFeatures() & FEXCore::Config::HostFeatures::ENABLESVE;
|
||||
LogMan::Throw::AFmt(!(DisableSVE && EnableSVE), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableAFP = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEAFP;
|
||||
const bool EnableAFP = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEAFP;
|
||||
LogMan::Throw::AFmt(!(DisableAFP && EnableAFP), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableLRCPC = HostFeatures() & FEXCore::Config::HostFeatures::DISABLELRCPC;
|
||||
const bool EnableLRCPC = HostFeatures() & FEXCore::Config::HostFeatures::ENABLELRCPC;
|
||||
LogMan::Throw::AFmt(!(DisableLRCPC && EnableLRCPC), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableLRCPC2 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLELRCPC2;
|
||||
const bool EnableLRCPC2 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLELRCPC2;
|
||||
LogMan::Throw::AFmt(!(DisableLRCPC2 && EnableLRCPC2), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableCSSC = HostFeatures() & FEXCore::Config::HostFeatures::DISABLECSSC;
|
||||
const bool EnableCSSC = HostFeatures() & FEXCore::Config::HostFeatures::ENABLECSSC;
|
||||
LogMan::Throw::AFmt(!(DisableCSSC && EnableCSSC), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisablePMULL128 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEPMULL128;
|
||||
const bool EnablePMULL128 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEPMULL128;
|
||||
LogMan::Throw::AFmt(!(DisablePMULL128 && EnablePMULL128), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableRNG = HostFeatures() & FEXCore::Config::HostFeatures::DISABLERNG;
|
||||
const bool EnableRNG = HostFeatures() & FEXCore::Config::HostFeatures::ENABLERNG;
|
||||
LogMan::Throw::AFmt(!(DisableRNG && EnableRNG), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableCLZERO = HostFeatures() & FEXCore::Config::HostFeatures::DISABLECLZERO;
|
||||
const bool EnableCLZERO = HostFeatures() & FEXCore::Config::HostFeatures::ENABLECLZERO;
|
||||
LogMan::Throw::AFmt(!(DisableCLZERO && EnableCLZERO), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableAtomics = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEATOMICS;
|
||||
const bool EnableAtomics = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEATOMICS;
|
||||
LogMan::Throw::AFmt(!(DisableAtomics && EnableAtomics), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableFCMA = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEFCMA;
|
||||
const bool EnableFCMA = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEFCMA;
|
||||
LogMan::Throw::AFmt(!(DisableFCMA && EnableFCMA), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableFlagM = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEFLAGM;
|
||||
const bool EnableFlagM = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEFLAGM;
|
||||
LogMan::Throw::AFmt(!(DisableFlagM && EnableFlagM), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableFlagM2 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEFLAGM2;
|
||||
const bool EnableFlagM2 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEFLAGM2;
|
||||
LogMan::Throw::AFmt(!(DisableFlagM2 && EnableFlagM2), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
|
||||
if (EnableAVX) {
|
||||
Features->SupportsAVX = true;
|
||||
@@ -144,10 +112,10 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
Features->SupportsSVE = false;
|
||||
}
|
||||
if (EnableAFP) {
|
||||
Features->SupportsFlushInputsToZero = true;
|
||||
Features->SupportsAFP = true;
|
||||
}
|
||||
else if (DisableAFP) {
|
||||
Features->SupportsFlushInputsToZero = false;
|
||||
Features->SupportsAFP = false;
|
||||
}
|
||||
if (EnableLRCPC) {
|
||||
Features->SupportsRCPC = true;
|
||||
@@ -209,11 +177,33 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
else if (DisableFlagM2) {
|
||||
Features->SupportsFlagM2 = false;
|
||||
}
|
||||
if (EnableCrypto) {
|
||||
Features->SupportsAES = true;
|
||||
Features->SupportsCRC = true;
|
||||
Features->SupportsSHA = true;
|
||||
Features->SupportsPMULL_128Bit = true;
|
||||
}
|
||||
else if (DisableCrypto) {
|
||||
Features->SupportsAES = false;
|
||||
Features->SupportsCRC = false;
|
||||
Features->SupportsSHA = false;
|
||||
Features->SupportsPMULL_128Bit = false;
|
||||
}
|
||||
if (EnableRPRES) {
|
||||
Features->SupportsRPRES = true;
|
||||
}
|
||||
else if (DisableRPRES) {
|
||||
Features->SupportsRPRES = false;
|
||||
}
|
||||
}
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
// Vixl simulator doesn't support AFP.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kAFP);
|
||||
// Vixl simulator doesn't support RPRES.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kRPRES);
|
||||
#elif !defined(_WIN32)
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#else
|
||||
@@ -223,11 +213,13 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) &&
|
||||
Features.Has(vixl::CPUFeatures::Feature::kSHA2);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
|
||||
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsAFP = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
@@ -235,6 +227,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsFCMA = Features.Has(vixl::CPUFeatures::Feature::kFcma);
|
||||
SupportsFlagM = Features.Has(vixl::CPUFeatures::Feature::kFlagM);
|
||||
SupportsFlagM2 = Features.Has(vixl::CPUFeatures::Feature::kAXFlag);
|
||||
SupportsRPRES = Features.Has(vixl::CPUFeatures::Feature::kRPRES);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
@@ -249,11 +242,15 @@ HostFeatures::HostFeatures() {
|
||||
#endif
|
||||
// TODO: AVX2 is currently unsupported. Disable until the remaining features are implemented.
|
||||
SupportsAVX2 = false;
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
SupportsCLWB = true;
|
||||
|
||||
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
|
||||
SupportsAFP = false;
|
||||
// RPRES has a dependency on AFP. Disable it until AFP is enabled.
|
||||
SupportsRPRES = false;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
@@ -291,6 +288,8 @@ HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// simulator doesn't support dc(ZVA)
|
||||
SupportsCLZERO = false;
|
||||
// Simulator doesn't support SHA
|
||||
SupportsSHA = false;
|
||||
#else
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
@@ -336,7 +335,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsCLZERO = data[1] & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsAFP = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
@@ -87,7 +87,7 @@ DEF_OP(Add) {
|
||||
|
||||
DEF_OP(AddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AddNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
@@ -98,78 +98,63 @@ DEF_OP(AddNZCV) {
|
||||
} else {
|
||||
cmn(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(GetReg(Node), ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(AdcNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AdcNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
// TODO: Optimize this out
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->NZCV.ID()));
|
||||
|
||||
adcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(SbbNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SbbNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
// Carry-in needs to be inverted for subtractions due to carry versus borrow
|
||||
// distinction between x86 and arm.
|
||||
// See below remarks on cfinv
|
||||
eor(ARMEmitter::Size::i32Bit, TMP1, GetReg(Op->NZCV.ID()), 1u << 29);
|
||||
|
||||
// TODO: Optimize this out
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
sbcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// The carry flag produced by arm64 sbcs is inverted compared to the x86 carry
|
||||
// flag. Invert it now.
|
||||
//
|
||||
// TODO: Once we optimize out the mrs, this will become a cfinv operation, but
|
||||
// that's only available with Feat_FlagM. For now the portable way is to flip
|
||||
// bit 29 (carry) manually.
|
||||
eor(ARMEmitter::Size::i32Bit, Dst, Dst, 1u << 29);
|
||||
}
|
||||
|
||||
DEF_OP(TestNZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src = GetReg(Op->Src1.ID());
|
||||
uint64_t Const;
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
// Shift the sign bit into place, clearing out the garbage in upper bits.
|
||||
// setf+rmif would avoid the scratch register, but higher latency on M1.
|
||||
// Adding zero does an effective test, setting NZ according to the result and
|
||||
// zeroing CV.
|
||||
if (OpSize < 4) {
|
||||
lsl(EmitSize, Dst, Src, 32 - (OpSize * 8));
|
||||
Src = Dst;
|
||||
// Cheaper to and+cmn than to lsl+lsl+tst, so do the and ourselves if
|
||||
// needed.
|
||||
if (Op->Src1 != Op->Src2) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
and_(EmitSize, TMP1, Src1, Const);
|
||||
} else {
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
and_(EmitSize, TMP1, Src1, Src2);
|
||||
}
|
||||
|
||||
Src1 = TMP1;
|
||||
}
|
||||
|
||||
unsigned Shift = 32 - (OpSize * 8);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Src1, ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
tst(EmitSize, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
tst(EmitSize, Src, Src);
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(Sub) {
|
||||
@@ -187,9 +172,19 @@ DEF_OP(Sub) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SubShift) {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(SubNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SubNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
@@ -203,20 +198,70 @@ DEF_OP(SubNZCV) {
|
||||
} else {
|
||||
cmp(EmitSize, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
DEF_OP(CarryInvert) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
cfinv();
|
||||
}
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
DEF_OP(RmifNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_RmifNZCV>();
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
|
||||
if (Op->InvertCarry) {
|
||||
// The carry flag produced by arm64 subs is inverted compared to the x86 carry
|
||||
// flag. Invert it now.
|
||||
//
|
||||
// TODO: Once we optimize out the mrs, this will become a cfinv operation, but
|
||||
// that's only available with Feat_FlagM. For now the portable way is to flip
|
||||
// bit 29 (carry) manually.
|
||||
eor(ARMEmitter::Size::i32Bit, Dst, Dst, 1u << 29);
|
||||
rmif(GetReg(Op->Src.ID()).X(), Op->Rotate, Op->Mask);
|
||||
}
|
||||
|
||||
DEF_OP(AXFlag) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM2, "Unsupported flagm2 op");
|
||||
axflag();
|
||||
}
|
||||
|
||||
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CondAddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
ARMEmitter::StatusFlags Flags = (ARMEmitter::StatusFlags)Op->FalseNZCV;
|
||||
uint64_t Const = 0;
|
||||
auto Src1 = IsInlineConstant(Op->Src1, &Const) ? ARMEmitter::Reg::zr :
|
||||
GetReg(Op->Src1.ID());
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "Unsupported inline constant");
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ccmn(EmitSize, Src1, Const, Flags, MapSelectCC(Op->Cond));
|
||||
} else {
|
||||
ccmn(EmitSize, Src1, GetReg(Op->Src2.ID()), Flags, MapSelectCC(Op->Cond));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -227,27 +272,10 @@ DEF_OP(Neg) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCSSC) {
|
||||
// On CSSC supporting processors, this turns in to one instruction and doesn't modify flags.
|
||||
abs(EmitSize, Dst, Src);
|
||||
}
|
||||
else {
|
||||
cmp(EmitSize, Src, 0);
|
||||
cneg(EmitSize, Dst, Src, ARMEmitter::Condition::CC_MI);
|
||||
}
|
||||
if (Op->Cond == FEXCore::IR::COND_AL)
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
else
|
||||
cneg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()), MapSelectCC(Op->Cond));
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
@@ -549,6 +577,16 @@ DEF_OP(Xor) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(XorShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
eor(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(Lshl) {
|
||||
auto Op = IROp->C<IR::IROp_Lshl>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -1217,6 +1255,11 @@ DEF_OP(Bfi) {
|
||||
// If Dst and SrcDst match then this turns in to a simple BFI instruction.
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else if (Dst != Src) {
|
||||
// If the destination isn't the source then we can move the DstSrc and insert directly.
|
||||
mov(EmitSize, Dst, SrcDst);
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else {
|
||||
// Destination didn't match the dst source register.
|
||||
// TODO: Inefficient until FEX can have RA constraints here.
|
||||
@@ -1246,6 +1289,11 @@ DEF_OP(Bfxil) {
|
||||
// If Dst and SrcDst match then this turns in to a single instruction.
|
||||
bfxil(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else if (Dst != Src) {
|
||||
// If the destination isn't the source then we can move the DstSrc and insert directly.
|
||||
mov(EmitSize, Dst, SrcDst);
|
||||
bfxil(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else {
|
||||
// Destination didn't match the dst source register.
|
||||
// TODO: Inefficient until FEX can have RA constraints here.
|
||||
@@ -1287,36 +1335,6 @@ DEF_OP(Sbfe) {
|
||||
sbfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
|
||||
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_ANDZ:
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_ANDNZ:
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU:return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -1325,28 +1343,15 @@ DEF_OP(Select) {
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
|
||||
Op->Cond == FEXCore::IR::COND_ANDNZ;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!tests || IsGPR(Op->Cmp1.ID()), "Only GPRs can be tested");
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
|
||||
if (tests) {
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
tst(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
tst(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
cmp(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
cmp(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
@@ -1385,6 +1390,38 @@ DEF_OP(Select) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelect) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelect>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
|
||||
uint64_t const_true, const_false;
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
uint64_t all_ones = OpSize == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
|
||||
if (is_const_true) {
|
||||
if (is_const_false != true || !(const_true == 1 || const_true == all_ones) || const_false != 0) {
|
||||
LOGMAN_MSG_A_FMT("NZCVSelect: Unsupported constant");
|
||||
}
|
||||
|
||||
if (const_true == all_ones)
|
||||
csetm(EmitSize, Dst, cc);
|
||||
else
|
||||
cset(EmitSize, Dst, cc);
|
||||
} else if (is_const_false) {
|
||||
LOGMAN_THROW_A_FMT(const_false == 0, "NZCVSelect: unsupported constant");
|
||||
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), ARMEmitter::Reg::zr, cc);
|
||||
} else {
|
||||
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), GetReg(Op->FalseVal.ID()), cc);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1499,41 +1536,10 @@ DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
const auto EmitSubSize = Op->ElementSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Scalar1 = GetVReg(Op->Scalar1.ID());
|
||||
ARMEmitter::VRegister Scalar2 = GetVReg(Op->Scalar2.ID());
|
||||
|
||||
fcmp(EmitSubSize, Scalar1, Scalar2);
|
||||
bool set = false;
|
||||
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ)) {
|
||||
LOGMAN_THROW_AA_FMT(IR::FCMP_FLAG_EQ == 0, "IR::FCMP_FLAG_EQ must equal 0");
|
||||
// EQ or unordered
|
||||
cset(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Condition::CC_EQ); // Z = 1
|
||||
csinc(ARMEmitter::Size::i64Bit, Dst, Dst, ARMEmitter::Reg::zr, ARMEmitter::Condition::CC_VC); // IF !V ? Z : 1
|
||||
set = true;
|
||||
}
|
||||
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
|
||||
// LT or unordered
|
||||
cset(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Condition::CC_LT);
|
||||
if (!set) {
|
||||
lsl(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_LT);
|
||||
set = true;
|
||||
} else {
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_LT, 1);
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED)) {
|
||||
cset(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Condition::CC_VS);
|
||||
if (!set) {
|
||||
lsl(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_UNORDERED);
|
||||
set = true;
|
||||
} else {
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP2, IR::FCMP_FLAG_UNORDERED, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
@@ -193,6 +193,33 @@ DEF_OP(AtomicAnd) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicCLR) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicCLR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stclrl(SubEmitSize, Src, MemSrc);
|
||||
}
|
||||
else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
bic(EmitSize, TMP2, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -247,6 +274,27 @@ DEF_OP(AtomicXor) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicNeg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
}
|
||||
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
@@ -75,9 +75,10 @@ DEF_OP(ExitFunction) {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x3, ARMEmitter::ShiftType::LSL, 4);
|
||||
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x1, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
|
||||
cmp(ARMEmitter::XReg::x0, RipReg.X());
|
||||
b(ARMEmitter::Condition::CC_NE, &FullLookup);
|
||||
sub(TMP1, ARMEmitter::XReg::x0, RipReg.X());
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
|
||||
br(ARMEmitter::Reg::r1);
|
||||
|
||||
Bind(&FullLookup);
|
||||
@@ -96,9 +97,7 @@ DEF_OP(Jump) {
|
||||
|
||||
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_ANDZ:
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_ANDNZ:
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
@@ -116,8 +115,8 @@ static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI:
|
||||
case FEXCore::IR::COND_PL:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unsupported compare type");
|
||||
return ARMEmitter::Condition::CC_NV;
|
||||
@@ -129,42 +128,27 @@ DEF_OP(CondJump) {
|
||||
|
||||
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
|
||||
Op->Cond == FEXCore::IR::COND_ANDNZ;
|
||||
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
const auto SubSize = ARMEmitter::ToVectorSizePair(Op->CompareSize == 4 ? ARMEmitter::SubRegSize::i32Bit : ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else if (isConst && Const == 0 && Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
if (Op->FromNZCV) {
|
||||
b(MapBranchCC(Op->Cond), TrueTargetLabel);
|
||||
} else {
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
if (tests) {
|
||||
if (isConst) {
|
||||
tst(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
tst(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
}
|
||||
} else {
|
||||
if (isConst) {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
}
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
fcmp(SubSize.Scalar, GetVReg(Op->Cmp1.ID()), GetVReg(Op->Cmp2.ID()));
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("CondJump: Expected GPR or FPR");
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
|
||||
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ ||
|
||||
Op->Cond.Val == FEXCore::IR::COND_NEQ,
|
||||
"CondJump: Expected simple condition");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else {
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
}
|
||||
|
||||
b(MapBranchCC(Op->Cond), TrueTargetLabel);
|
||||
// TODO: Wire up tbz/tbnz
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
@@ -467,7 +451,7 @@ DEF_OP(CPUID) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
DEF_OP(XGetBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -12,12 +12,12 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
DEF_OP(VAESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetVReg(Node), GetVReg(Op->Vector.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
DEF_OP(VAESEnc) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -44,7 +44,7 @@ DEF_OP(AESEnc) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
DEF_OP(VAESEncLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -69,7 +69,7 @@ DEF_OP(AESEncLast) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
DEF_OP(VAESDec) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -96,7 +96,7 @@ DEF_OP(AESDec) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
DEF_OP(VAESDecLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -121,7 +121,7 @@ DEF_OP(AESDecLast) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
DEF_OP(VAESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
@@ -179,6 +179,32 @@ DEF_OP(CRC32) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
|
||||
sha1h(Dst.S(), Src.S());
|
||||
}
|
||||
|
||||
DEF_OP(VSha256U0) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U0>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256su0(Dst, Src2);
|
||||
}
|
||||
else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256su0(VTMP1, Src2);
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -530,9 +530,11 @@ void Arm64JITCore::Op_NoOp(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx, 0)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128{ctx->HostFeatures.SupportsSVE}
|
||||
, HostSupportsSVE256{ctx->HostFeatures.SupportsAVX}
|
||||
, HostSupportsRPRES{ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP{ctx->HostFeatures.SupportsAFP}
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
@@ -687,8 +689,7 @@ bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
@@ -700,7 +701,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
@@ -744,16 +745,11 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
adr(TMP1, &JITCodeHeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
#ifdef _WIN32
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) -
|
||||
offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(CodeData.BlockEntry, Entry);
|
||||
CursorIncrement(GDBSize);
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) -
|
||||
offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
@@ -800,272 +796,9 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
#define REGISTER_OP(op, x) case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
// ALU ops
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
REGISTER_OP(CONSTANT, Constant);
|
||||
REGISTER_OP(ENTRYPOINTOFFSET, EntrypointOffset);
|
||||
REGISTER_OP(INLINECONSTANT, InlineConstant);
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(ADDNZCV, AddNZCV);
|
||||
REGISTER_OP(ADCNZCV, AdcNZCV);
|
||||
REGISTER_OP(SBBNZCV, SbbNZCV);
|
||||
REGISTER_OP(TESTNZ, TestNZ);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(SUBNZCV, SubNZCV);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
REGISTER_OP(UDIV, UDiv);
|
||||
REGISTER_OP(REM, Rem);
|
||||
REGISTER_OP(UREM, URem);
|
||||
REGISTER_OP(MULH, MulH);
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(ORLSHL, Orlshl);
|
||||
REGISTER_OP(ORLSHR, Orlshr);
|
||||
REGISTER_OP(ORNROR, Ornror);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
REGISTER_OP(LSHL, Lshl);
|
||||
REGISTER_OP(LSHR, Lshr);
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
REGISTER_OP(LUREM, LURem);
|
||||
REGISTER_OP(NOT, Not);
|
||||
REGISTER_OP(POPCOUNT, Popcount);
|
||||
REGISTER_OP(FINDLSB, FindLSB);
|
||||
REGISTER_OP(FINDMSB, FindMSB);
|
||||
REGISTER_OP(FINDTRAILINGZEROES, FindTrailingZeroes);
|
||||
REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes);
|
||||
REGISTER_OP(REV, Rev);
|
||||
REGISTER_OP(BFI, Bfi);
|
||||
REGISTER_OP(BFXIL, Bfxil);
|
||||
REGISTER_OP(BFE, Bfe);
|
||||
REGISTER_OP(SBFE, Sbfe);
|
||||
REGISTER_OP(SELECT, Select);
|
||||
REGISTER_OP(VEXTRACTTOGPR, VExtractToGPR);
|
||||
REGISTER_OP(FLOAT_TOGPR_ZS, Float_ToGPR_ZS);
|
||||
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
|
||||
REGISTER_OP(FCMP, FCmp);
|
||||
|
||||
// Atomic ops
|
||||
REGISTER_OP(CASPAIR, CASPair);
|
||||
REGISTER_OP(CAS, CAS);
|
||||
REGISTER_OP(ATOMICADD, AtomicAdd);
|
||||
REGISTER_OP(ATOMICSUB, AtomicSub);
|
||||
REGISTER_OP(ATOMICAND, AtomicAnd);
|
||||
REGISTER_OP(ATOMICOR, AtomicOr);
|
||||
REGISTER_OP(ATOMICXOR, AtomicXor);
|
||||
REGISTER_OP(ATOMICSWAP, AtomicSwap);
|
||||
REGISTER_OP(ATOMICFETCHADD, AtomicFetchAdd);
|
||||
REGISTER_OP(ATOMICFETCHSUB, AtomicFetchSub);
|
||||
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
|
||||
REGISTER_OP(ATOMICFETCHCLR, AtomicFetchCLR);
|
||||
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
|
||||
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
REGISTER_OP(TELEMETRYSETVALUE, TelemetrySetValue);
|
||||
|
||||
// Branch ops
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
REGISTER_OP(JUMP, Jump);
|
||||
REGISTER_OP(CONDJUMP, CondJump);
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(VDUPFROMGPR, VDupFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
|
||||
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
|
||||
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
|
||||
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
|
||||
|
||||
// Encryption ops
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
|
||||
// Flag ops
|
||||
REGISTER_OP(GETHOSTFLAG, GetHostFlag);
|
||||
|
||||
// Memory ops
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP_RT(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP_RT(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
REGISTER_OP(FILLREGISTER, FillRegister);
|
||||
REGISTER_OP(LOADFLAG, LoadFlag);
|
||||
REGISTER_OP(STOREFLAG, StoreFlag);
|
||||
REGISTER_OP(LOADMEM, LoadMem);
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP_RT(LOADMEMTSO, LoadMemTSO);
|
||||
REGISTER_OP_RT(STOREMEMTSO, StoreMemTSO);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
REGISTER_OP(VLOADVECTORELEMENT, VLoadVectorElement);
|
||||
REGISTER_OP(VSTOREVECTORELEMENT, VStoreVectorElement);
|
||||
REGISTER_OP(VBROADCASTFROMMEM, VBroadcastFromMem);
|
||||
REGISTER_OP(PUSH, Push);
|
||||
REGISTER_OP(MEMSET, MemSet);
|
||||
REGISTER_OP(MEMCPY, MemCpy);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINECLEAN, CacheLineClean);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
// Misc ops
|
||||
REGISTER_OP(DUMMY, NoOp);
|
||||
REGISTER_OP(IRHEADER, NoOp);
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PRINT, Print);
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
|
||||
// Vector ops
|
||||
REGISTER_OP(VECTORZERO, VectorZero);
|
||||
REGISTER_OP(VECTORIMM, VectorImm);
|
||||
REGISTER_OP(LOADNAMEDVECTORCONSTANT, LoadNamedVectorConstant);
|
||||
REGISTER_OP(LOADNAMEDVECTORINDEXEDCONSTANT, LoadNamedVectorIndexedConstant);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
REGISTER_OP(VAND, VAnd);
|
||||
REGISTER_OP(VBIC, VBic);
|
||||
REGISTER_OP(VOR, VOr);
|
||||
REGISTER_OP(VXOR, VXor);
|
||||
REGISTER_OP(VADD, VAdd);
|
||||
REGISTER_OP(VSUB, VSub);
|
||||
REGISTER_OP(VUQADD, VUQAdd);
|
||||
REGISTER_OP(VUQSUB, VUQSub);
|
||||
REGISTER_OP(VSQADD, VSQAdd);
|
||||
REGISTER_OP(VSQSUB, VSQSub);
|
||||
REGISTER_OP(VADDP, VAddP);
|
||||
REGISTER_OP(VADDV, VAddV);
|
||||
REGISTER_OP(VUMINV, VUMinV);
|
||||
REGISTER_OP(VURAVG, VURAvg);
|
||||
REGISTER_OP(VABS, VAbs);
|
||||
REGISTER_OP(VFABS, VFAbs);
|
||||
REGISTER_OP(VPOPCOUNT, VPopcount);
|
||||
REGISTER_OP(VFADD, VFAdd);
|
||||
REGISTER_OP(VFADDP, VFAddP);
|
||||
REGISTER_OP(VFSUB, VFSub);
|
||||
REGISTER_OP(VFMUL, VFMul);
|
||||
REGISTER_OP(VFDIV, VFDiv);
|
||||
REGISTER_OP(VFMIN, VFMin);
|
||||
REGISTER_OP(VFMAX, VFMax);
|
||||
REGISTER_OP(VFRECP, VFRecp);
|
||||
REGISTER_OP(VFSQRT, VFSqrt);
|
||||
REGISTER_OP(VFRSQRT, VFRSqrt);
|
||||
REGISTER_OP(VNEG, VNeg);
|
||||
REGISTER_OP(VFNEG, VFNeg);
|
||||
REGISTER_OP(VNOT, VNot);
|
||||
REGISTER_OP(VUMIN, VUMin);
|
||||
REGISTER_OP(VSMIN, VSMin);
|
||||
REGISTER_OP(VUMAX, VUMax);
|
||||
REGISTER_OP(VSMAX, VSMax);
|
||||
REGISTER_OP(VZIP, VZip);
|
||||
REGISTER_OP(VZIP2, VZip2);
|
||||
REGISTER_OP(VUNZIP, VUnZip);
|
||||
REGISTER_OP(VUNZIP2, VUnZip2);
|
||||
REGISTER_OP(VTRN, VTrn);
|
||||
REGISTER_OP(VTRN2, VTrn2);
|
||||
REGISTER_OP(VBSL, VBSL);
|
||||
REGISTER_OP(VCMPEQ, VCMPEQ);
|
||||
REGISTER_OP(VCMPEQZ, VCMPEQZ);
|
||||
REGISTER_OP(VCMPGT, VCMPGT);
|
||||
REGISTER_OP(VCMPGTZ, VCMPGTZ);
|
||||
REGISTER_OP(VCMPLTZ, VCMPLTZ);
|
||||
REGISTER_OP(VFCMPEQ, VFCMPEQ);
|
||||
REGISTER_OP(VFCMPNEQ, VFCMPNEQ);
|
||||
REGISTER_OP(VFCMPLT, VFCMPLT);
|
||||
REGISTER_OP(VFCMPGT, VFCMPGT);
|
||||
REGISTER_OP(VFCMPLE, VFCMPLE);
|
||||
REGISTER_OP(VFCMPORD, VFCMPORD);
|
||||
REGISTER_OP(VFCMPUNO, VFCMPUNO);
|
||||
REGISTER_OP(VUSHL, VUShl);
|
||||
REGISTER_OP(VUSHR, VUShr);
|
||||
REGISTER_OP(VSSHR, VSShr);
|
||||
REGISTER_OP(VUSHLS, VUShlS);
|
||||
REGISTER_OP(VUSHRS, VUShrS);
|
||||
REGISTER_OP(VUSHRSWIDE, VUShrSWide);
|
||||
REGISTER_OP(VSSHRSWIDE, VSShrSWide);
|
||||
REGISTER_OP(VUSHLSWIDE, VUShlSWide);
|
||||
REGISTER_OP(VSSHRS, VSShrS);
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VUSHRI, VUShrI);
|
||||
REGISTER_OP(VSSHRI, VSShrI);
|
||||
REGISTER_OP(VSHLI, VShlI);
|
||||
REGISTER_OP(VUSHRNI, VUShrNI);
|
||||
REGISTER_OP(VUSHRNI2, VUShrNI2);
|
||||
REGISTER_OP(VSXTL, VSXTL);
|
||||
REGISTER_OP(VSXTL2, VSXTL2);
|
||||
REGISTER_OP(VUXTL, VUXTL);
|
||||
REGISTER_OP(VUXTL2, VUXTL2);
|
||||
REGISTER_OP(VSQXTN, VSQXTN);
|
||||
REGISTER_OP(VSQXTN2, VSQXTN2);
|
||||
REGISTER_OP(VSQXTNPAIR, VSQXTNPair);
|
||||
REGISTER_OP(VSQXTUN, VSQXTUN);
|
||||
REGISTER_OP(VSQXTUN2, VSQXTUN2);
|
||||
REGISTER_OP(VSQXTUNPAIR, VSQXTUNPair);
|
||||
REGISTER_OP(VSRSHR, VSRSHR);
|
||||
REGISTER_OP(VSQSHL, VSQSHL);
|
||||
REGISTER_OP(VUMUL, VMul);
|
||||
REGISTER_OP(VSMUL, VMul);
|
||||
REGISTER_OP(VUMULL, VUMull);
|
||||
REGISTER_OP(VSMULL, VSMull);
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUMULH, VUMulH);
|
||||
REGISTER_OP(VSMULH, VSMulH);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VTBL2, VTBL2);
|
||||
REGISTER_OP(VREV32, VRev32);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VFCADD, VFCADD);
|
||||
#define IROP_DISPATCH_DISPATCH
|
||||
#include <FEXCore/IR/IRDefines_Dispatch.inc>
|
||||
#undef REGISTER_OP
|
||||
|
||||
default:
|
||||
|
||||
@@ -26,6 +26,7 @@ $end_info$
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
@@ -43,7 +44,7 @@ public:
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -58,6 +59,8 @@ private:
|
||||
|
||||
const bool HostSupportsSVE128{};
|
||||
const bool HostSupportsSVE256{};
|
||||
const bool HostSupportsRPRES{};
|
||||
const bool HostSupportsAFP{};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
@@ -113,6 +116,15 @@ private:
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
// Converts IR-base shift type to ARMEmitter shift type.
|
||||
// Will be a no-op, only a type conversion since the two definitions match.
|
||||
[[nodiscard]] ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
|
||||
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
|
||||
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
|
||||
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
|
||||
ARMEmitter::ShiftType::ROR;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
@@ -208,6 +220,11 @@ private:
|
||||
uint32_t SpillSlots{};
|
||||
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
|
||||
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store register style.
|
||||
OpType RT_LoadRegister;
|
||||
@@ -218,278 +235,21 @@ private:
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
// Dynamic Dispatcher supporting operations
|
||||
DEF_OP(LoadRegisterSRA);
|
||||
DEF_OP(StoreRegisterSRA);
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
|
||||
///< No-op Handler
|
||||
DEF_OP(NoOp);
|
||||
|
||||
///< ALU Ops
|
||||
DEF_OP(TruncElementPair);
|
||||
DEF_OP(Constant);
|
||||
DEF_OP(EntrypointOffset);
|
||||
DEF_OP(InlineConstant);
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(AddNZCV);
|
||||
DEF_OP(AdcNZCV);
|
||||
DEF_OP(SbbNZCV);
|
||||
DEF_OP(TestNZ);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(SubNZCV);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
DEF_OP(UDiv);
|
||||
DEF_OP(Rem);
|
||||
DEF_OP(URem);
|
||||
DEF_OP(MulH);
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(Orlshl);
|
||||
DEF_OP(Orlshr);
|
||||
DEF_OP(Ornror);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
DEF_OP(Lshl);
|
||||
DEF_OP(Lshr);
|
||||
DEF_OP(Ashr);
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
DEF_OP(LURem);
|
||||
DEF_OP(Zext);
|
||||
DEF_OP(Not);
|
||||
DEF_OP(Popcount);
|
||||
DEF_OP(FindLSB);
|
||||
DEF_OP(FindMSB);
|
||||
DEF_OP(FindTrailingZeroes);
|
||||
DEF_OP(CountLeadingZeroes);
|
||||
DEF_OP(Rev);
|
||||
DEF_OP(Bfi);
|
||||
DEF_OP(Bfxil);
|
||||
DEF_OP(Bfe);
|
||||
DEF_OP(Sbfe);
|
||||
DEF_OP(Select);
|
||||
DEF_OP(VExtractToGPR);
|
||||
DEF_OP(Float_ToGPR_ZU);
|
||||
DEF_OP(Float_ToGPR_ZS);
|
||||
DEF_OP(Float_ToGPR_S);
|
||||
DEF_OP(FCmp);
|
||||
|
||||
///< Atomic ops
|
||||
DEF_OP(CASPair);
|
||||
DEF_OP(CAS);
|
||||
DEF_OP(AtomicAdd);
|
||||
DEF_OP(AtomicSub);
|
||||
DEF_OP(AtomicAnd);
|
||||
DEF_OP(AtomicOr);
|
||||
DEF_OP(AtomicXor);
|
||||
DEF_OP(AtomicSwap);
|
||||
DEF_OP(AtomicFetchAdd);
|
||||
DEF_OP(AtomicFetchSub);
|
||||
DEF_OP(AtomicFetchAnd);
|
||||
DEF_OP(AtomicFetchCLR);
|
||||
DEF_OP(AtomicFetchOr);
|
||||
DEF_OP(AtomicFetchXor);
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
DEF_OP(TelemetrySetValue);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
DEF_OP(Jump);
|
||||
DEF_OP(CondJump);
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(VDupFromGPR);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
DEF_OP(Vector_FToZS);
|
||||
DEF_OP(Vector_FToS);
|
||||
DEF_OP(Vector_FToF);
|
||||
DEF_OP(Vector_FToI);
|
||||
|
||||
///< Flag ops
|
||||
DEF_OP(GetHostFlag);
|
||||
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadRegisterSRA);
|
||||
DEF_OP(StoreRegisterSRA);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
DEF_OP(FillRegister);
|
||||
DEF_OP(LoadFlag);
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(LoadMemTSO);
|
||||
DEF_OP(StoreMemTSO);
|
||||
DEF_OP(VLoadVectorMasked);
|
||||
DEF_OP(VStoreVectorMasked);
|
||||
DEF_OP(VLoadVectorElement);
|
||||
DEF_OP(VStoreVectorElement);
|
||||
DEF_OP(VBroadcastFromMem);
|
||||
DEF_OP(Push);
|
||||
DEF_OP(MemSet);
|
||||
DEF_OP(MemCpy);
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineClean);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(LoadNamedVectorConstant);
|
||||
DEF_OP(LoadNamedVectorIndexedConstant);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
DEF_OP(VOr);
|
||||
DEF_OP(VXor);
|
||||
DEF_OP(VAdd);
|
||||
DEF_OP(VSub);
|
||||
DEF_OP(VUQAdd);
|
||||
DEF_OP(VUQSub);
|
||||
DEF_OP(VSQAdd);
|
||||
DEF_OP(VSQSub);
|
||||
DEF_OP(VAddP);
|
||||
DEF_OP(VAddV);
|
||||
DEF_OP(VUMinV);
|
||||
DEF_OP(VURAvg);
|
||||
DEF_OP(VAbs);
|
||||
DEF_OP(VFAbs);
|
||||
DEF_OP(VPopcount);
|
||||
DEF_OP(VFAdd);
|
||||
DEF_OP(VFAddP);
|
||||
DEF_OP(VFSub);
|
||||
DEF_OP(VFMul);
|
||||
DEF_OP(VFDiv);
|
||||
DEF_OP(VFMin);
|
||||
DEF_OP(VFMax);
|
||||
DEF_OP(VFRecp);
|
||||
DEF_OP(VFSqrt);
|
||||
DEF_OP(VFRSqrt);
|
||||
DEF_OP(VNeg);
|
||||
DEF_OP(VFNeg);
|
||||
DEF_OP(VNot);
|
||||
DEF_OP(VUMin);
|
||||
DEF_OP(VSMin);
|
||||
DEF_OP(VUMax);
|
||||
DEF_OP(VSMax);
|
||||
DEF_OP(VZip);
|
||||
DEF_OP(VZip2);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VUnZip2);
|
||||
DEF_OP(VTrn);
|
||||
DEF_OP(VTrn2);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
DEF_OP(VCMPGT);
|
||||
DEF_OP(VCMPGTZ);
|
||||
DEF_OP(VCMPLTZ);
|
||||
DEF_OP(VFCMPEQ);
|
||||
DEF_OP(VFCMPNEQ);
|
||||
DEF_OP(VFCMPLT);
|
||||
DEF_OP(VFCMPGT);
|
||||
DEF_OP(VFCMPLE);
|
||||
DEF_OP(VFCMPORD);
|
||||
DEF_OP(VFCMPUNO);
|
||||
DEF_OP(VUShl);
|
||||
DEF_OP(VUShr);
|
||||
DEF_OP(VSShr);
|
||||
DEF_OP(VUShlS);
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VUShrSWide);
|
||||
DEF_OP(VSShrSWide);
|
||||
DEF_OP(VUShlSWide);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
DEF_OP(VUXTL2);
|
||||
DEF_OP(VSQXTN);
|
||||
DEF_OP(VSQXTN2);
|
||||
DEF_OP(VSQXTNPair);
|
||||
DEF_OP(VSQXTUN);
|
||||
DEF_OP(VSQXTUN2);
|
||||
DEF_OP(VSQXTUNPair);
|
||||
DEF_OP(VSRSHR);
|
||||
DEF_OP(VSQSHL);
|
||||
DEF_OP(VMul);
|
||||
DEF_OP(VUMull);
|
||||
DEF_OP(VSMull);
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUMulH);
|
||||
DEF_OP(VSMulH);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VTBL2);
|
||||
DEF_OP(VRev32);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VFCADD);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
DEF_OP(AESEnc);
|
||||
DEF_OP(AESEncLast);
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#define IROP_DISPATCH_DEFS
|
||||
#include <FEXCore/IR/IRDefines_Dispatch.inc>
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
@@ -295,7 +296,11 @@ DEF_OP(LoadRegisterSRA) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
const auto regId =
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ? (StaticRegisters.size() - 2) :
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ? (StaticRegisters.size() - 1) :
|
||||
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
|
||||
@@ -473,10 +478,14 @@ DEF_OP(StoreRegisterSRA) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
const auto regId =
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ? (StaticRegisters.size() - 2) :
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ? (StaticRegisters.size() - 1) :
|
||||
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = StaticRegisters[regId];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
@@ -1031,23 +1040,37 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadNZCV) {
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(StoreNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_StoreNZCV>();
|
||||
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
ldr(Dst.W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
LOGMAN_THROW_A_FMT(Op->Flag != X86State::RFLAG_PF_RAW_LOC &&
|
||||
Op->Flag != X86State::RFLAG_AF_RAW_LOC,
|
||||
"PF/AF must be accessed as registers");
|
||||
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
str(GetReg(Op->Value.ID()).W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
LOGMAN_THROW_A_FMT(Op->Flag != X86State::RFLAG_PF_RAW_LOC &&
|
||||
Op->Flag != X86State::RFLAG_AF_RAW_LOC,
|
||||
"PF/AF must be accessed as registers");
|
||||
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize,
|
||||
@@ -1520,13 +1543,13 @@ DEF_OP(VBroadcastFromMem) {
|
||||
ElementSize == 4 || ElementSize == 8 ||
|
||||
ElementSize == 16, "Invalid element size");
|
||||
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use SVE 256-bit broadcast");
|
||||
}
|
||||
if (Is256Bit && !HostSupportsSVE256) {
|
||||
LOGMAN_MSG_A_FMT("{}: 256-bit vectors must support SVE256", __func__);
|
||||
return;
|
||||
}
|
||||
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B.Zeroing()
|
||||
: PRED_TMP_16B.Zeroing();
|
||||
if (Is256Bit && HostSupportsSVE256) {
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
@@ -1840,9 +1863,15 @@ DEF_OP(MemSet) {
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Direction = GetReg(Op->Direction.ID());
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
|
||||
// If Direction == 0 then:
|
||||
// MemReg is incremented (by size)
|
||||
// else:
|
||||
@@ -1862,8 +1891,10 @@ DEF_OP(MemSet) {
|
||||
add(TMP2, Prefix.X(), MemReg.X());
|
||||
}
|
||||
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, Direction, &BackwardImpl);
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
|
||||
switch (OpSize) {
|
||||
@@ -1917,8 +1948,7 @@ DEF_OP(MemSet) {
|
||||
}
|
||||
};
|
||||
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
auto EmitMemset = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
|
||||
@@ -1978,15 +2008,26 @@ DEF_OP(MemSet) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
if (DirectionIsInline) {
|
||||
// If the direction constant is set then the direction is negative.
|
||||
EmitMemset(DirectionConstant ? -1 : 1);
|
||||
}
|
||||
else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
EmitMemset(Direction);
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
@@ -2001,7 +2042,12 @@ DEF_OP(MemCpy) {
|
||||
const auto MemRegSrc = GetReg(Op->AddrSrc.ID());
|
||||
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Direction = GetReg(Op->Direction.ID());
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
|
||||
auto Dst = GetRegPair(Node);
|
||||
// If Direction == 0 then:
|
||||
@@ -2038,8 +2084,10 @@ DEF_OP(MemCpy) {
|
||||
// TMP3 = Src
|
||||
// TMP4 = load+store temp value
|
||||
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, Direction, &BackwardImpl);
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
cbnz(ARMEmitter::Size::i64Bit, DirectionReg, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
|
||||
switch (OpSize) {
|
||||
@@ -2161,8 +2209,7 @@ DEF_OP(MemCpy) {
|
||||
}
|
||||
};
|
||||
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
auto EmitMemcpy = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
|
||||
@@ -2235,15 +2282,24 @@ DEF_OP(MemCpy) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
if (DirectionIsInline) {
|
||||
// If the direction constant is set then the direction is negative.
|
||||
EmitMemcpy(DirectionConstant ? -1 : 1);
|
||||
}
|
||||
else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
for (int32_t Direction : { 1, -1 }) {
|
||||
EmitMemcpy(Direction);
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
|
||||
@@ -143,7 +143,17 @@ DEF_OP(Print) {
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
|
||||
}
|
||||
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -27,6 +27,50 @@ namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class PassManager;
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
DEFAULT,
|
||||
// TSO access behaviour
|
||||
TSO,
|
||||
// Non-TSO access behaviour
|
||||
NONTSO,
|
||||
// Non-temporal streaming
|
||||
STREAM,
|
||||
};
|
||||
|
||||
enum class BTAction {
|
||||
BTNone,
|
||||
BTClear,
|
||||
BTSet,
|
||||
BTComplement,
|
||||
};
|
||||
|
||||
struct LoadSourceOptions {
|
||||
// Alignment of the load in bytes. -1 signifies unaligned
|
||||
int8_t Align = -1;
|
||||
|
||||
// Whether or not to load the data if a memory access occurs.
|
||||
// If set to false, then the address that would have been loaded from
|
||||
// will be returned instead.
|
||||
//
|
||||
// Note: If returning the address, make sure to apply the segment offset
|
||||
// after with AppendSegmentOffset().
|
||||
//
|
||||
bool LoadData = true;
|
||||
|
||||
// Use to force a load even if the underlying type isn't loadable.
|
||||
bool ForceLoad = false;
|
||||
|
||||
// Specifies the access type of the load.
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT;
|
||||
|
||||
// Whether or not a zero extend should clear the upper bits
|
||||
// in the register (e.g. an 8-bit load would clear the upper 24 bits
|
||||
// or 56 bits depending on the operating mode).
|
||||
// If true, no zero-extension occurs.
|
||||
bool AllowUpperGarbage = false;
|
||||
};
|
||||
|
||||
class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
@@ -52,7 +96,6 @@ public:
|
||||
TYPE_RORI,
|
||||
TYPE_ROL,
|
||||
TYPE_ROLI,
|
||||
TYPE_FCMP,
|
||||
TYPE_BEXTR,
|
||||
TYPE_BLSI,
|
||||
TYPE_BLSMSK,
|
||||
@@ -61,7 +104,6 @@ public:
|
||||
TYPE_BZHI,
|
||||
TYPE_TZCNT,
|
||||
TYPE_LZCNT,
|
||||
TYPE_BITSELECT,
|
||||
TYPE_RDRAND,
|
||||
};
|
||||
|
||||
@@ -85,11 +127,10 @@ public:
|
||||
}
|
||||
|
||||
void StartNewBlock() {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
|
||||
// If we loaded flags but didn't change them, invalidate the cached copy and move on.
|
||||
// Changes get stored out by CalculateDeferredFlags.
|
||||
CachedNZCV = nullptr;
|
||||
PossiblySetNZCVBits = ~0U;
|
||||
|
||||
// New block needs to reset segment telemetry.
|
||||
SegmentsNeedReadCheck = ~0U;
|
||||
@@ -98,6 +139,35 @@ public:
|
||||
ClearCachedNamedConstants();
|
||||
}
|
||||
|
||||
IRPair<IROp_Jump> Jump() {
|
||||
CalculateDeferredFlags();
|
||||
return _Jump();
|
||||
}
|
||||
IRPair<IROp_Jump> Jump(OrderedNode *_TargetBlock) {
|
||||
CalculateDeferredFlags();
|
||||
return _Jump(_TargetBlock);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(OrderedNode *_Cmp1, OrderedNode *_Cmp2, OrderedNode *_TrueBlock, OrderedNode *_FalseBlock, CondClassType _Cond = {COND_NEQ}, uint8_t _CompareSize = 0) {
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(_Cmp1, _Cmp2, _TrueBlock, _FalseBlock, _Cond, _CompareSize);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(OrderedNode *ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(ssa0, cond);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(OrderedNode *ssa0, OrderedNode *ssa1, OrderedNode *ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(ssa0, ssa1, ssa2, cond);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// The jump will ignore the sources, so it doesn't matter what we put here.
|
||||
// Put an inline constant so RA+codegen will ignore altogether.
|
||||
auto Placeholder = _InlineConstant(0);
|
||||
return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true);
|
||||
}
|
||||
|
||||
bool FinishOp(uint64_t NextRIP, bool LastOp) {
|
||||
// If we are switching to a new block and this current block has yet to set a RIP
|
||||
// Then we need to insert an unconditional jump from the current block to the one we are going to
|
||||
@@ -123,7 +193,7 @@ public:
|
||||
_ExitFunction(RelocatedNextRIP);
|
||||
}
|
||||
else if (it != JumpTargets.end()) {
|
||||
_Jump(it->second.BlockEntry);
|
||||
Jump(it->second.BlockEntry);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -185,7 +255,8 @@ public:
|
||||
template<uint32_t SrcIndex>
|
||||
void MOVGPROp(OpcodeArgs);
|
||||
void MOVGPRNTOp(OpcodeArgs);
|
||||
void MOVVectorOp(OpcodeArgs);
|
||||
void MOVVectorAlignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp>
|
||||
void ALUOp(OpcodeArgs);
|
||||
@@ -265,14 +336,10 @@ public:
|
||||
void RCLOp1Bit(OpcodeArgs);
|
||||
void RCLOp(OpcodeArgs);
|
||||
void RCLSmallerOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
|
||||
template<uint32_t SrcIndex, enum BTAction Action>
|
||||
void BTOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void BTROp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void BTSOp(OpcodeArgs);
|
||||
template<uint32_t SrcIndex>
|
||||
void BTCOp(OpcodeArgs);
|
||||
|
||||
void IMUL1SrcOp(OpcodeArgs);
|
||||
void IMUL2SrcOp(OpcodeArgs);
|
||||
void IMULOp(OpcodeArgs);
|
||||
@@ -322,8 +389,6 @@ public:
|
||||
void SGDTOp(OpcodeArgs);
|
||||
|
||||
// SSE
|
||||
void MOVAPS_MOVAPDOp(OpcodeArgs);
|
||||
void MOVUPS_MOVUPDOp(OpcodeArgs);
|
||||
void MOVLPOp(OpcodeArgs);
|
||||
void MOVHPDOp(OpcodeArgs);
|
||||
void MOVSDOp(OpcodeArgs);
|
||||
@@ -333,13 +398,12 @@ public:
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorALUROp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorScalarALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize, bool Scalar>
|
||||
void VectorUnaryOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorUnaryDuplicateOp(OpcodeArgs);
|
||||
|
||||
void MOVQOp(OpcodeArgs);
|
||||
void MOVQMMXOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void MOVMSKOp(OpcodeArgs);
|
||||
void MOVMSKOpOne(OpcodeArgs);
|
||||
@@ -379,7 +443,6 @@ public:
|
||||
void Vector_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Widen>
|
||||
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
@@ -387,7 +450,7 @@ public:
|
||||
void MOVBetweenGPR_FPR(OpcodeArgs);
|
||||
void TZCNT(OpcodeArgs);
|
||||
void LZCNT(OpcodeArgs);
|
||||
template<size_t ElementSize, bool Scalar>
|
||||
template<size_t ElementSize>
|
||||
void VFCMPOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void SHUFOp(OpcodeArgs);
|
||||
@@ -424,11 +487,9 @@ public:
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorALUOp(OpcodeArgs);
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarALUOp(OpcodeArgs);
|
||||
template <IROps IROp, size_t ElementSize, bool Scalar>
|
||||
void AVXVectorUnaryOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
template <size_t ElementSize>
|
||||
void AVXVectorRound(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize, size_t SrcElementSize>
|
||||
@@ -440,10 +501,44 @@ public:
|
||||
template <size_t SrcElementSize, bool Widen>
|
||||
void AVXVector_CVT_Int_To_Float(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorScalarInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarInsertALUOp(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorScalarUnaryInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarUnaryInsertALUOp(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void AVXVector_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
void InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize>
|
||||
void InsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
template <size_t DstElementSize>
|
||||
void AVXInsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void InsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void InsertScalarRound(OpcodeArgs);
|
||||
template <size_t ElementSize>
|
||||
void AVXInsertScalarRound(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void InsertScalarFCMPOp(OpcodeArgs);
|
||||
template <size_t ElementSize>
|
||||
void AVXInsertScalarFCMPOp(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize>
|
||||
void AVXCVTGPR_To_FPR(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
template <size_t ElementSize>
|
||||
void AVXVFCMPOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
@@ -489,6 +584,9 @@ public:
|
||||
void VMOVSDOp(OpcodeArgs);
|
||||
void VMOVSSOp(OpcodeArgs);
|
||||
|
||||
void VMOVAPS_VMOVAPDOp(OpcodeArgs);
|
||||
void VMOVUPS_VMOVUPDOp(OpcodeArgs);
|
||||
|
||||
void VMPSADBWOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
@@ -787,7 +885,7 @@ public:
|
||||
|
||||
template<size_t ElementSize, size_t DstElementSize, bool Signed>
|
||||
void ExtendVectorElements(OpcodeArgs);
|
||||
template<size_t ElementSize, bool Scalar>
|
||||
template<size_t ElementSize>
|
||||
void VectorRound(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
@@ -817,22 +915,45 @@ public:
|
||||
|
||||
static inline constexpr unsigned IndexNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_OF_LOC: return 28;
|
||||
case FEXCore::X86State::RFLAG_CF_LOC: return 29;
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC: return 30;
|
||||
case FEXCore::X86State::RFLAG_SF_LOC: return 31;
|
||||
case FEXCore::X86State::RFLAG_OF_RAW_LOC: return 28;
|
||||
case FEXCore::X86State::RFLAG_CF_RAW_LOC: return 29;
|
||||
case FEXCore::X86State::RFLAG_ZF_RAW_LOC: return 30;
|
||||
case FEXCore::X86State::RFLAG_SF_RAW_LOC: return 31;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FCMP, // flags were set by a ucomis* / comis*
|
||||
};
|
||||
protected:
|
||||
void SaveNZCV(IROps Op = OP_DUMMY) override {
|
||||
/* Some opcodes are conservatively marked as clobbering flags, but in fact
|
||||
* do not clobber flags in certain conditions. Check for that here as an
|
||||
* optimization.
|
||||
*/
|
||||
switch (Op) {
|
||||
case OP_VFMINSCALARINSERT:
|
||||
case OP_VFMAXSCALARINSERT:
|
||||
/* On AFP platforms, becomes fmin/fmax and preserves NZCV. Otherwise
|
||||
* becomes fcmp and clobbers.
|
||||
*/
|
||||
if (CTX->HostFeatures.SupportsAFP)
|
||||
return;
|
||||
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
// Invariant: When executing instructions that clobber NZCV, the flags must
|
||||
// be resident in a GPR, which is equivalent to CachedNZCV != nullptr. Get
|
||||
// the NZCV which fills the cache if necessary.
|
||||
if (CachedNZCV == nullptr)
|
||||
GetNZCV();
|
||||
|
||||
// Assume we'll need a reload.
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
private:
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
@@ -840,18 +961,11 @@ private:
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX{};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
OrderedNode* flagsOpSrc{};
|
||||
OrderedNode* flagsOpDestSigned{};
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
constexpr static unsigned FullNZCVMask =
|
||||
(1U << FEXCore::X86State::RFLAG_CF_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_ZF_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_SF_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_OF_LOC);
|
||||
(1U << FEXCore::X86State::RFLAG_CF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_SF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
static bool ContainsNZCV(unsigned BitMask) {
|
||||
return (BitMask & FullNZCVMask) != 0;
|
||||
@@ -859,10 +973,10 @@ private:
|
||||
|
||||
static bool IsNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_CF_LOC:
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC:
|
||||
case FEXCore::X86State::RFLAG_SF_LOC:
|
||||
case FEXCore::X86State::RFLAG_OF_LOC:
|
||||
case FEXCore::X86State::RFLAG_CF_RAW_LOC:
|
||||
case FEXCore::X86State::RFLAG_ZF_RAW_LOC:
|
||||
case FEXCore::X86State::RFLAG_SF_RAW_LOC:
|
||||
case FEXCore::X86State::RFLAG_OF_RAW_LOC:
|
||||
return true;
|
||||
|
||||
default:
|
||||
@@ -889,8 +1003,7 @@ private:
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
void AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
|
||||
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void AVXVectorVariableBlend(OpcodeArgs);
|
||||
@@ -907,6 +1020,10 @@ private:
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, size_t ElementSize);
|
||||
|
||||
OrderedNode* VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
OrderedNode* ExtendVectorElementsImpl(OpcodeArgs, size_t ElementSize,
|
||||
size_t DstElementSize, bool Signed);
|
||||
|
||||
@@ -930,7 +1047,8 @@ private:
|
||||
|
||||
OrderedNode* PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
const X86Tables::DecodedOperand& Imm,
|
||||
bool IsAVX);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
|
||||
|
||||
@@ -997,25 +1115,69 @@ private:
|
||||
void MOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
void VMOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool Scalar,
|
||||
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src1, OrderedNode *Src2, uint8_t CompType);
|
||||
|
||||
void VTESTOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
|
||||
void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
|
||||
// x86 ALU scalar operations operate in three different ways
|
||||
// - AVX512: Writemask shenanigans that we don't care about.
|
||||
// - AVX/VEX: Two source
|
||||
// - Example 32bit VADDSS Dest, Src1, Src2
|
||||
// - Dest[31:0] = Src1[31:0] + Src2[31:0]
|
||||
// - Dest[127:32] = Src1[127:32]
|
||||
// - SSE: Scalar operation inserts in to the low bits, upper bits completely unaffected.
|
||||
// - Example 32bit ADDSS Dest, Src
|
||||
// - Dest[31:0] = Dest[31:0] + Src[31:0]
|
||||
// - Dest[{256,128}:32] = (Unmodified)
|
||||
OrderedNode* VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertCVTGPR_To_FPRImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t DstElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
OrderedNode* InsertScalarRoundImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
uint64_t Mode, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertScalarFCMPOpImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
uint8_t CompType, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* VectorRoundImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, uint64_t Mode, bool IsScalar);
|
||||
OrderedNode *Src, uint64_t Mode);
|
||||
|
||||
OrderedNode* Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
void Vector_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize);
|
||||
void Vector_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize, bool IsAVX);
|
||||
|
||||
OrderedNode* Vector_CVT_Float_To_IntImpl(OpcodeArgs, size_t SrcElementSize, bool Narrow, bool HostRoundingMode);
|
||||
|
||||
@@ -1045,27 +1207,22 @@ private:
|
||||
|
||||
void UpdatePrefixFromSegment(OrderedNode *Segment, uint32_t SegmentReg);
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
ACCESS_DEFAULT,
|
||||
// TSO access behaviour
|
||||
ACCESS_TSO,
|
||||
// Non-TSO access behaviour
|
||||
ACCESS_NONTSO,
|
||||
// Non-temporal streaming
|
||||
ACCESS_STREAM,
|
||||
};
|
||||
OrderedNode *LoadGPRRegister(uint32_t GPR, int8_t Size = -1, uint8_t Offset = 0, bool AllowUpperGarbage = false);
|
||||
OrderedNode *LoadXMMRegister(uint32_t XMM);
|
||||
void StoreGPRRegister(uint32_t GPR, OrderedNode *const Src, int8_t Size = -1, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, OrderedNode *const Src);
|
||||
|
||||
OrderedNode *GetRelocatedPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT, bool AllowUpperGarbage = false);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT, bool AllowUpperGarbage = false);
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
|
||||
OrderedNode *LoadSource(RegisterClassType Class, X86Tables::DecodedOp const& Op, X86Tables::DecodedOperand const& Operand, uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
OrderedNode *LoadSource_WithOpSize(RegisterClassType Class, X86Tables::DecodedOp const& Op, X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
|
||||
constexpr OpSize GetGuestVectorLength() const {
|
||||
return CTX->HostFeatures.SupportsAVX ? OpSize::i256Bit : OpSize::i128Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]] static uint32_t GPROffset(X86State::X86Reg reg) {
|
||||
LOGMAN_THROW_AA_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
@@ -1089,24 +1246,24 @@ private:
|
||||
|
||||
static inline constexpr unsigned NZCVIndexMask(unsigned BitMask) {
|
||||
unsigned NZCVMask{};
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_OF_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_OF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_CF_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_ZF_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_SF_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC);
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_SF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
}
|
||||
return NZCVMask;
|
||||
}
|
||||
|
||||
OrderedNode *GetNZCV() {
|
||||
if (!CachedNZCV) {
|
||||
CachedNZCV = _LoadFlag(FEXCore::X86State::RFLAG_NZCV_LOC);
|
||||
CachedNZCV = _LoadNZCV();
|
||||
|
||||
// We don't know what's set
|
||||
PossiblySetNZCVBits = ~0;
|
||||
@@ -1132,43 +1289,91 @@ private:
|
||||
|
||||
// Mask out the NZ bits, clearing CV. Even if the code sets CV after, this can end up faster
|
||||
// moves by allowing orlshl to be used instead of bfi.
|
||||
PossiblySetNZCVBits = (1u << IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_LOC));
|
||||
PossiblySetNZCVBits = (1u << IndexNZCV(FEXCore::X86State::RFLAG_SF_RAW_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
SetNZCV(_And(OpSize::i32Bit, OldNZCV, _Constant(PossiblySetNZCVBits)));
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
CachedNZCV = _TestNZ(SrcSize, Res);
|
||||
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
|
||||
CachedNZCV = _LoadNZCV();
|
||||
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
|
||||
NZCVDirty = true;
|
||||
NZCVDirty = false;
|
||||
}
|
||||
|
||||
OrderedNode *InsertNZCV(OrderedNode *NZCV, unsigned BitOffset, OrderedNode *Value) {
|
||||
unsigned Bit = IndexNZCV(BitOffset);
|
||||
void InsertNZCV(unsigned BitOffset, OrderedNode *Value, signed FlagOffset, bool MustMask) {
|
||||
signed Bit = IndexNZCV(BitOffset);
|
||||
|
||||
// If NZCV is not dirty, we always want to use rmif, it's 1 instruction to
|
||||
// implement this. But if NZCV is dirty, it might still be cheaper to copy
|
||||
// the GPR flags to NZCV and rmif. This is a heuristic for cases where we
|
||||
// expect that 2 instruction sequence to be a win (versus something like
|
||||
// bfe+mov+bfi+mov which can happen with our RA..). It's not totally
|
||||
// conservative but it's pretty good in practice.
|
||||
bool PreferRmif = !NZCVDirty || FlagOffset || MustMask ||
|
||||
(PossiblySetNZCVBits & (1u << Bit));
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM && PreferRmif) {
|
||||
// Update NZCV
|
||||
if (NZCVDirty && CachedNZCV)
|
||||
_StoreNZCV(CachedNZCV);
|
||||
|
||||
CachedNZCV = nullptr;
|
||||
NZCVDirty = false;
|
||||
|
||||
// Insert as NZCV.
|
||||
signed RmifBit = Bit - 28;
|
||||
_RmifNZCV(Value, (64 + FlagOffset - RmifBit) % 64, 1u << RmifBit);
|
||||
CachedNZCV = nullptr;
|
||||
} else {
|
||||
// Insert as GPR
|
||||
if (FlagOffset || MustMask)
|
||||
Value = _Bfe(OpSize::i64Bit, 1, FlagOffset, Value);
|
||||
|
||||
if (PossiblySetNZCVBits == 0)
|
||||
SetNZCV(_Lshl(OpSize::i64Bit, Value, _Constant(Bit)));
|
||||
else if ((PossiblySetNZCVBits & (1u << Bit)) == 0)
|
||||
SetNZCV(_Orlshl(OpSize::i32Bit, GetNZCV(), Value, Bit));
|
||||
else
|
||||
SetNZCV(_Bfi(OpSize::i32Bit, 1, Bit, GetNZCV(), Value));
|
||||
}
|
||||
|
||||
uint32_t SetBits = PossiblySetNZCVBits;
|
||||
PossiblySetNZCVBits |= (1u << Bit);
|
||||
}
|
||||
|
||||
if (SetBits == 0)
|
||||
return _Lshl(OpSize::i64Bit, Value, _Constant(Bit));
|
||||
else if ((SetBits & (1u << Bit)) == 0)
|
||||
return _Orlshl(OpSize::i32Bit, NZCV, Value, Bit);
|
||||
else
|
||||
return _Bfi(OpSize::i32Bit, 1, Bit, NZCV, Value);
|
||||
void CarryInvert() {
|
||||
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM && !NZCVDirty) {
|
||||
// Invert as NZCV.
|
||||
_CarryInvert();
|
||||
CachedNZCV = nullptr;
|
||||
} else {
|
||||
// Invert as a GPR
|
||||
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), _Constant(1u << Bit)));
|
||||
}
|
||||
|
||||
PossiblySetNZCVBits |= 1u << Bit;
|
||||
}
|
||||
|
||||
template<unsigned BitOffset>
|
||||
void SetRFLAG(OrderedNode *Value) {
|
||||
SetRFLAG(Value, BitOffset);
|
||||
void SetRFLAG(OrderedNode *Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
SetRFLAG(Value, BitOffset, ValueOffset, MustMask);
|
||||
}
|
||||
|
||||
void SetRFLAG(OrderedNode *Value, unsigned BitOffset) {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
void SetRFLAG(OrderedNode *Value, unsigned BitOffset, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
InsertNZCV(BitOffset, Value, ValueOffset, MustMask);
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
_StoreRegister(Value, false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
_StoreRegister(Value, false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
} else {
|
||||
if (ValueOffset || MustMask)
|
||||
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
|
||||
|
||||
if (IsNZCV(BitOffset))
|
||||
SetNZCV(InsertNZCV(PossiblySetNZCVBits ? GetNZCV() : nullptr, BitOffset, Value));
|
||||
else
|
||||
_StoreFlag(Value, BitOffset);
|
||||
}
|
||||
}
|
||||
|
||||
void SetAF(unsigned Constant) {
|
||||
@@ -1176,22 +1381,139 @@ private:
|
||||
// bits. This allows us to defer the extract in the usual case. When it is
|
||||
// read, bit 4 is extracted. In order to write a constant value of AF, that
|
||||
// means we need to left-shift here to compensate.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(Constant << 4));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(Constant << 4));
|
||||
}
|
||||
|
||||
void ZeroMultipleFlags(uint32_t BitMask);
|
||||
|
||||
OrderedNode *GetRFLAG(unsigned BitOffset) {
|
||||
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_SF_RAW_LOC:
|
||||
return Invert ? CondClassType{COND_PL} : CondClassType{COND_MI};
|
||||
|
||||
case FEXCore::X86State::RFLAG_ZF_RAW_LOC:
|
||||
return Invert ? CondClassType{COND_NEQ} : CondClassType{COND_EQ};
|
||||
|
||||
case FEXCore::X86State::RFLAG_CF_RAW_LOC:
|
||||
return Invert ? CondClassType{COND_ULT} : CondClassType{COND_UGE};
|
||||
|
||||
case FEXCore::X86State::RFLAG_OF_RAW_LOC:
|
||||
return Invert ? CondClassType{COND_FNU} : CondClassType{COND_FU};
|
||||
|
||||
default:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *GetRFLAG(unsigned BitOffset, bool Invert = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
if (!CachedNZCV || (PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset))))
|
||||
return _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV());
|
||||
else
|
||||
return _Constant(0);
|
||||
if (!(PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset)))) {
|
||||
return _Constant(Invert ? 1 : 0);
|
||||
} else if (NZCVDirty) {
|
||||
auto Value = _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV());
|
||||
|
||||
if (Invert)
|
||||
return _Xor(OpSize::i32Bit, Value, _Constant(1));
|
||||
else
|
||||
return Value;
|
||||
} else {
|
||||
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert),
|
||||
_Constant(1), _Constant(0));
|
||||
}
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
} else {
|
||||
return _LoadFlag(BitOffset);
|
||||
}
|
||||
}
|
||||
|
||||
// Set SSE comparison flags based on the result set by Arm FCMP. This converts
|
||||
// NZCV from the Arm representation to an eXternal representation that's
|
||||
// totally not a euphemism for x86 or anything, nuh-uh.
|
||||
void ConvertNZCVToSSE() {
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
// We need to set PF according to the unordered flag. We'd rather do this
|
||||
// after axflag, since some impls fuse fcmp+axflag, so we want to do this
|
||||
// after. We can recover "unordered" after axflag as (Z && !C), but
|
||||
// there's no condition code for this so it would take 2 instructions
|
||||
// instead of one, which seems worse than doing 1 op before and breaking
|
||||
// the fusion.
|
||||
//
|
||||
// We set PF to unordered (V), but our PF representation is inverted so we
|
||||
// actually set to !V. This is one instruction with the VC cond code.
|
||||
OrderedNode *PFInvert =
|
||||
_NZCVSelect(OpSize::i32Bit, CondClassType{COND_FNU}, _Constant(1), _Constant(0));
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PFInvert);
|
||||
|
||||
// For the rest, this one weird a64 instruction maps exactly to what x86
|
||||
// needs. What a coincidence!
|
||||
_AXFlag();
|
||||
PossiblySetNZCVBits = ~0;
|
||||
|
||||
// It does assume we invert CF internally, which is still TODO for us. For
|
||||
// now, add a cfinv to deal. Hopefully we delete this later.
|
||||
CarryInvert();
|
||||
} else {
|
||||
OrderedNode *Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
OrderedNode *C_inv = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
OrderedNode *V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
// We want to zero SF/OF, and then set CF/ZF. Zeroing up front lets us do
|
||||
// this all with shifted-or's on non-flagm platforms.
|
||||
ZeroNZCV();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Or(OpSize::i32Bit, C_inv, V));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Or(OpSize::i32Bit, Z, V));
|
||||
|
||||
// Note that we store PF inverted.
|
||||
// TODO: We could maybe optimize this xor out for non-flagm platforms with
|
||||
// bfi/bfxil?
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Xor(OpSize::i32Bit, V, _Constant(1)));
|
||||
}
|
||||
}
|
||||
|
||||
// Set x87 comparison flags based on the result set by Arm FCMP. Clobbers
|
||||
// NZCV on flagm2 platforms.
|
||||
void ConvertNZCVToX87() {
|
||||
OrderedNode *V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
// Convert to x86 flags, saves us from or'ing after.
|
||||
_AXFlag();
|
||||
PossiblySetNZCVBits = ~0;
|
||||
|
||||
// Copy the values. CF is inverted from the axflag result, ZF is as-is.
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
} else {
|
||||
OrderedNode *Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
OrderedNode *N = GetRFLAG(FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Or(OpSize::i32Bit, N, V));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Or(OpSize::i32Bit, Z, V));
|
||||
}
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(V);
|
||||
}
|
||||
|
||||
// Helper to derive Dest by a given builder-using Expression with the opcode
|
||||
// replaced with NewOp. Useful for generic building code. Not safe in general.
|
||||
// but does the right handling of ImplicitFlagClobber at least and must be
|
||||
// used instead of raw Op mutation.
|
||||
#define DeriveOp(Dest, NewOp, Expr) \
|
||||
if (ImplicitFlagClobber(NewOp)) \
|
||||
SaveNZCV(NewOp); \
|
||||
auto Dest = (Expr); \
|
||||
Dest.first->Header.Op = (NewOp)
|
||||
|
||||
// Named constant cache for the current block.
|
||||
// Different arrays for sizes 1,2,4,8,16,32.
|
||||
OrderedNode *CachedNamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_MAX][6]{};
|
||||
@@ -1245,8 +1567,8 @@ private:
|
||||
CachedIndexedNamedVectorConstants.clear();
|
||||
}
|
||||
|
||||
OrderedNode *SelectMask(OrderedNode *Cmp, uint64_t Mask, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
OrderedNode *SelectNZCV(unsigned BitOffset, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP) const;
|
||||
OrderedNode *SelectBit(OrderedNode *Cmp, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
OrderedNode *SelectCC(uint8_t OP, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
|
||||
/**
|
||||
@@ -1279,7 +1601,7 @@ private:
|
||||
OrderedNode *Res{};
|
||||
|
||||
union {
|
||||
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, BITSELECT, RDRAND
|
||||
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, RDRAND
|
||||
struct {
|
||||
} NoSource;
|
||||
|
||||
@@ -1349,13 +1671,48 @@ private:
|
||||
return CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE;
|
||||
}
|
||||
|
||||
template <typename F>
|
||||
void CalculateFlags_ShiftVariable(OrderedNode *Shift, F&& CalculateFlags) {
|
||||
// We are the ones calculating the deferred flags. Don't recurse!
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
// RCR can call this with constants, so handle that without branching.
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Shift), &Const)) {
|
||||
if (Const)
|
||||
CalculateFlags();
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
// Otherwise, prepare to branch.
|
||||
uint32_t OldSetNZCVBits = PossiblySetNZCVBits;
|
||||
auto Zero = _Constant(0);
|
||||
|
||||
// If the shift is zero, do not touch the flags.
|
||||
auto SetBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto EndBlock = CreateNewCodeBlockAfter(SetBlock);
|
||||
CondJump(Shift, Zero, EndBlock, SetBlock, {COND_EQ});
|
||||
|
||||
SetCurrentCodeBlock(SetBlock);
|
||||
StartNewBlock();
|
||||
{
|
||||
CalculateFlags();
|
||||
Jump(EndBlock);
|
||||
}
|
||||
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
StartNewBlock();
|
||||
PossiblySetNZCVBits |= OldSetNZCVBits;
|
||||
}
|
||||
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
OrderedNode *LoadPFRaw();
|
||||
OrderedNode *LoadAF();
|
||||
void FixupAF();
|
||||
void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr);
|
||||
void CalculatePF(OrderedNode *Res);
|
||||
void CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
void CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool Sub);
|
||||
@@ -1378,7 +1735,6 @@ private:
|
||||
void CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BLSMSK(OrderedNode *Src);
|
||||
@@ -1387,7 +1743,6 @@ private:
|
||||
void CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BITSELECT(OrderedNode *Src);
|
||||
void CalculateFlags_RDRAND(OrderedNode *Src);
|
||||
/** @} */
|
||||
|
||||
@@ -1690,20 +2045,6 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_FCMP(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_FCMP,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BEXTR,
|
||||
@@ -1778,14 +2119,6 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BITSELECT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BITSELECT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RDRAND(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_RDRAND,
|
||||
|
||||
@@ -23,19 +23,34 @@ class OrderedNode;
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Tmp = _Ror(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), _Constant(32, 2));
|
||||
auto Top = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Src, 3), Tmp);
|
||||
auto Result = _VInsGPR(16, 4, 3, Src, Top);
|
||||
OrderedNode *RotatedNode{};
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
// Move the element to zero, rotate, and then move back (Using duplicates).
|
||||
// Saves one instruction versus that path that doesn't support SHA extension.
|
||||
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto Sha1HRotated = _VSha1H(Duplicated);
|
||||
RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
|
||||
}
|
||||
else {
|
||||
// SHA1 extension missing, manually rotate.
|
||||
// Emulate rotate.
|
||||
auto ShiftLeft = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Dest, 30);
|
||||
RotatedNode = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeft, Dest, 2);
|
||||
}
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode *NewVec = _VExtr(16, 8, Dest, Src, 1);
|
||||
|
||||
@@ -46,26 +61,34 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ROR by 31 is equivalent to a ROL by 1
|
||||
auto ThirtyOne = _Constant(32, 31);
|
||||
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
|
||||
// Do all the work without it.
|
||||
|
||||
auto W13 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W16 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), W13), ThirtyOne);
|
||||
auto W17 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 2), W14), ThirtyOne);
|
||||
auto W18 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 1), W15), ThirtyOne);
|
||||
auto W19 = _Ror(OpSize::i32Bit, _Xor(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 0), W16), ThirtyOne);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(OpSize::i32Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, W16);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, W17);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, W18);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, W19);
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
|
||||
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
|
||||
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
|
||||
|
||||
// Element0 didn't get XOR'd with anything, so do it now.
|
||||
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
|
||||
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
|
||||
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
|
||||
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
|
||||
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
@@ -102,8 +125,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(32, k_array[Imm8]);
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W1 = _VExtractToGPR(16, 4, Src, 2);
|
||||
@@ -151,30 +174,37 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
};
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Result{};
|
||||
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
Result = _VSha256U0(Dest, Src);
|
||||
}
|
||||
else {
|
||||
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
};
|
||||
|
||||
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
|
||||
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
|
||||
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
|
||||
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, Sig0);
|
||||
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
|
||||
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
|
||||
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
|
||||
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
|
||||
Result = _VInsGPR(16, 4, 0, D1, Sig0);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
@@ -182,8 +212,8 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))), _Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 3);
|
||||
@@ -214,8 +244,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), _Ror(OpSize::i32Bit, E, _Constant(32, 11))), _Ror(OpSize::i32Bit, E, _Constant(32, 25)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
@@ -260,14 +290,14 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEnc(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -280,8 +310,8 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -289,8 +319,8 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -303,8 +333,8 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -312,8 +342,8 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDec(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -326,8 +356,8 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDec(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -335,8 +365,8 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -349,8 +379,8 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -358,7 +388,7 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t RCON = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
@@ -375,8 +405,8 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Data.Literal.Value);
|
||||
|
||||
auto Res = _PCLMUL(16, Dest, Src, Selector);
|
||||
@@ -388,8 +418,8 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
|
||||
|
||||
OrderedNode *Res = _PCLMUL(DstSize, Src1, Src2, Selector);
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -139,7 +139,7 @@ void OpDispatchBuilder::FLD(OpcodeArgs) {
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Read from memory
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
}
|
||||
else {
|
||||
// Implicit arg
|
||||
@@ -178,7 +178,7 @@ void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
// Read from memory
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags, -1);
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode *converted = _F80BCDLoad(data);
|
||||
_StoreContextIndexed(converted, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -238,7 +238,7 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
size_t read_width = GetSrcSize(Op);
|
||||
|
||||
// Read from memory
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
|
||||
auto zero = _Constant(0);
|
||||
|
||||
@@ -247,10 +247,13 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
data = _Sbfe(OpSize::i64Bit, read_width * 8, 0, data);
|
||||
}
|
||||
|
||||
// Extract sign and make interger absolute
|
||||
auto sign = _Select(COND_SLT, data, zero, _Constant(0x8000), zero);
|
||||
// We're about to clobber flags to grab the sign, so save NZCV.
|
||||
SaveNZCV();
|
||||
|
||||
auto absolute = _Abs(OpSize::i64Bit, data);
|
||||
// Extract sign and make interger absolute
|
||||
_SubNZCV(OpSize::i64Bit, data, zero);
|
||||
auto sign = _NZCVSelect(OpSize::i64Bit, CondClassType{COND_SLT}, _Constant(0x8000), zero);
|
||||
auto absolute = _Neg(OpSize::i64Bit, data, CondClassType{COND_MI});
|
||||
|
||||
// left justify the absolute interger
|
||||
auto shift = _Sub(OpSize::i64Bit, _Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
@@ -334,11 +337,11 @@ void OpDispatchBuilder::FADD(OpcodeArgs) {
|
||||
// Memory arg
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -395,11 +398,11 @@ void OpDispatchBuilder::FMUL(OpcodeArgs) {
|
||||
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -458,11 +461,11 @@ void OpDispatchBuilder::FDIV(OpcodeArgs) {
|
||||
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -543,11 +546,11 @@ void OpDispatchBuilder::FSUB(OpcodeArgs) {
|
||||
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -724,11 +727,11 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
// Memory arg
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -763,13 +766,13 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(PF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
}
|
||||
|
||||
if constexpr (poptwice) {
|
||||
@@ -856,9 +859,7 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F80Round(a);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F80Round(a));
|
||||
|
||||
if constexpr (IROp == IR::OP_F80SIN ||
|
||||
IROp == IR::OP_F80COS) {
|
||||
@@ -889,9 +890,7 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
st1 = _LoadContextIndexed(st1, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F80Add(a, st1);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F80Add(a, st1));
|
||||
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
@@ -1014,7 +1013,7 @@ void OpDispatchBuilder::X87ATAN(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
@@ -1052,7 +1051,7 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
{
|
||||
@@ -1099,7 +1098,7 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
@@ -1110,7 +1109,7 @@ void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87LDSW(OpcodeArgs) {
|
||||
OrderedNode *NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
ReconstructX87StateFromFSW(NewFSW);
|
||||
}
|
||||
|
||||
@@ -1139,7 +1138,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
OrderedNode *Top = GetX87Top();
|
||||
@@ -1213,7 +1212,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
|
||||
@@ -66,7 +66,7 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
@@ -89,7 +89,7 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
//ignore the rounding precision, we're always 64-bit in F64.
|
||||
//extract rounding mode
|
||||
OrderedNode *roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
@@ -112,7 +112,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Read from memory
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
if constexpr (width == 32) {
|
||||
converted = _Float_FToF(8, 4, data);
|
||||
@@ -153,7 +153,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
// Read from memory
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags, -1);
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode *converted = _F80BCDLoad(data);
|
||||
converted = _F80CVT(8, converted);
|
||||
_StoreContextIndexed(converted, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -210,7 +210,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
|
||||
size_t read_width = GetSrcSize(Op);
|
||||
// Read from memory
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
if(read_width == 2) {
|
||||
data = _Sbfe(OpSize::i64Bit, read_width * 8, 0, data);
|
||||
}
|
||||
@@ -292,16 +292,16 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -353,16 +353,16 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -417,16 +417,16 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -503,16 +503,16 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -601,21 +601,13 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
|
||||
auto low = _Constant(0);
|
||||
OrderedNode *data = _VCastFromGPR(8, 8, low);
|
||||
|
||||
OrderedNode *Res = _FCmp(8, a, data,
|
||||
(1 << FCMP_FLAG_EQ) |
|
||||
(1 << FCMP_FLAG_LT) |
|
||||
(1 << FCMP_FLAG_UNORDERED));
|
||||
// We are going to clobber NZCV, make sure it's in a GPR first.
|
||||
GetNZCV();
|
||||
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
|
||||
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
// Now we do our comparison.
|
||||
_FCmp(8, a, data);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
|
||||
//TODO: This should obey rounding mode
|
||||
@@ -661,16 +653,16 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -681,36 +673,22 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode *Res = _FCmp(8, a, b,
|
||||
(1 << FCMP_FLAG_EQ) |
|
||||
(1 << FCMP_FLAG_LT) |
|
||||
(1 << FCMP_FLAG_UNORDERED));
|
||||
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
|
||||
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
|
||||
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
|
||||
|
||||
if constexpr (whichflags == FCOMIFlags::FLAGS_X87) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
// We are going to clobber NZCV, make sure it's in a GPR first.
|
||||
GetNZCV();
|
||||
|
||||
_FCmp(8, a, b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
else {
|
||||
// Invalidate deferred flags early
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(PF);
|
||||
_FCmp(8, a, b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToSSE();
|
||||
}
|
||||
|
||||
if constexpr (poptwice) {
|
||||
@@ -767,9 +745,7 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F64SIN(a);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F64SIN(a));
|
||||
|
||||
if constexpr (IROp == IR::OP_F64SIN ||
|
||||
IROp == IR::OP_F64COS) {
|
||||
@@ -799,9 +775,7 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
st1 = _LoadContextIndexed(st1, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F64ATAN(a, st1);
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
DeriveOp(result, IROp, _F64ATAN(a, st1));
|
||||
|
||||
if constexpr (IROp == IR::OP_F64FPREM ||
|
||||
IROp == IR::OP_F64FPREM1) {
|
||||
@@ -921,7 +895,7 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
OrderedNode *Top = GetX87Top();
|
||||
@@ -999,7 +973,7 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
|
||||
@@ -1,41 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <unistd.h>
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore {
|
||||
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SetHostSignalHandler(Signal, Func, Required);
|
||||
FrontendRegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void SignalDelegator::HandleSignal(int Signal, void *Info, void *UContext) {
|
||||
// Let the host take first stab at handling the signal
|
||||
auto Thread = GetTLSThread();
|
||||
HostSignalHandler &Handler = HostHandlers[Signal];
|
||||
|
||||
if (!Thread) {
|
||||
LogMan::Msg::AFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
|
||||
}
|
||||
else {
|
||||
for (auto &Handler : Handler.Handlers) {
|
||||
if (Handler(Thread, Signal, Info, UContext)) {
|
||||
// If the host handler handled the fault then we can continue now
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (Handler.FrontendHandler &&
|
||||
Handler.FrontendHandler(Thread, Signal, Info, UContext)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Now let the frontend handle the signal
|
||||
// It's clearly a guest signal and this ends up being an OS specific issue
|
||||
HandleGuestSignal(Thread, Signal, Info, UContext);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,84 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#ifndef NDEBUG
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore::X86Tables::X86InstDebugInfo {
|
||||
void InstallDebugInfo() {
|
||||
const std::tuple<uint8_t, uint8_t, Flags> BaseOpTable[] = {
|
||||
{0x50, 8, {FLAGS_MEM_ACCESS}},
|
||||
{0x58, 8, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0x68, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0x6A, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xAA, 4, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xC8, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xCC, 2, {FLAGS_DEBUG}},
|
||||
|
||||
{0xD7, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xF1, 1, {FLAGS_DEBUG}},
|
||||
{0xF4, 1, {FLAGS_DEBUG}},
|
||||
};
|
||||
|
||||
const std::tuple<uint8_t, uint8_t, Flags> TwoByteOpTable[] = {
|
||||
{0x0B, 1, {FLAGS_DEBUG}},
|
||||
{0x19, 7, {FLAGS_DEBUG}},
|
||||
{0x28, 2, {FLAGS_MEM_ALIGN_16}},
|
||||
|
||||
{0x31, 1, {FLAGS_DEBUG}},
|
||||
|
||||
{0xA2, 1, {FLAGS_DEBUG}},
|
||||
{0xA3, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0xAB, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0xB3, 1, {FLAGS_MEM_ACCESS}},
|
||||
{0xBB, 1, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
{0xFF, 1, {FLAGS_DEBUG}},
|
||||
};
|
||||
|
||||
const std::tuple<uint8_t, uint8_t, Flags> PrimaryGroupOpTable[] = {
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 2, {FLAGS_DIVIDE}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 6), 2, {FLAGS_DIVIDE}},
|
||||
#undef OPD
|
||||
};
|
||||
|
||||
const std::tuple<uint16_t, uint8_t, Flags> SecondaryExtensionOpTable[] = {
|
||||
#define PF_NONE 0
|
||||
#define PF_F3 1
|
||||
#define PF_66 2
|
||||
#define PF_F2 3
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_6) << 5) | (prefix) << 3 | (Reg))
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 2), 1, {FLAGS_DEBUG}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 3), 1, {FLAGS_DEBUG}},
|
||||
#undef PF_F3
|
||||
#undef PF_66
|
||||
#undef PF_F2
|
||||
#undef OPD
|
||||
};
|
||||
|
||||
auto GenerateDebugTable = [](auto& FinalTable, auto& LocalTable) {
|
||||
for (auto Op : LocalTable) {
|
||||
auto OpNum = std::get<0>(Op);
|
||||
auto DebugInfo = std::get<2>(Op);
|
||||
for (uint8_t i = 0; i < std::get<1>(Op); ++i) {
|
||||
memcpy(&FinalTable[OpNum+i].DebugInfo, &DebugInfo, sizeof(X86InstDebugInfo::Flags));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
GenerateDebugTable(BaseOps, BaseOpTable);
|
||||
GenerateDebugTable(SecondBaseOps, TwoByteOpTable);
|
||||
GenerateDebugTable(PrimaryInstGroupOps, PrimaryGroupOpTable);
|
||||
|
||||
GenerateDebugTable(SecondInstGroupOps, SecondaryExtensionOpTable);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
@@ -44,11 +44,6 @@ void InitializeVEXTables();
|
||||
void InitializeXOPTables();
|
||||
void InitializeEVEXTables();
|
||||
|
||||
#ifndef NDEBUG
|
||||
uint64_t Total{};
|
||||
uint64_t NumInsts{};
|
||||
#endif
|
||||
|
||||
void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeBaseTables(Mode);
|
||||
InitializeSecondaryTables(Mode);
|
||||
@@ -62,10 +57,6 @@ void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeVEXTables();
|
||||
InitializeXOPTables();
|
||||
InitializeEVEXTables();
|
||||
|
||||
#ifndef NDEBUG
|
||||
X86InstDebugInfo::InstallDebugInfo();
|
||||
#endif
|
||||
}
|
||||
|
||||
}
|
||||
@@ -100,10 +100,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x6B, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
|
||||
// This should just throw a GP
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x70, 1, X86InstInfo{"JO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x71, 1, X86InstInfo{"JNO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
@@ -147,19 +147,19 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
|
||||
{0xB0, 8, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_REX_IN_BYTE , 1, nullptr}},
|
||||
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2, 4, nullptr}},
|
||||
@@ -169,7 +169,7 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xC8, 1, X86InstInfo{"ENTER", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 3, nullptr}},
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0xCA, 2, X86InstInfo{"RETF", TYPE_PRIV, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1, nullptr}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
@@ -192,8 +192,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xEC, 2, X86InstInfo{"IN", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xEE, 2, X86InstInfo{"OUT", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xF5, 1, X86InstInfo{"CMC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF8, 1, X86InstInfo{"CLC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF9, 1, X86InstInfo{"STC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -183,41 +183,41 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_9, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 10
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_NONE, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F3, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_10, PF_66, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_66, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 0), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 1), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 2), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 3), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 4), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 5), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 6), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_10, PF_F2, 7), 1, X86InstInfo{"UD1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
// GROUP 12
|
||||
{OPD(TYPE_GROUP_12, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -29,7 +29,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x09, 1, X86InstInfo{"WBINVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0A, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0C, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0D, 1, X86InstInfo{"", TYPE_GROUP_P, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0E, 1, X86InstInfo{"FEMMS", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -44,7 +44,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x16, 1, X86InstInfo{"MOVLHPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x17, 1, X86InstInfo{"MOVHPS", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x18, 1, X86InstInfo{"", TYPE_GROUP_16, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x19, 7, X86InstInfo{"NOP", TYPE_INST, FLAGS_DEBUG | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x19, 7, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
{0x20, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x22, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -59,7 +59,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x2F, 1, X86InstInfo{"COMISS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x30, 1, X86InstInfo{"WRMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x31, 1, X86InstInfo{"RDTSC", TYPE_INST, FLAGS_DEBUG | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x31, 1, X86InstInfo{"RDTSC", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x32, 1, X86InstInfo{"RDMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x33, 1, X86InstInfo{"RDPMC", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x34, 1, X86InstInfo{"SYSENTER", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -166,7 +166,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SETLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"SETNLE", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_DEBUG | FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -254,7 +254,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0xFC, 1, X86InstInfo{"PADDB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFD, 1, X86InstInfo{"PADDW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFE, 1, X86InstInfo{"PADDD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_DEBUG | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
// FEX reserved instructions
|
||||
// Unused x86 encoding instruction.
|
||||
|
||||
@@ -279,9 +279,15 @@ namespace InstFlags {
|
||||
using InstFlagType = uint64_t;
|
||||
|
||||
constexpr InstFlagType FLAGS_NONE = 0;
|
||||
constexpr InstFlagType FLAGS_DEBUG = (1ULL << 1);
|
||||
// The secondary Opcode Map uses prefix bytes to overlay more instruction
|
||||
// But some instructions need to ignore this overlay and consume these prefixes.
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY = (1ULL << 0);
|
||||
// Some instructions partially ignore overlay
|
||||
// Ignore OpSize (0x66) in this case
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY66 = (1ULL << 1);
|
||||
constexpr InstFlagType FLAGS_DEBUG_MEM_ACCESS = (1ULL << 2);
|
||||
constexpr InstFlagType FLAGS_SUPPORTS_REP = (1ULL << 3);
|
||||
// Only SEXT if the instruction is operating in 64bit operand size
|
||||
constexpr InstFlagType FLAGS_SRC_SEXT64BIT = (1ULL << 3);
|
||||
constexpr InstFlagType FLAGS_BLOCK_END = (1ULL << 4);
|
||||
constexpr InstFlagType FLAGS_SETS_RIP = (1ULL << 5);
|
||||
|
||||
@@ -331,27 +337,17 @@ constexpr InstFlagType FLAGS_MODRM = (1ULL << 16);
|
||||
constexpr InstFlagType FLAGS_SF_MOD_MEM_ONLY = (1ULL << 18);
|
||||
constexpr InstFlagType FLAGS_SF_MOD_REG_ONLY = (1ULL << 19);
|
||||
|
||||
// The secondary Opcode Map uses prefix bytes to overlay more instruction
|
||||
// But some instructions need to ignore this overlay and consume these prefixes.
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY = (1ULL << 20);
|
||||
// Some instructions partially ignore overlay
|
||||
// Ignore OpSize (0x66) in this case
|
||||
constexpr InstFlagType FLAGS_NO_OVERLAY66 = (1ULL << 21);
|
||||
|
||||
// x87
|
||||
constexpr InstFlagType FLAGS_POP = (1ULL << 22);
|
||||
|
||||
// Only SEXT if the instruction is operating in 64bit operand size
|
||||
constexpr InstFlagType FLAGS_SRC_SEXT64BIT = (1ULL << 23);
|
||||
constexpr InstFlagType FLAGS_POP = (1ULL << 20);
|
||||
|
||||
// Whether or not the instruction has a VEX prefix for the first source operand
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 24);
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 21);
|
||||
// Whether or not the instruction has a VEX prefix for the second source operand
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 25);
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 22);
|
||||
// Whether or not the instruction has a VEX prefix for the destination
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 26);
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 23);
|
||||
// Whether or not the instruction has a VSIB byte
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 27);
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 24);
|
||||
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
@@ -419,35 +415,12 @@ constexpr uint8_t OpToIndex(uint8_t Op) {
|
||||
using DecodedOp = DecodedInst const*;
|
||||
using OpDispatchPtr = void (IR::OpDispatchBuilder::*)(DecodedOp);
|
||||
|
||||
#ifndef NDEBUG
|
||||
namespace X86InstDebugInfo {
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_4 = (1 << 0);
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_8 = (1 << 1);
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_16 = (1 << 2);
|
||||
constexpr uint64_t FLAGS_MEM_ALIGN_SIZE = (1 << 3); // If instruction size changes depending on prefixes
|
||||
constexpr uint64_t FLAGS_MEM_ACCESS = (1 << 4);
|
||||
constexpr uint64_t FLAGS_DEBUG = (1 << 5);
|
||||
constexpr uint64_t FLAGS_DIVIDE = (1 << 6);
|
||||
|
||||
|
||||
struct Flags {
|
||||
uint64_t DebugFlags;
|
||||
};
|
||||
void InstallDebugInfo();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
struct X86InstInfo {
|
||||
char const *Name;
|
||||
InstType Type;
|
||||
InstFlags::InstFlagType Flags; ///< Must be larger than InstFlags enum
|
||||
uint8_t MoreBytes;
|
||||
OpDispatchPtr OpcodeDispatcher;
|
||||
#ifndef NDEBUG
|
||||
X86InstDebugInfo::Flags DebugInfo;
|
||||
uint32_t NumUnitTestsGenerated;
|
||||
#endif
|
||||
|
||||
bool operator==(const X86InstInfo &b) const {
|
||||
if (strcmp(Name, b.Name) != 0 ||
|
||||
@@ -524,12 +497,6 @@ extern std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps;
|
||||
// EVEX
|
||||
extern std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps;
|
||||
|
||||
|
||||
#ifndef NDEBUG
|
||||
extern uint64_t Total;
|
||||
extern uint64_t NumInsts;
|
||||
#endif
|
||||
|
||||
template <typename OpcodeType>
|
||||
struct X86TablesInfoStruct {
|
||||
OpcodeType first;
|
||||
@@ -548,11 +515,6 @@ static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInfoStruct<Op
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
FinalTable[OpNum + i] = Info;
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -570,11 +532,6 @@ static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86TablesInfoS
|
||||
}
|
||||
else {
|
||||
FinalTable[OpNum + i] = Info;
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -602,11 +559,6 @@ static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct
|
||||
}
|
||||
}
|
||||
}
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -79,7 +79,7 @@ namespace FEXCore {
|
||||
const char *Name;
|
||||
};
|
||||
|
||||
static thread_local FEXCore::Core::InternalThreadState *Thread;
|
||||
static thread_local FEXCore::Core::InternalThreadState *Thread = nullptr;
|
||||
|
||||
|
||||
struct ExportEntry { uint8_t *sha256; ThunkedFunction* Fn; };
|
||||
@@ -171,8 +171,18 @@ namespace FEXCore {
|
||||
Set arg0/1 to arg regs, use CTX::HandleCallback to handle the callback
|
||||
*/
|
||||
static void CallCallback(void *callback, void *arg0, void* arg1) {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = (uintptr_t)arg1;
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to invoke guest callback asynchronously");
|
||||
}
|
||||
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = (uintptr_t)arg1;
|
||||
} else {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RCX] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDX] = (uintptr_t)arg1;
|
||||
}
|
||||
|
||||
Thread->CTX->HandleCallback(Thread, (uintptr_t)callback);
|
||||
}
|
||||
@@ -220,7 +230,7 @@ namespace FEXCore {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, gregs[X86State::REG_R11]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
}
|
||||
else {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, mm[0][0]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(8, 8, emit->_Constant(Entrypoint)), offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
}, CTX->ThunkHandler.get(), (void*)args->target_addr);
|
||||
|
||||
@@ -363,18 +363,9 @@ namespace FEXCore::IR {
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// Add to thread local ir cache
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), std::move(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
Thread->DebugStore.insert({GuestRIP, std::move(Entry)});
|
||||
}
|
||||
else {
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
if (IRList->IsCopy()) delete IRList;
|
||||
}
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
if (IRList->IsCopy()) delete IRList;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+440
-121
@@ -67,8 +67,6 @@
|
||||
"constexpr uint8_t COND_SLT = 11",
|
||||
"constexpr uint8_t COND_SGT = 12",
|
||||
"constexpr uint8_t COND_SLE = 13",
|
||||
"constexpr uint8_t COND_ANDZ = 14 /* (a & b) == 0 */",
|
||||
"constexpr uint8_t COND_ANDNZ = 15 /* (a & b) != 0 */",
|
||||
|
||||
"constexpr uint8_t COND_FLU = 16 /* float less or unordred */",
|
||||
"constexpr uint8_t COND_FGE = 17 /* float greater or equal */",
|
||||
@@ -77,6 +75,8 @@
|
||||
"constexpr uint8_t COND_FU = 20 /* float unordred */",
|
||||
"constexpr uint8_t COND_FNU = 21 /* float not unordred */",
|
||||
|
||||
"constexpr uint8_t COND_AL = 32 /* always */",
|
||||
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRClass {0}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {2}",
|
||||
@@ -156,35 +156,43 @@
|
||||
"MemOffsetType": "MemOffsetType",
|
||||
"BreakDefinition": "BreakDefinition",
|
||||
"RoundType": "RoundType",
|
||||
"FloatCompareOp": "FloatCompareOp",
|
||||
"NamedVectorConstant": "FEXCore::IR::NamedVectorConstant",
|
||||
"IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant"
|
||||
"IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant",
|
||||
"ShiftType": "FEXCore::IR::ShiftType"
|
||||
},
|
||||
"Ops": {
|
||||
"Misc": {
|
||||
"Dummy": {
|
||||
"HasSideEffects": true,
|
||||
"SwitchGen": false
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions": {
|
||||
"SwitchGen": false
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"CodeBlock SSA:$Begin, SSA:$Last": {
|
||||
"SwitchGen": false,
|
||||
"RAOverride": "0"
|
||||
"RAOverride": "0",
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"BeginBlock SSA:$BlockHeader": {
|
||||
"HasSideEffects": true,
|
||||
"SwitchGen": false,
|
||||
"RAOverride": "0"
|
||||
"RAOverride": "0",
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"InvalidateFlags u64:$Flags": {
|
||||
"HasSideEffects": true
|
||||
"HasSideEffects": true,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
|
||||
"EndBlock SSA:$BlockHeader": {
|
||||
"HasSideEffects": true,
|
||||
"SwitchGen": false,
|
||||
"RAOverride": "0"
|
||||
"RAOverride": "0",
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
|
||||
"GuestOpcode u32:$GuestEntryOffset": {
|
||||
@@ -195,6 +203,7 @@
|
||||
"GPR = ValidateCode u64:$CodeOriginalLow, u64:$CodeOriginalhigh, i64:$Offset, u8:$CodeLength": {
|
||||
"HasSideEffects": true,
|
||||
"HasDest": true,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
@@ -240,6 +249,7 @@
|
||||
"The second GPR pair element is a bool if the number is valid",
|
||||
"RNG hardware is allowed to fail early and return. Software must always check this"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
},
|
||||
@@ -254,7 +264,7 @@
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "0"
|
||||
},
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, u8:$CompareSize{0}": {
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, u8:$CompareSize{0}, i1:$FromNZCV{false}": {
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2",
|
||||
"EmitValidation": [
|
||||
@@ -337,7 +347,8 @@
|
||||
"Desc": ["Loads a value from the static-ra context with offset",
|
||||
"Dest = Ctx[Offset]"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true
|
||||
},
|
||||
|
||||
"StoreRegister SSA:$Value, i1:$IsPrewrite, u32:$Offset, RegisterClass:$Class, RegisterClass:$StaticClass, u8:#Size": {
|
||||
@@ -348,6 +359,7 @@
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true,
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class"
|
||||
]
|
||||
@@ -435,6 +447,17 @@
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = LoadNZCV": {
|
||||
"Desc": ["Loads value of NZCV register"],
|
||||
"DestSize": "4"
|
||||
},
|
||||
|
||||
"StoreNZCV GPR:$Value": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Stores value to NZCV register"],
|
||||
"DestSize": "4"
|
||||
},
|
||||
|
||||
"GPR = LoadFlag u32:$Flag": {
|
||||
"Desc": ["Loads an x86-64 flag from the context object",
|
||||
"Specialized to allow flexible implementation of flag handling"
|
||||
@@ -472,7 +495,8 @@
|
||||
"SSA = LoadMemTSO RegisterClass:$Class, u8:#Size, GPR:$Addr, GPR:$Offset, u8:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a x86 TSO compatible load from memory. Offset must be Invalid()."
|
||||
],
|
||||
"DestSize": "Size"
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true
|
||||
},
|
||||
|
||||
"StoreMemTSO RegisterClass:$Class, u8:#Size, SSA:$Value, GPR:$Addr, GPR:$Offset, u8:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
@@ -480,6 +504,7 @@
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true,
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class"
|
||||
]
|
||||
@@ -488,6 +513,7 @@
|
||||
"FPR = VLoadVectorMasked u8:#RegisterSize, u8:#ElementSize, FPR:$Mask, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a masked load similar to VPMASKMOV/VMASKMOV where the upper bit of each element",
|
||||
"determines whether or not that element will be loaded from memory"],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -495,6 +521,7 @@
|
||||
"Desc": ["Does a masked store similar to VPMASKMOV/VMASKMOV where the upper bit of each element",
|
||||
"determines whether or not that element will be stored to memory"],
|
||||
"HasSideEffects": true,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -586,6 +613,7 @@
|
||||
],
|
||||
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -600,6 +628,7 @@
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"NumElements": "2",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
|
||||
@@ -607,7 +636,8 @@
|
||||
},
|
||||
"AtomicAdd OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer add"
|
||||
"Desc": ["Atomic integer add",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -616,7 +646,8 @@
|
||||
},
|
||||
"AtomicSub OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer sub"
|
||||
"Desc": ["Atomic integer sub",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -625,7 +656,18 @@
|
||||
},
|
||||
"AtomicAnd OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer and"
|
||||
"Desc": ["Atomic integer and",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"AtomicCLR OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer binary clear",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -634,7 +676,8 @@
|
||||
},
|
||||
"AtomicOr OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer or"
|
||||
"Desc": ["Atomic integer or",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -643,7 +686,18 @@
|
||||
},
|
||||
"AtomicXor OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer xor"
|
||||
"Desc": ["Atomic integer xor",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"AtomicNeg OpSize:#Size, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer two's complement negate",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -663,7 +717,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and add",
|
||||
"Atomically fetches %Addr and adds %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -674,7 +729,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and sub",
|
||||
"Atomically fetches %Addr and subtracts %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -685,7 +741,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and binary and",
|
||||
"Atomically fetches %Addr and binary ands %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -698,7 +755,8 @@
|
||||
"Atomically fetches %Addr and binary clears %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"Matches ARM ldclral semantics",
|
||||
"eg: Dest[Addr] &= ~Value"
|
||||
"eg: Dest[Addr] &= ~Value",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -709,7 +767,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and binary or",
|
||||
"Atomically fetches %Addr and binary ors %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -720,7 +779,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and binary exclusive or",
|
||||
"Atomically fetches %Addr and binary exclusive ors %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -730,7 +790,8 @@
|
||||
"GPR = AtomicFetchNeg OpSize:#Size, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and two's complement negate",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -742,6 +803,7 @@
|
||||
"Desc": ["Set Telemetry value if the passed in 32-bit value isn't zero.",
|
||||
"Only useful for 32-bit applications."
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "8"
|
||||
}
|
||||
},
|
||||
@@ -803,19 +865,9 @@
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
"GPR = Neg OpSize:#Size, GPR:$Src": {
|
||||
"Desc": ["Integer negation",
|
||||
"Dest = -Src",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Abs OpSize:#Size, GPR:$Src": {
|
||||
"Desc": ["Integer 2's complement absolute value",
|
||||
"Dest = std::abs(Src)",
|
||||
"GPR = Neg OpSize:#Size, GPR:$Src, CondClass:$Cond{{COND_AL}}": {
|
||||
"Desc": ["Integer negation, with optional predication",
|
||||
"Dest = Cond ? -Src : Src",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
@@ -848,6 +900,7 @@
|
||||
"In the case of zero returns ~0U"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -900,25 +953,48 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AddNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs"],
|
||||
"DestSize": "4",
|
||||
"AddNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the sum of two GPRs"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AdcNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"DestSize": "4",
|
||||
"CarryInvert": {
|
||||
"Desc": ["Invert carry flag in NZCV"],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"AXFlag": {
|
||||
"Desc": ["After an FCmp, converts NZCV flags from the Arm format to a mysterious eXternal format"],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"RmifNZCV GPR:$Src, u8:$Rotate, u8:$Mask": {
|
||||
"Desc": ["Rotate, mask, and insert into NZCV on FlagM platforms"],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"CondAddNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2, CondClass:$Cond, u8:$FalseNZCV": {
|
||||
"Desc": ["If condition is true, set NZCV per sum of GPRs, else force NZCV to a constant."],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = SbbNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"DestSize": "4",
|
||||
"AdcNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"SbbNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the difference of two GPRs and carry-in given as NZCV"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Sub OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
@@ -930,13 +1006,24 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = SubNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, u8:$InvertCarry": {
|
||||
"Desc": ["Return NZCV for the difference of two GPRs. ",
|
||||
"If InvertCarry is nonzero, carry flag uses x86 definition, inverted from arm64.",
|
||||
""],
|
||||
"DestSize": "4",
|
||||
"GPR = SubShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
|
||||
"Desc": [ "Integer Sub with shifted register",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"_Shift != ShiftType::ROR"
|
||||
]
|
||||
},
|
||||
"SubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the difference of two GPRs. ",
|
||||
"Carry flag uses arm64 definition, inverted x86.",
|
||||
""],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Or OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
@@ -979,6 +1066,13 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = XorShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
|
||||
"Desc": [ "Integer binary exclusive or with shifted register"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = And OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary and"
|
||||
],
|
||||
@@ -994,9 +1088,10 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = TestNZ u8:$Size, GPR:$Src1": {
|
||||
"Desc": ["Return NZCV for a GPR, setting N and Z accordingly and zeroing C and V"],
|
||||
"DestSize": "4"
|
||||
"TestNZ OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the binary AND of two GPRs, setting N and Z accordingly and zeroing C and V"],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift left"
|
||||
@@ -1141,12 +1236,23 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = NZCVSelect OpSize:#ResultSize, CondClass:$Cond, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
"Desc": ["Select based on value in NZCV flags",
|
||||
"op:",
|
||||
"Dest = Cond ? TrueVal : FalseVal"
|
||||
],
|
||||
"DestSize": "ResultSize",
|
||||
"EmitValidation": [
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Select OpSize:#ResultSize, OpSize:$CompareSize, CondClass:$Cond, SSA:$Cmp1, SSA:$Cmp2, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
"Desc": ["Ternary selection of GPRs",
|
||||
"op:",
|
||||
"Dest = Cmp1 <Cond> Cmp2 ? TrueVal : FalseVal"
|
||||
],
|
||||
"DestSize": "ResultSize",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_CompareSize == FEXCore::IR::OpSize::i32Bit || _CompareSize == FEXCore::IR::OpSize::i64Bit || _CompareSize == FEXCore::IR::OpSize::i128Bit",
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit",
|
||||
@@ -1251,11 +1357,166 @@
|
||||
"DestSize": "DestElementSize"
|
||||
},
|
||||
|
||||
"GPR = FCmp u8:$ElementSize, FPR:$Scalar1, FPR:$Scalar2, u32:$Flags": {
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"FCmp u8:$ElementSize, FPR:$Scalar1, FPR:$Scalar2": {
|
||||
"Desc": ["Does a scalar unordered compare and sets NZCV accordingly.",
|
||||
"NZCV follows Arm conventions, a separate AXFLAG instruction is required for x86",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"DestSize": "4"
|
||||
"HasSideEffects": true
|
||||
}
|
||||
},
|
||||
"VectorScalar": {
|
||||
"FPR = VFAddScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'add' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFSubScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'sub' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMulScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'mul' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFDivScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'div' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMinScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'min' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
|
||||
"Additionally matches x86 zero and NaN semantics",
|
||||
"If both source operands are zero, return the second operand (in the case of negative and positive zero)",
|
||||
"If either source operand is NaN then return the second operand."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"ImplicitFlagClobber": true
|
||||
},
|
||||
"FPR = VFMaxScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'max' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
|
||||
"Additionally matches x86 zero and NaN semantics",
|
||||
"If both source operands are zero, return the second operand (in the case of negative and positive zero)",
|
||||
"If either source operand is NaN then return the second operand."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"ImplicitFlagClobber": true
|
||||
},
|
||||
"FPR = VFSqrtScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'sqrt' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFRSqrtScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'rsqrt' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFRecpScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'recip' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFToFScalarInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cvt' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
},
|
||||
"FPR = VSToFVectorInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i8:$HasTwoElements, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a Vector 'scvt' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
|
||||
"HasTwoElements is slightly different than most of these scalar operations.",
|
||||
"Handles the edge case of cvtpi2ps xmm0, mm0 which is two elements in the lower 64-bits"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
},
|
||||
"FPR = VSToFGPRInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector, GPR:$Src, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cvt' between Vector1 and GPR.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
},
|
||||
"FPR = VFToIScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, RoundType:$Round, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar round float to integral on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Rounding mode determined by argument",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFCMPScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FloatCompareOp:$Op, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cmp' between Vector1 and Vecto2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Compare op determined by argument",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
@@ -1274,7 +1535,7 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"FPR = VectorImm u8:#RegisterSize, u8:#ElementSize, u8:$Immediate": {
|
||||
"FPR = VectorImm u8:#RegisterSize, u8:#ElementSize, u8:$Immediate, u8:$ShiftAmount{0}": {
|
||||
"Desc": ["Generates a vector with each element containg the immediate zexted"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -1389,6 +1650,10 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUShraI u8:#RegisterSize, u8:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSShrI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1600,6 +1865,13 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFAddV u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Does a horizontal float vector add of elements across the source vector",
|
||||
"Result is a zero extended scalar"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFSub u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1621,11 +1893,7 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUMul u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSMul u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"FPR = VMul u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1790,6 +2058,15 @@
|
||||
],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VTBX1 u8:#RegisterSize, FPR:$VectorSrcDst, FPR:$VectorTable, FPR:$VectorIndices": {
|
||||
"Desc": ["Does a vector table lookup from one register in to the destination",
|
||||
"Lookup is byte sized per byte element.",
|
||||
"Any index larger than what the registers provide will result in not modifying that element",
|
||||
"Table is always treated as a 128bit register",
|
||||
"Indices matches destination size. Either 64bit or 128bit"
|
||||
],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VBSL u8:#RegisterSize, FPR:$VectorMask, FPR:$VectorTrue, FPR:$VectorFalse": {
|
||||
"Desc": ["Does a vector bitwise select.",
|
||||
"If the bit in the field is 1 then the corresponding bit is pulled from VectorTrue",
|
||||
@@ -1807,7 +2084,8 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"DestSize": "4"
|
||||
"DestSize": "4",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = VPCMPISTRX FPR:$LHS, FPR:$RHS, u8:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPISTRI/PCMPISTRM instruction",
|
||||
@@ -1818,7 +2096,8 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"DestSize": "4"
|
||||
"DestSize": "4",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = VFCADD u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, u16:$Rotate": {
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -1908,6 +2187,14 @@
|
||||
"Desc": "Assists in key generation",
|
||||
"DestSize": "16"
|
||||
},
|
||||
"FPR = VSha1H FPR:$Src": {
|
||||
"Desc": "Does vector scalar SHA1H instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i32Bit"
|
||||
},
|
||||
"FPR = VSha256U0 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar VSha256U0 instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit"
|
||||
},
|
||||
"GPR = CRC32 GPR:$Src1, GPR:$Src2, u8:$SrcSize": {
|
||||
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
|
||||
],
|
||||
@@ -1925,111 +2212,143 @@
|
||||
}
|
||||
},
|
||||
"F64": {
|
||||
"FPR = F64ATAN FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64FPREM FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64FPREM1 FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64SCALE FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64F2XM1 FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64FYL2X FPR:$Src, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64TAN FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64SIN FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64COS FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
}
|
||||
"FPR = F64ATAN FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64FPREM FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64FPREM1 FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64SCALE FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64F2XM1 FPR:$Src": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64FYL2X FPR:$Src, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64TAN FPR:$Src": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64SIN FPR:$Src": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64COS FPR:$Src": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
}
|
||||
},
|
||||
"F80": {
|
||||
"FPR = F80Add FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80Sub FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80Mul FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
|
||||
"FPR = F80Div FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80ATAN FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80FPREM FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80FPREM1 FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80SCALE FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80CVT u8:#Size, FPR:$X80Src": {
|
||||
"DestSize": "Size"
|
||||
"DestSize": "Size",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = F80CVTInt u8:#Size, FPR:$X80Src, i1:$Truncate": {
|
||||
"DestSize": "Size"
|
||||
"DestSize": "Size",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80CVTTo FPR:$X80Src, u8:$SrcSize": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80CVTToInt GPR:$Src, u8:$SrcSize": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80Round FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80F2XM1 FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80TAN FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80SIN FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80COS FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80SQRT FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80XTRACT_EXP FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80XTRACT_SIG FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = F80Cmp FPR:$X80Src1, FPR:$X80Src2, u32:$Flags": {
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"DestSize": "4"
|
||||
"DestSize": "4",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80BCDLoad FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80BCDStore FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
|
||||
"FPR = F80FYL2X FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
}
|
||||
},
|
||||
"Backend": {
|
||||
|
||||
@@ -44,6 +44,11 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, CondClassType Arg) {
|
||||
if (Arg == COND_AL) {
|
||||
*out << "ALWAYS";
|
||||
return;
|
||||
}
|
||||
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
@@ -238,6 +243,18 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::FloatCompareOp Arg) {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: *out << "FEQ"; break;
|
||||
case FloatCompareOp::LT: *out << "FLT"; break;
|
||||
case FloatCompareOp::LE: *out << "FLE"; break;
|
||||
case FloatCompareOp::UNO: *out << "UNO"; break;
|
||||
case FloatCompareOp::NEQ: *out << "NEQ"; break;
|
||||
case FloatCompareOp::ORD: *out << "ORD"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::BreakDefinition Arg) {
|
||||
*out << "{" << Arg.ErrorRegister << ".";
|
||||
*out << static_cast<uint32_t>(Arg.Signal) << ".";
|
||||
@@ -245,6 +262,16 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
*out << static_cast<uint32_t>(Arg.si_code) << "}";
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::ShiftType Arg) {
|
||||
switch (Arg) {
|
||||
case ShiftType::LSL: *out << "LSL"; break;
|
||||
case ShiftType::LSR: *out << "LSR"; break;
|
||||
case ShiftType::ASR: *out << "ASR"; break;
|
||||
case ShiftType::ROR: *out << "ROR"; break;
|
||||
default: *out << "<Unknown Shift Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
|
||||
@@ -6,12 +6,11 @@ tags: ir|parser
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/StringUtils.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
@@ -194,8 +194,6 @@ public:
|
||||
|
||||
private:
|
||||
bool HandleConstantPools(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
bool ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& CurrentIR,
|
||||
OrderedNode* CodeNode, IROp_Header* IROp);
|
||||
@@ -267,91 +265,6 @@ bool ConstProp::HandleConstantPools(IREmitter *IREmit, const IRListView& Current
|
||||
return Changed;
|
||||
}
|
||||
|
||||
// Code motion around selects
|
||||
// Moves unary ops that depend on a select before the select, if both inputs are constants
|
||||
// assumes that unary ops without side effects on constants will be constprop'd
|
||||
void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
// Code motion around selects
|
||||
// Moves unary ops that depend on a select before the select, if both inputs are constants
|
||||
// assumes that unary ops without side effects on constants will be constprop'd
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto BlockOp = BlockIROp->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
for (auto [UnaryOpNode, UnaryOpHdr] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)) {
|
||||
// could be moved
|
||||
auto SelectOpNode = IREmit->UnwrapNode(UnaryOpHdr->Args[0]);
|
||||
auto SelectOpHdr = IREmit->GetOpHeader(UnaryOpHdr->Args[0]);
|
||||
auto SelectOp = SelectOpHdr->CW<IR::IROp_Select>();
|
||||
|
||||
// the value isn't used after the select otherwise
|
||||
// make sure the sizes match
|
||||
if (SelectOpHdr->Size == UnaryOpHdr->Size && SelectOpHdr->Op == OP_SELECT && SelectOpNode->NumUses == 1
|
||||
&& IREmit->IsValueConstant(SelectOp->TrueVal)
|
||||
&& IREmit->IsValueConstant(SelectOp->FalseVal)) {
|
||||
|
||||
IREmit->SetWriteCursor(IREmit->UnwrapNode(SelectOpNode->Header.Previous));
|
||||
|
||||
size_t OpSize = FEXCore::IR::GetSize(UnaryOpHdr->Op);
|
||||
|
||||
/// copy for TrueVal ///
|
||||
auto NewUnaryOp1 = IREmit->AllocateRawOp(OpSize);
|
||||
|
||||
// Copy over the op
|
||||
memcpy(NewUnaryOp1.first, UnaryOpHdr, OpSize);
|
||||
|
||||
for (int i = 0; i < IR::GetArgs(NewUnaryOp1.first->Op); i++) {
|
||||
NewUnaryOp1.first->Args[i] = IREmit->WrapNode(IREmit->Invalid());
|
||||
}
|
||||
// Set New Op to operate on the constant
|
||||
IREmit->ReplaceNodeArgument(NewUnaryOp1, 0, IREmit->UnwrapNode(SelectOp->TrueVal));
|
||||
// Make select use the operated constant
|
||||
IREmit->ReplaceNodeArgument(SelectOpNode, 2, NewUnaryOp1);
|
||||
|
||||
/// copy for FalseVal ///
|
||||
auto NewUnaryOp2 = IREmit->AllocateRawOp(OpSize);
|
||||
|
||||
// Copy over the op
|
||||
memcpy(NewUnaryOp2.first, UnaryOpHdr, OpSize);
|
||||
|
||||
for (int i = 0; i < IR::GetArgs(NewUnaryOp2.first->Op); i++) {
|
||||
NewUnaryOp2.first->Args[i] = IREmit->WrapNode(IREmit->Invalid());
|
||||
}
|
||||
// Set New Op to operate on the constant
|
||||
IREmit->ReplaceNodeArgument(NewUnaryOp2, 0, IREmit->UnwrapNode(SelectOp->FalseVal));
|
||||
// Make select use the operated constant
|
||||
IREmit->ReplaceNodeArgument(SelectOpNode, 3, NewUnaryOp2);
|
||||
|
||||
// Replace uses of the defuct unary op w/ select
|
||||
IREmit->ReplaceAllUsesWithRange(UnaryOpNode, SelectOpNode, IREmit->GetIterator(IREmit->WrapNode(UnaryOpNode)), IREmit->GetIterator(BlockOp->Last));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ConstProp::FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
// Make all FCMPs set no flags
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
if (IROp->Op == OP_FCMP) {
|
||||
auto fcmp = IROp->CW<IR::IROp_FCmp>();
|
||||
fcmp->Flags = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Set needed flags
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
if (IROp->Op == OP_GETHOSTFLAG) {
|
||||
auto ghf = IROp->CW<IR::IROp_GetHostFlag>();
|
||||
|
||||
auto fcmp = IREmit->GetOpHeader(ghf->Value)->CW<IR::IROp_FCmp>();
|
||||
LOGMAN_THROW_AA_FMT(fcmp->Header.Op == OP_FCMP || fcmp->Header.Op == OP_F80CMP, "Unexpected OP_GETHOSTFLAG source");
|
||||
if(fcmp->Header.Op == OP_FCMP) {
|
||||
fcmp->Flags |= 1 << ghf->Flag;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// LoadMem / StoreMem imm pooling
|
||||
// If imms are close by, use address gen to generate the values instead of using a new imm
|
||||
void ConstProp::LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
@@ -419,7 +332,6 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_AND: {
|
||||
// if AND's arguments are imms, they are masking
|
||||
for (int i = 0; i < IR::GetArgs(IROp->Op); i++) {
|
||||
@@ -494,27 +406,6 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VFADD:
|
||||
case OP_VFSUB:
|
||||
case OP_VFMUL:
|
||||
case OP_VFDIV:
|
||||
case OP_FCMP: {
|
||||
auto flopSize = IROp->Size;
|
||||
for (int i = 0; i < IR::GetArgs(IROp->Op); i++) {
|
||||
auto argHeader = IREmit->GetOpHeader(IROp->Args[i]);
|
||||
|
||||
if (argHeader->Op == OP_VMOV) {
|
||||
auto source = argHeader->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
if (sourceHeader->Size >= flopSize) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(source));
|
||||
//printf("VMOV bypassed\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VMOV: {
|
||||
// elim from load mem
|
||||
auto source = IROp->Args[0];
|
||||
@@ -668,13 +559,31 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
bool IsConstant1 = IREmit->IsValueConstant(Op->Header.Args[0], &Constant1);
|
||||
bool IsConstant2 = IREmit->IsValueConstant(Op->Header.Args[1], &Constant2);
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsConstant1 && IsConstant2) {
|
||||
uint64_t NewConstant = (Constant1 + Constant2) & getMask(Op) ;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
else if (IsConstant2 && !IsImmAddSub(Constant2) && IsImmAddSub(-Constant2)) {
|
||||
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
|
||||
// This means we can convert the operation in to a subtract.
|
||||
// Change the IR operation itself.
|
||||
IROp->Op = OP_SUB;
|
||||
// Set the write cursor to just before this operation.
|
||||
auto CodeIter = CurrentIR.at(CodeNode);
|
||||
--CodeIter;
|
||||
IREmit->SetWriteCursor(std::get<0>(*CodeIter));
|
||||
|
||||
// Negate the constant.
|
||||
auto NegConstant = IREmit->_Constant(-Constant2);
|
||||
|
||||
// Replace the second source with the negated constant.
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Src2_Index, NegConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUB: {
|
||||
@@ -690,6 +599,21 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUBSHIFT: {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
|
||||
uint64_t Constant1, Constant2;
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(IROp->Args[1], &Constant2) &&
|
||||
Op->Shift == IR::ShiftType::LSL) {
|
||||
// Optimize the LSL case when we know both sources are constant.
|
||||
// This is a pattern that shows up with direction flag calculations if DF was set just before the operation.
|
||||
uint64_t NewConstant = (Constant1 - (Constant2 << Op->ShiftAmount)) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_AND: {
|
||||
auto Op = IROp->CW<IR::IROp_And>();
|
||||
uint64_t Constant1{};
|
||||
@@ -721,6 +645,8 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
/* TODO: restore this when we have rmif or something? */
|
||||
#if 0
|
||||
case OP_TESTNZ: {
|
||||
auto Op = IROp->CW<IR::IROp_TestNZ>();
|
||||
uint64_t Constant1{};
|
||||
@@ -735,6 +661,7 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
#endif
|
||||
case OP_OR: {
|
||||
auto Op = IROp->CW<IR::IROp_Or>();
|
||||
uint64_t Constant1{};
|
||||
@@ -993,37 +920,6 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
auto Select = IREmit->GetOpHeader(Op->Header.Args[0]);
|
||||
|
||||
uint64_t Constant;
|
||||
// Fold the select into the CondJump if possible. Could handle more complex cases, too.
|
||||
if (Op->Cond.Val == COND_NEQ && IREmit->IsValueConstant(Op->Cmp2, &Constant) && Constant == 0 && Select->Op == OP_SELECT) {
|
||||
|
||||
const auto SelectCmpClass = IREmit->WalkFindRegClass(Select->Args[0]);
|
||||
if (SelectCmpClass == GPRPairClass) {
|
||||
// If the comparison class is a GPRPair then don't fold the select since it isn't free.
|
||||
break;
|
||||
}
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Select->Args[2], &Constant1) && IREmit->IsValueConstant(Select->Args[3], &Constant2)) {
|
||||
if (Constant1 == 1 && Constant2 == 0) {
|
||||
auto slc = Select->C<IR::IROp_Select>();
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(Select->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->UnwrapNode(Select->Args[1]));
|
||||
Op->Cond = slc->Cond;
|
||||
Op->CompareSize = slc->CompareSize;
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -1090,16 +986,54 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDADDNZCV:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_TESTNZ:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
|
||||
if (IsImmLogical(Constant1, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SELECT:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
|
||||
bool Bitwise = Op->Cond == COND_ANDZ ||
|
||||
Op->Cond == COND_ANDNZ;
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
|
||||
if (Bitwise ? IsImmLogical(Constant1, IROp->Size * 8) : IsImmAddSub(Constant1)) {
|
||||
if (IsImmAddSub(Constant1)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
@@ -1130,6 +1064,33 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_NZCVSELECT:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelect>();
|
||||
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
// We always allow source 1 to be zero, but source 0 can only be a
|
||||
// special 1/~0 constant if source 1 is 0.
|
||||
uint64_t Constant0{};
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1) &&
|
||||
Constant1 == 0)
|
||||
{
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant0) &&
|
||||
(Constant0 == 1 || Constant0 == AllOnes))
|
||||
{
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, Constant0));
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
@@ -1257,6 +1218,35 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_MemCpy>();
|
||||
|
||||
uint64_t Constant{};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MEMSET:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
|
||||
uint64_t Constant{};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant & 1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -1276,8 +1266,6 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
CodeMotionAroundSelects(IREmit, CurrentIR);
|
||||
FCMPOptimization(IREmit, CurrentIR);
|
||||
LoadMemStoreMemImmediatePooling(IREmit, CurrentIR);
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
|
||||
@@ -26,7 +26,7 @@ private:
|
||||
bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DCE");
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
int NumRemoved = 0;
|
||||
bool Changed = false;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
|
||||
@@ -41,28 +41,71 @@ bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
auto [CodeNode, IROp] = CodeLast();
|
||||
|
||||
bool HasSideEffects = IR::HasSideEffects(IROp->Op);
|
||||
if (IROp->Op == OP_SYSCALL ||
|
||||
IROp->Op == OP_INLINESYSCALL) {
|
||||
FEXCore::IR::SyscallFlags Flags{};
|
||||
if (IROp->Op == OP_SYSCALL) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
else {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) == FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) {
|
||||
HasSideEffects = false;
|
||||
switch (IROp->Op) {
|
||||
case OP_SYSCALL:
|
||||
case OP_INLINESYSCALL: {
|
||||
FEXCore::IR::SyscallFlags Flags{};
|
||||
if (IROp->Op == OP_SYSCALL) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
else {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) == FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) {
|
||||
HasSideEffects = false;
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_ATOMICFETCHADD:
|
||||
case OP_ATOMICFETCHSUB:
|
||||
case OP_ATOMICFETCHAND:
|
||||
case OP_ATOMICFETCHCLR:
|
||||
case OP_ATOMICFETCHOR:
|
||||
case OP_ATOMICFETCHXOR:
|
||||
case OP_ATOMICFETCHNEG: {
|
||||
// If the result of the atomic fetch is completely unused, convert it to a non-fetching atomic operation.
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
switch (IROp->Op) {
|
||||
case OP_ATOMICFETCHADD:
|
||||
IROp->Op = OP_ATOMICADD;
|
||||
break;
|
||||
case OP_ATOMICFETCHSUB:
|
||||
IROp->Op = OP_ATOMICSUB;
|
||||
break;
|
||||
case OP_ATOMICFETCHAND:
|
||||
IROp->Op = OP_ATOMICAND;
|
||||
break;
|
||||
case OP_ATOMICFETCHCLR:
|
||||
IROp->Op = OP_ATOMICCLR;
|
||||
break;
|
||||
case OP_ATOMICFETCHOR:
|
||||
IROp->Op = OP_ATOMICOR;
|
||||
break;
|
||||
case OP_ATOMICFETCHXOR:
|
||||
IROp->Op = OP_ATOMICXOR;
|
||||
break;
|
||||
case OP_ATOMICFETCHNEG:
|
||||
IROp->Op = OP_ATOMICNEG;
|
||||
break;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
|
||||
// Skip over anything that has side effects
|
||||
// Use count tracking can't safely remove anything with side effects
|
||||
if (!HasSideEffects) {
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
NumRemoved++;
|
||||
Changed = true;
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
@@ -74,7 +117,7 @@ bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
return NumRemoved != 0;
|
||||
return Changed;
|
||||
}
|
||||
|
||||
void DeadCodeElimination::markUsed(OrderedNodeWrapper *CodeOp, IROp_Header *IROp) {
|
||||
|
||||
@@ -277,6 +277,24 @@ namespace {
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, pf_raw),
|
||||
sizeof(FEXCore::Core::CPUState::pf_raw),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, af_raw),
|
||||
sizeof(FEXCore::Core::CPUState::af_raw),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
@@ -419,6 +437,10 @@ namespace {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
// PF/AF
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
@@ -518,12 +519,21 @@ namespace {
|
||||
const auto GetRegAndClassFromOffset = [&, this](uint32_t Offset) {
|
||||
const auto beginGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0]);
|
||||
const auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[16]);
|
||||
const auto pf = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
const auto af = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
|
||||
const auto [beginFpr, endFpr] = GetFPRBeginAndEnd();
|
||||
|
||||
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr), "Unexpected Offset {}", Offset);
|
||||
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr) || (Offset == pf) || (Offset == af), "Unexpected Offset {}", Offset);
|
||||
|
||||
if (Offset >= beginGpr && Offset < endGpr) {
|
||||
unsigned FlagOffset =
|
||||
Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount - 2;
|
||||
|
||||
if (Offset == pf) {
|
||||
return PhysicalRegister(GPRFixedClass, FlagOffset);
|
||||
} else if (Offset == af) {
|
||||
return PhysicalRegister(GPRFixedClass, FlagOffset + 1);
|
||||
} else if (Offset >= beginGpr && Offset < endGpr) {
|
||||
auto reg = (Offset - beginGpr) / Core::CPUState::GPR_REG_SIZE;
|
||||
return PhysicalRegister(GPRFixedClass, reg);
|
||||
} else if (Offset >= beginFpr && Offset < endFpr) {
|
||||
@@ -544,12 +554,21 @@ namespace {
|
||||
const auto GetStaticMapFromOffset = [&](uint32_t Offset) -> LiveRange** {
|
||||
const auto beginGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0]);
|
||||
const auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[16]);
|
||||
const auto pf = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
const auto af = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
|
||||
const auto [beginFpr, endFpr] = GetFPRBeginAndEnd();
|
||||
|
||||
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr), "Unexpected Offset {}", Offset);
|
||||
LOGMAN_THROW_AA_FMT((Offset >= beginGpr && Offset < endGpr) || (Offset >= beginFpr && Offset < endFpr) || (Offset == pf) || (Offset == af), "Unexpected Offset {}", Offset);
|
||||
|
||||
if (Offset >= beginGpr && Offset < endGpr) {
|
||||
unsigned FlagOffset =
|
||||
Graph->Set.Classes[GPRFixedClass.Val].PhysicalCount - 2;
|
||||
|
||||
if (Offset == pf) {
|
||||
return &StaticMaps[FlagOffset];
|
||||
} else if (Offset == af) {
|
||||
return &StaticMaps[FlagOffset + 1];
|
||||
} else if (Offset >= beginGpr && Offset < endGpr) {
|
||||
auto reg = (Offset - beginGpr) / Core::CPUState::GPR_REG_SIZE;
|
||||
return &StaticMaps[reg];
|
||||
} else if (Offset >= beginFpr && Offset < endFpr) {
|
||||
|
||||
@@ -5,8 +5,8 @@
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -272,7 +272,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
size_t NumberOfPages = length / FHU::FEX_PAGE_SIZE;
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
|
||||
|
||||
uint64_t AllocatedOffset{};
|
||||
LiveVMARegion *LiveRegion{};
|
||||
@@ -460,7 +460,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
}
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
|
||||
|
||||
length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
@@ -585,7 +585,7 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
|
||||
OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
|
||||
|
||||
// Walk the pages and deallocate
|
||||
// First walk the live regions
|
||||
|
||||
@@ -24,24 +24,48 @@ static bool LoadFileImpl(T &Data, const fextl::string &Filepath, size_t FixedSiz
|
||||
size_t FileSize{};
|
||||
if (FixedSize == 0) {
|
||||
struct stat buf;
|
||||
if (fstat(FD, &buf) != 0) {
|
||||
close(FD);
|
||||
return false;
|
||||
if (fstat(FD, &buf) == 0) {
|
||||
FileSize = buf.st_size;
|
||||
}
|
||||
|
||||
FileSize = buf.st_size;
|
||||
}
|
||||
else {
|
||||
FileSize = FixedSize;
|
||||
}
|
||||
|
||||
ssize_t Read = -1;
|
||||
if (FileSize > 0) {
|
||||
bool LoadedFile{};
|
||||
if (FileSize) {
|
||||
// File size is known upfront
|
||||
Data.resize(FileSize);
|
||||
Read = pread(FD, &Data.at(0), FileSize, 0);
|
||||
|
||||
LoadedFile = Read == FileSize;
|
||||
}
|
||||
else {
|
||||
// The file is either empty or its size is unknown (e.g. procfs data).
|
||||
// Try reading in chunks instead
|
||||
ssize_t CurrentOffset = 0;
|
||||
constexpr size_t READ_SIZE = 4096;
|
||||
Data.resize(READ_SIZE);
|
||||
|
||||
while ((Read = pread(FD, &Data.at(CurrentOffset), READ_SIZE, CurrentOffset)) == READ_SIZE) {
|
||||
CurrentOffset += Read;
|
||||
Data.resize(CurrentOffset + Read);
|
||||
}
|
||||
|
||||
if (Read == -1) {
|
||||
Data.clear();
|
||||
close(FD);
|
||||
return false;
|
||||
}
|
||||
|
||||
// Final resize to ensure there is no garbage data past the end.
|
||||
Data.resize(CurrentOffset + Read);
|
||||
|
||||
LoadedFile = true;
|
||||
}
|
||||
close(FD);
|
||||
return Read == FileSize;
|
||||
return LoadedFile;
|
||||
}
|
||||
|
||||
ssize_t LoadFileToBuffer(const fextl::string &Filepath, std::span<char> Buffer) {
|
||||
|
||||
@@ -118,7 +118,7 @@ namespace Handler {
|
||||
}
|
||||
Begin = End + 1;
|
||||
End = View.find_first_of(',', Begin);
|
||||
Option = View.substr(Begin, End);
|
||||
Option = View.substr(Begin, End - Begin);
|
||||
}
|
||||
|
||||
return fextl::fmt::format("{}", EnumMask);
|
||||
|
||||
@@ -136,7 +136,7 @@ namespace CPU {
|
||||
[[nodiscard]] virtual CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) = 0;
|
||||
FEXCore::IR::RegisterAllocationData *RAData) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
@@ -155,13 +155,6 @@ namespace CPU {
|
||||
*/
|
||||
[[nodiscard]] virtual void *MapRegion(void *HostPtr, uint64_t GuestPtr, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief This is post-setup initialization that is called just before code executino
|
||||
*
|
||||
* Guest memory is available at this point and ThreadState is valid
|
||||
*/
|
||||
virtual void Initialize() {}
|
||||
|
||||
/**
|
||||
* @brief Lets FEXCore know if this CPUBackend needs IR and DebugData for CompileCode
|
||||
*
|
||||
|
||||
@@ -87,6 +87,11 @@ namespace FEXCore::Context {
|
||||
void *VDSO_kernel_rt_sigreturn;
|
||||
};
|
||||
|
||||
struct ThreadsState {
|
||||
FEXCore::Core::InternalThreadState* ParentThread;
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*>* Threads;
|
||||
};
|
||||
|
||||
using CodeRangeInvalidationFn = std::function<void(uint64_t start, uint64_t Length)>;
|
||||
|
||||
using CustomCPUFactoryType = std::function<fextl::unique_ptr<CPU::CPUBackend>(Context*, Core::InternalThreadState *Thread)>;
|
||||
@@ -111,16 +116,6 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY static fextl::unique_ptr<FEXCore::Context::Context> CreateNewContext();
|
||||
|
||||
/**
|
||||
* @brief Post creation context initialization
|
||||
* Once configurations have been set, do the post-creation initialization with that configuration
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
*
|
||||
* @return true if we managed to initialize correctly
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual bool InitializeContext() = 0;
|
||||
|
||||
/**
|
||||
* @brief Allows setting up in memory code and other things prior to launchign code execution
|
||||
*
|
||||
@@ -141,6 +136,18 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void Pause() = 0;
|
||||
|
||||
/**
|
||||
* @brief Waits for all threads to be idle.
|
||||
*
|
||||
* Idling can happen when the process is shutting down or the debugger has asked for all threads to pause.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void WaitForIdle() = 0;
|
||||
|
||||
/**
|
||||
* @brief When resuming from a paused state, waits for all threads to start executing before returning.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void WaitForThreadsToRun() = 0;
|
||||
|
||||
/**
|
||||
* @brief Starts (or continues) the CPU core
|
||||
*
|
||||
@@ -182,25 +189,7 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) = 0;
|
||||
|
||||
/**
|
||||
* @brief Gets the program exit status
|
||||
*
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
*
|
||||
* @return The program exit status
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual int GetProgramStatus() const = 0;
|
||||
|
||||
/**
|
||||
* @brief [[threadsafe]] Returns the ExitReason of the parent thread. Typically used for async result status
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
*
|
||||
* @return The ExitReason for the parentthread
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual ExitReason GetExitReason() = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) = 0;
|
||||
|
||||
/**
|
||||
* @brief [[theadsafe]] Checks if the Context is either done working or paused(in the case of single stepping)
|
||||
@@ -213,22 +202,6 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual bool IsDone() const = 0;
|
||||
|
||||
/**
|
||||
* @brief Gets a copy the CPUState of the parent thread
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
* @param State The state object to populate
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void GetCPUState(FEXCore::Core::CPUState *State) const = 0;
|
||||
|
||||
/**
|
||||
* @brief Copies the CPUState provided to the parent thread
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
* @param State The satate object to copy from
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetCPUState(const FEXCore::Core::CPUState *State) = 0;
|
||||
|
||||
/**
|
||||
* @brief Allows the frontend to pass in a custom CPUBackend creation factory
|
||||
*
|
||||
@@ -252,12 +225,36 @@ namespace FEXCore::Context {
|
||||
///< State reconstruction helpers
|
||||
///< Reconstructs the guest RIP from the passed in thread context and related Host PC.
|
||||
FEX_DEFAULT_VISIBILITY virtual uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) = 0;
|
||||
///< Reconstructs a compacted EFLAGS from FEX's internal EFLAG representation.
|
||||
FEX_DEFAULT_VISIBILITY virtual uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
/**
|
||||
* @brief Reconstructs a compacted EFLAGS from FEX's internal EFLAG representation.
|
||||
*
|
||||
* @param Thread The thread getting the state reconstructed
|
||||
* @param WasInJIT If the code was in the JIT at the time.
|
||||
* @param HostGPRs The host Arm64 GPRs at the point of state inside the JIT.
|
||||
* @param PSTATE The Arm64 PState value.
|
||||
*
|
||||
* If WasInJIT is false then HostGPRs and PSTATE is ignored, with the assumption that the FEX JIT has already stored all state in to the
|
||||
* ThreadState object.
|
||||
*
|
||||
* @return x86 EFLAGS reconstructed
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, bool WasInJIT, uint64_t *HostGPRs, uint64_t PSTATE) = 0;
|
||||
///< Sets FEX's internal EFLAGS representation to the passed in compacted form.
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState *Thread, uint32_t EFLAGS) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) = 0;
|
||||
/**
|
||||
* @brief Create a new thread object that doesn't inherit any state.
|
||||
* Used to create FEX thread objects in preparation for creating a true OS thread.
|
||||
*
|
||||
* @param InitialRIP The starting RIP of this thread
|
||||
* @param StackPointer The starting RSP of this thread
|
||||
* @param NewThreadState The thread state to inherit from if not nullptr.
|
||||
* @param ParentTID The thread ID that the parent is inheriting from
|
||||
*
|
||||
* @return A new InternalThreadState object for using with a new guest thread.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::Core::InternalThreadState* CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState *NewThreadState = nullptr, uint64_t ParentTID = 0) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ExecutionThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InitializeThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void RunThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
@@ -317,6 +314,14 @@ namespace FEXCore::Context {
|
||||
*
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void EnableExitOnHLT() = 0;
|
||||
|
||||
/**
|
||||
* @brief Gets the thread data for FEX's internal tracked threads.
|
||||
*
|
||||
* @return struct containing all the thread information.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual ThreadsState GetThreads() = 0;
|
||||
|
||||
private:
|
||||
};
|
||||
|
||||
|
||||
@@ -72,7 +72,7 @@ namespace FEXCore::Core {
|
||||
static_assert(std::is_trivially_copyable_v<NonAtomicRefCounter<uint64_t>>, "needs to be trivially copyable");
|
||||
static_assert(sizeof(NonAtomicRefCounter<uint64_t>) == sizeof(uint64_t), "Needs to be correct size");
|
||||
|
||||
struct FEX_PACKED CPUState {
|
||||
struct CPUState {
|
||||
// Allows more efficient handling of the register
|
||||
// file in the event AVX is not supported.
|
||||
union XMMRegs {
|
||||
@@ -102,6 +102,8 @@ namespace FEXCore::Core {
|
||||
uint64_t InlineJITBlockHeader{};
|
||||
XMMRegs xmm{};
|
||||
uint8_t flags[48]{};
|
||||
uint64_t pf_raw{};
|
||||
uint64_t af_raw{};
|
||||
uint64_t mm[8][2]{};
|
||||
|
||||
// 32bit x86 state
|
||||
@@ -335,7 +337,4 @@ namespace FEXCore::Core {
|
||||
static_assert(sizeof(CpuStateFrame::SynchronousFaultData) == 8, "This needs to be 8 bytes");
|
||||
static_assert(std::alignment_of_v<CpuStateFrame::SynchronousFaultDataStruct> == 8, "This needs to be 8 bytes");
|
||||
static_assert(offsetof(CpuStateFrame, SynchronousFaultData) % 8 == 0, "This needs to be aligned");
|
||||
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetFlagName(unsigned Flag);
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetGRegName(unsigned Reg);
|
||||
}
|
||||
@@ -36,9 +36,10 @@ class HostFeatures final {
|
||||
bool SupportsFCMA{};
|
||||
bool SupportsFlagM{};
|
||||
bool SupportsFlagM2{};
|
||||
bool SupportsRPRES{};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsFlushInputsToZero{};
|
||||
bool SupportsAFP{};
|
||||
bool SupportsFloatExceptions{};
|
||||
};
|
||||
}
|
||||
@@ -35,8 +35,6 @@ namespace Core {
|
||||
#endif
|
||||
};
|
||||
}
|
||||
using HostSignalDelegatorFunction = std::function<bool(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext)>;
|
||||
|
||||
class SignalDelegator {
|
||||
public:
|
||||
virtual ~SignalDelegator() = default;
|
||||
@@ -49,16 +47,6 @@ namespace Core {
|
||||
virtual void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
virtual void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleSignal(int Signal, void *Info, void *UContext);
|
||||
|
||||
/**
|
||||
* @brief Check to ensure the XID handler is still set to the FEX handler
|
||||
*
|
||||
@@ -67,12 +55,6 @@ namespace Core {
|
||||
*/
|
||||
virtual void CheckXIDHandler() = 0;
|
||||
|
||||
constexpr static size_t MAX_SIGNALS {64};
|
||||
|
||||
// Use the last signal just so we are less likely to ever conflict with something that the guest application is using
|
||||
// 64 is used internally by Valgrind
|
||||
constexpr static size_t SIGNAL_FOR_PAUSE {63};
|
||||
|
||||
struct SignalDelegatorConfig {
|
||||
bool StaticRegisterAllocation{};
|
||||
bool SupportsAVX{};
|
||||
@@ -124,31 +106,5 @@ namespace Core {
|
||||
|
||||
protected:
|
||||
SignalDelegatorConfig Config;
|
||||
|
||||
virtual FEXCore::Core::InternalThreadState *GetTLSThread() = 0;
|
||||
virtual void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
virtual void FrontendRegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) = 0;
|
||||
virtual void FrontendRegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) = 0;
|
||||
|
||||
private:
|
||||
struct HostSignalHandler {
|
||||
fextl::vector<FEXCore::HostSignalDelegatorFunction> Handlers{};
|
||||
FEXCore::HostSignalDelegatorFunction FrontendHandler{};
|
||||
};
|
||||
std::array<HostSignalHandler, MAX_SIGNALS + 1> HostHandlers{};
|
||||
|
||||
protected:
|
||||
void SetHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
HostHandlers[Signal].Handlers.push_back(std::move(Func));
|
||||
}
|
||||
void SetFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
HostHandlers[Signal].FrontendHandler = std::move(Func);
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -56,24 +56,24 @@ enum X86Reg : uint32_t {
|
||||
* @name RFLAG register bit locations
|
||||
* @{ */
|
||||
enum X86RegLocation : uint32_t {
|
||||
RFLAG_CF_LOC = 0,
|
||||
RFLAG_RESERVED_LOC = 1, // Reserved Bit, Read-as-1
|
||||
RFLAG_PF_LOC = 2,
|
||||
RFLAG_AF_LOC = 4,
|
||||
RFLAG_ZF_LOC = 6,
|
||||
RFLAG_SF_LOC = 7,
|
||||
RFLAG_TF_LOC = 8,
|
||||
RFLAG_IF_LOC = 9,
|
||||
RFLAG_DF_LOC = 10,
|
||||
RFLAG_OF_LOC = 11,
|
||||
RFLAG_IOPL_LOC = 12,
|
||||
RFLAG_NT_LOC = 14,
|
||||
RFLAG_RF_LOC = 16,
|
||||
RFLAG_VM_LOC = 17,
|
||||
RFLAG_AC_LOC = 18,
|
||||
RFLAG_VIF_LOC = 19,
|
||||
RFLAG_VIP_LOC = 20,
|
||||
RFLAG_ID_LOC = 21,
|
||||
RFLAG_CF_RAW_LOC = 0, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_RESERVED_LOC = 1, // Reserved Bit, Read-as-1
|
||||
RFLAG_PF_RAW_LOC = 2, // Contains multiple bits, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_AF_RAW_LOC = 4, // Contains multiple bits, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_ZF_RAW_LOC = 6, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_SF_RAW_LOC = 7, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_TF_LOC = 8,
|
||||
RFLAG_IF_LOC = 9,
|
||||
RFLAG_DF_LOC = 10,
|
||||
RFLAG_OF_RAW_LOC = 11, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_IOPL_LOC = 12,
|
||||
RFLAG_NT_LOC = 14,
|
||||
RFLAG_RF_LOC = 16,
|
||||
RFLAG_VM_LOC = 17,
|
||||
RFLAG_AC_LOC = 18,
|
||||
RFLAG_VIF_LOC = 19,
|
||||
RFLAG_VIP_LOC = 20,
|
||||
RFLAG_ID_LOC = 21,
|
||||
|
||||
// So we can implement arm64-like flag manipulaton on the x86 jit..
|
||||
// SF/ZF/CF/OF packed into a 32-bit word, matching arm64's NZCV structure (not semantics).
|
||||
|
||||
@@ -527,6 +527,12 @@ enum NamedVectorConstant : uint8_t {
|
||||
NAMED_VECTOR_PADDSUBPD_INVERT_UPPER,
|
||||
NAMED_VECTOR_MOVMSKPS_SHIFT,
|
||||
NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE,
|
||||
NAMED_VECTOR_BLENDPS_0110B,
|
||||
NAMED_VECTOR_BLENDPS_0111B,
|
||||
NAMED_VECTOR_BLENDPS_1001B,
|
||||
NAMED_VECTOR_BLENDPS_1011B,
|
||||
NAMED_VECTOR_BLENDPS_1101B,
|
||||
NAMED_VECTOR_BLENDPS_1110B,
|
||||
NAMED_VECTOR_CONST_POOL_MAX,
|
||||
// Beginning of named constants that don't have a constant pool backing.
|
||||
NAMED_VECTOR_ZERO = NAMED_VECTOR_CONST_POOL_MAX,
|
||||
@@ -541,6 +547,9 @@ enum IndexNamedVectorConstant : uint8_t {
|
||||
INDEXED_NAMED_VECTOR_PSHUFHW,
|
||||
INDEXED_NAMED_VECTOR_PSHUFD,
|
||||
INDEXED_NAMED_VECTOR_SHUFPS,
|
||||
INDEXED_NAMED_VECTOR_DPPS_MASK,
|
||||
INDEXED_NAMED_VECTOR_DPPD_MASK,
|
||||
INDEXED_NAMED_VECTOR_PBLENDW,
|
||||
INDEXED_NAMED_VECTOR_MAX,
|
||||
};
|
||||
|
||||
@@ -555,6 +564,22 @@ enum OpSize : uint8_t {
|
||||
i256Bit = 32,
|
||||
};
|
||||
|
||||
enum class FloatCompareOp : uint8_t {
|
||||
EQ = 0,
|
||||
LT,
|
||||
LE,
|
||||
UNO,
|
||||
NEQ,
|
||||
ORD,
|
||||
};
|
||||
|
||||
enum class ShiftType : uint8_t {
|
||||
LSL = 0,
|
||||
LSR,
|
||||
ASR,
|
||||
ROR,
|
||||
};
|
||||
|
||||
// Converts a size stored as an integer in to an OpSize enum.
|
||||
// This is a nop operation and will be eliminated by the compiler.
|
||||
static inline OpSize SizeToOpSize(uint8_t Size) {
|
||||
@@ -568,6 +593,7 @@ static inline OpSize SizeToOpSize(uint8_t Size) {
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
#define IROP_ENUM
|
||||
#define IROP_STRUCTS
|
||||
#define IROP_SIZES
|
||||
|
||||
@@ -27,6 +27,8 @@ friend class FEXCore::IR::PassManager;
|
||||
ResetWorkingList();
|
||||
}
|
||||
|
||||
virtual ~IREmitter() = default;
|
||||
|
||||
void ReownOrClaimBuffer() {
|
||||
DualListData.ReownOrClaimBuffer();
|
||||
}
|
||||
@@ -332,6 +334,10 @@ friend class FEXCore::IR::PassManager;
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
virtual void SaveNZCV(IROps Op) {
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
}
|
||||
|
||||
OrderedNode *CurrentWriteCursor = nullptr;
|
||||
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
|
||||
@@ -1,338 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#ifndef _WIN32
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
#ifndef _WIN32
|
||||
// Replacement for std::mutexes to deal with unlocking issues in the face of Linux fork() semantics.
|
||||
//
|
||||
// A fork() only clones the parent's calling thread. Other threads are silently dropped, which permanently leaves any mutexes owned by them locked.
|
||||
// To address this issue, ForkableUniqueMutex and ForkableSharedMutex provide a way to forcefully remove any dangling locks and reset the mutexes to their default state.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex()
|
||||
: Mutex (PTHREAD_MUTEX_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = default;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_mutex_t Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex()
|
||||
: Mutex (PTHREAD_RWLOCK_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = default;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
void lock_shared() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_rdlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
unlock();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
const auto Result = pthread_rwlock_trywrlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_RWLOCK_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_rwlock_t Mutex;
|
||||
};
|
||||
#else
|
||||
// Windows doesn't support forking, so these can be standard mutexes.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex() = default;
|
||||
|
||||
// Non-moveable
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = delete;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = delete;
|
||||
|
||||
void lock() {
|
||||
Mutex.lock();
|
||||
}
|
||||
void unlock() {
|
||||
Mutex.unlock();
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
private:
|
||||
std::mutex Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex() = default;
|
||||
|
||||
// Non-moveable
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = delete;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = delete;
|
||||
|
||||
void lock() {
|
||||
Mutex.lock();
|
||||
}
|
||||
void unlock() {
|
||||
Mutex.unlock();
|
||||
}
|
||||
void lock_shared() {
|
||||
Mutex.lock_shared();
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
Mutex.unlock_shared();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
return Mutex.try_lock();
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
return Mutex.try_lock_shared();
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
private:
|
||||
std::shared_mutex Mutex;
|
||||
};
|
||||
#endif
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedDeferredSignalWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread)
|
||||
: Mutex {&_Mutex}
|
||||
, Thread {Thread} {
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedDeferredSignalWithMutexBase(const ScopedDeferredSignalWithMutexBase&) = delete;
|
||||
ScopedDeferredSignalWithMutexBase& operator=(ScopedDeferredSignalWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedDeferredSignalWithMutexBase(ScopedDeferredSignalWithMutexBase &&rhs)
|
||||
: Mutex {rhs.Mutex}
|
||||
, Thread {rhs.Thread} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedDeferredSignalWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the recount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
using ScopedDeferredSignalWithMutex = ScopedDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedDeferredSignalWithSharedLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithUniqueLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedDeferredSignalWithForkableMutex = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedDeferredSignalWithForkableSharedLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithForkableUniqueLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
|
||||
class ScopedSignalMasker final {
|
||||
public:
|
||||
ScopedSignalMasker() = default;
|
||||
|
||||
void Mask(uint64_t Mask) {
|
||||
#ifndef _WIN32
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
#endif
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ScopedSignalMasker(const ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker(ScopedSignalMasker &&rhs) = default;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker &&) = default;
|
||||
|
||||
void Unmask() {
|
||||
#ifndef _WIN32
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
#endif
|
||||
}
|
||||
private:
|
||||
#ifndef _WIN32
|
||||
uint64_t OriginalMask{};
|
||||
#endif
|
||||
};
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedPotentialDeferredSignalWithMutexBase final {
|
||||
public:
|
||||
ScopedPotentialDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex}
|
||||
, Thread {Thread} {
|
||||
if (Thread) {
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
}
|
||||
else {
|
||||
Masker.Mask(Mask);
|
||||
}
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedPotentialDeferredSignalWithMutexBase(const ScopedPotentialDeferredSignalWithMutexBase&) = delete;
|
||||
ScopedPotentialDeferredSignalWithMutexBase& operator=(ScopedPotentialDeferredSignalWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedPotentialDeferredSignalWithMutexBase(ScopedPotentialDeferredSignalWithMutexBase &&rhs)
|
||||
: Mutex {rhs.Mutex}
|
||||
, Thread {rhs.Thread} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedPotentialDeferredSignalWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
|
||||
if (Thread) {
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the refcount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
else {
|
||||
// Unmask back to the original signal mask
|
||||
Masker.Unmask();
|
||||
}
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
ScopedSignalMasker Masker;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
using ScopedPotentialDeferredSignalWithMutex = ScopedPotentialDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithSharedLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedPotentialDeferredSignalWithForkableMutex = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithForkableSharedLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithForkableUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
}
|
||||
@@ -0,0 +1,243 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <signal.h>
|
||||
#ifndef _WIN32
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore {
|
||||
#ifndef _WIN32
|
||||
// Replacement for std::mutexes to deal with unlocking issues in the face of Linux fork() semantics.
|
||||
//
|
||||
// A fork() only clones the parent's calling thread. Other threads are silently dropped, which permanently leaves any mutexes owned by them locked.
|
||||
// To address this issue, ForkableUniqueMutex and ForkableSharedMutex provide a way to forcefully remove any dangling locks and reset the mutexes to their default state.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex()
|
||||
: Mutex (PTHREAD_MUTEX_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = default;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_mutex_t Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex()
|
||||
: Mutex (PTHREAD_RWLOCK_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = default;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
void lock_shared() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_rdlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
unlock();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
const auto Result = pthread_rwlock_trywrlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_RWLOCK_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_rwlock_t Mutex;
|
||||
};
|
||||
#else
|
||||
// Windows doesn't support forking, so these can be standard mutexes.
|
||||
class ForkableUniqueMutex final : public std::mutex {
|
||||
public:
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final : public std::shared_mutex {
|
||||
public:
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
};
|
||||
#endif
|
||||
|
||||
// Helper class to manage deferred signal refcounting within a block scope
|
||||
class DeferredSignalRefCountGuard final {
|
||||
public:
|
||||
explicit DeferredSignalRefCountGuard(FEXCore::Core::InternalThreadState *Thread) : Thread(Thread) {
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
DeferredSignalRefCountGuard(const DeferredSignalRefCountGuard&) = delete;
|
||||
DeferredSignalRefCountGuard& operator=(DeferredSignalRefCountGuard&) = delete;
|
||||
DeferredSignalRefCountGuard(DeferredSignalRefCountGuard&& rhs) : Thread(rhs.Thread) {
|
||||
rhs.Thread = nullptr;
|
||||
}
|
||||
|
||||
~DeferredSignalRefCountGuard() {
|
||||
if (Thread) {
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the refcount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
private:
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
// Helper class to mask POSIX signals within a block scope
|
||||
class ScopedSignalMasker final {
|
||||
public:
|
||||
explicit ScopedSignalMasker(uint64_t Mask) : OriginalMask(0) {
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &*OriginalMask, sizeof(*OriginalMask));
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ScopedSignalMasker(const ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker(ScopedSignalMasker&& rhs) : OriginalMask(rhs.OriginalMask) {
|
||||
rhs.OriginalMask.reset();
|
||||
}
|
||||
|
||||
~ScopedSignalMasker() {
|
||||
if (OriginalMask) {
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(*OriginalMask));
|
||||
}
|
||||
}
|
||||
private:
|
||||
std::optional<uint64_t> OriginalMask{};
|
||||
};
|
||||
#endif
|
||||
|
||||
/**
|
||||
* @brief Produces a wrapper object around a scoped lock of the given mutex
|
||||
* while ensuring POSIX signals are masked while the mutex is locked
|
||||
*
|
||||
* Use this to prevent reentrancy issues of C++ mutexes with certain signal handlers.
|
||||
* Common examples of such issues are:
|
||||
* - C++ mutexes not unlocking due to a signal handler calling longjmp from within a scope owning the mutex
|
||||
* - The signal handler itself using a mutex that would be re-locked if the handler gets invoked
|
||||
* again before unlocking
|
||||
*
|
||||
* Ownership of the returned object may be moved, but it is NOT SAFE to move across threads.
|
||||
*/
|
||||
template<template<typename> class LockType = std::unique_lock, typename MutexType>
|
||||
[[nodiscard]] static auto MaskSignalsAndLockMutex(MutexType& mutex, uint64_t Mask = ~0ULL) {
|
||||
#ifndef _WIN32
|
||||
// Signals are masked first, and then the lock is acquired
|
||||
struct {
|
||||
ScopedSignalMasker mask;
|
||||
LockType<MutexType> lock;
|
||||
} scope_guard { ScopedSignalMasker { Mask }, LockType<MutexType> { mutex } };
|
||||
return scope_guard;
|
||||
#else
|
||||
// TODO: Doesn't block signals which may or may not cause issues.
|
||||
return LockType<MutexType> { mutex };
|
||||
#endif
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Produces a wrapper object around a scoped lock of the given mutex
|
||||
* while bumping the Thread's deferred signal refcount while the mutex is
|
||||
* locked.
|
||||
*/
|
||||
template<template<typename> class LockType = std::unique_lock, typename MutexType>
|
||||
[[nodiscard]] static auto GuardSignalDeferringSection(MutexType& mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL) {
|
||||
// Refcount is incremented first, and then the lock is acquired.
|
||||
struct {
|
||||
std::optional<DeferredSignalRefCountGuard> refcount;
|
||||
LockType<MutexType> lock;
|
||||
} scope_guard = { DeferredSignalRefCountGuard { Thread }, LockType<MutexType> { mutex } };
|
||||
return scope_guard;
|
||||
}
|
||||
|
||||
// Like GuardSignalDeferringSection but falls back to masking signals when Thread is nullptr
|
||||
template<template<typename> class LockType = std::unique_lock, typename MutexType>
|
||||
[[nodiscard]] static auto GuardSignalDeferringSectionWithFallback(MutexType& mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL) {
|
||||
#ifndef _WIN32
|
||||
using ExtraGuard = std::variant<ScopedSignalMasker, DeferredSignalRefCountGuard>;
|
||||
#else
|
||||
using ExtraGuard = std::variant<std::monostate, DeferredSignalRefCountGuard>;
|
||||
#endif
|
||||
|
||||
struct {
|
||||
ExtraGuard refcount_or_mask;
|
||||
LockType<MutexType> lock;
|
||||
} scope_guard {
|
||||
Thread ? ExtraGuard { DeferredSignalRefCountGuard { Thread } }
|
||||
#ifndef _WIN32
|
||||
: ExtraGuard { ScopedSignalMasker { Mask } }
|
||||
#else
|
||||
: ExtraGuard { }
|
||||
#endif
|
||||
};
|
||||
scope_guard.lock = LockType<MutexType> { mutex };
|
||||
return scope_guard;
|
||||
}
|
||||
}
|
||||
File renamed without changes.
@@ -0,0 +1,22 @@
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <catch2/catch.hpp>
|
||||
|
||||
TEST_CASE("LoadFile-Doesn'tExist") {
|
||||
fextl::string MapsFile;
|
||||
auto Read = FEXCore::FileLoading::LoadFile(MapsFile, "/tmp/a/b/c/d/e/z");
|
||||
REQUIRE(MapsFile.size() == 0);
|
||||
REQUIRE(Read == false);
|
||||
}
|
||||
|
||||
TEST_CASE("LoadFile-procfs") {
|
||||
fextl::string MapsFile;
|
||||
FEXCore::FileLoading::LoadFile(MapsFile, "/proc/self/maps");
|
||||
REQUIRE(MapsFile.size() != 0);
|
||||
}
|
||||
|
||||
TEST_CASE("LoadFile-Buffer") {
|
||||
fextl::string MapsFile;
|
||||
MapsFile.resize(16);
|
||||
auto Read = FEXCore::FileLoading::LoadFileToBuffer("/proc/self/maps", MapsFile);
|
||||
REQUIRE(MapsFile.size() == Read);
|
||||
}
|
||||
@@ -1709,6 +1709,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Evaluate into flags") {
|
||||
TEST_SINGLE(setf8(WReg::w30), "setf8 w30");
|
||||
TEST_SINGLE(setf16(WReg::w30), "setf16 w30");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Carry flag invert") {
|
||||
TEST_SINGLE(cfinv(), "cfinv");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Arm to eXternal FLAG") {
|
||||
TEST_SINGLE(axflag(), "axflag");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: eXternal to Arm FLAG") {
|
||||
TEST_SINGLE(xaflag(), "xaflag");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Conditional compare - register") {
|
||||
TEST_SINGLE(ccmn(Size::i32Bit, Reg::r29, Reg::r28, StatusFlags::None, Condition::CC_AL), "ccmn w29, w28, #nzcv, al");
|
||||
TEST_SINGLE(ccmn(Size::i32Bit, Reg::r29, Reg::r28, StatusFlags::Flag_N, Condition::CC_AL), "ccmn w29, w28, #Nzcv, al");
|
||||
|
||||
@@ -1805,175 +1805,175 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE predicate read from FFR (u
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE predicate initialize") {
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.b, pow2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.h, pow2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.s, pow2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.d, pow2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.b, pow2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.h, pow2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.s, pow2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.d, pow2");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.b, pow2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.h, pow2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.s, pow2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.d, pow2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.b, pow2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.h, pow2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.s, pow2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.d, pow2");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.b, vl1");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.h, vl1");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.s, vl1");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.d, vl1");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.b, vl1");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.h, vl1");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.s, vl1");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.d, vl1");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.b, vl1");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.h, vl1");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.s, vl1");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.d, vl1");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.b, vl1");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.h, vl1");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.s, vl1");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.d, vl1");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.b, vl2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.h, vl2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.s, vl2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.d, vl2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.b, vl2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.h, vl2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.s, vl2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.d, vl2");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.b, vl2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.h, vl2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.s, vl2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.d, vl2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.b, vl2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.h, vl2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.s, vl2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.d, vl2");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.b, vl3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.h, vl3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.s, vl3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.d, vl3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.b, vl3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.h, vl3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.s, vl3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.d, vl3");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.b, vl3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.h, vl3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.s, vl3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.d, vl3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.b, vl3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.h, vl3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.s, vl3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.d, vl3");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.b, vl4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.h, vl4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.s, vl4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.d, vl4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.b, vl4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.h, vl4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.s, vl4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.d, vl4");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.b, vl4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.h, vl4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.s, vl4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.d, vl4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.b, vl4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.h, vl4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.s, vl4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.d, vl4");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.b, vl5");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.h, vl5");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.s, vl5");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.d, vl5");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.b, vl5");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.h, vl5");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.s, vl5");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.d, vl5");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.b, vl5");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.h, vl5");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.s, vl5");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.d, vl5");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.b, vl5");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.h, vl5");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.s, vl5");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.d, vl5");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.b, vl6");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.h, vl6");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.s, vl6");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.d, vl6");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.b, vl6");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.h, vl6");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.s, vl6");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.d, vl6");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.b, vl6");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.h, vl6");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.s, vl6");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.d, vl6");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.b, vl6");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.h, vl6");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.s, vl6");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.d, vl6");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.b, vl7");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.h, vl7");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.s, vl7");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.d, vl7");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.b, vl7");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.h, vl7");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.s, vl7");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.d, vl7");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.b, vl7");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.h, vl7");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.s, vl7");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.d, vl7");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.b, vl7");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.h, vl7");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.s, vl7");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.d, vl7");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.b, vl8");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.h, vl8");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.s, vl8");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.d, vl8");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.b, vl8");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.h, vl8");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.s, vl8");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.d, vl8");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.b, vl8");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.h, vl8");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.s, vl8");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.d, vl8");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.b, vl8");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.h, vl8");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.s, vl8");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.d, vl8");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.b, vl16");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.h, vl16");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.s, vl16");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.d, vl16");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.b, vl16");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.h, vl16");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.s, vl16");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.d, vl16");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.b, vl16");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.h, vl16");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.s, vl16");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.d, vl16");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.b, vl16");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.h, vl16");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.s, vl16");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.d, vl16");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.b, vl32");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.h, vl32");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.s, vl32");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.d, vl32");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.b, vl32");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.h, vl32");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.s, vl32");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.d, vl32");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.b, vl32");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.h, vl32");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.s, vl32");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.d, vl32");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.b, vl32");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.h, vl32");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.s, vl32");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.d, vl32");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.b, vl64");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.h, vl64");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.s, vl64");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.d, vl64");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.b, vl64");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.h, vl64");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.s, vl64");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.d, vl64");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.b, vl64");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.h, vl64");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.s, vl64");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.d, vl64");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.b, vl64");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.h, vl64");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.s, vl64");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.d, vl64");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.b, vl128");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.h, vl128");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.s, vl128");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.d, vl128");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.b, vl128");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.h, vl128");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.s, vl128");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.d, vl128");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.b, vl128");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.h, vl128");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.s, vl128");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.d, vl128");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.b, vl128");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.h, vl128");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.s, vl128");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.d, vl128");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.b, vl256");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.h, vl256");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.s, vl256");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.d, vl256");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.b, vl256");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.h, vl256");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.s, vl256");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.d, vl256");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.b, vl256");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.h, vl256");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.s, vl256");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.d, vl256");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.b, vl256");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.h, vl256");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.s, vl256");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.d, vl256");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.b, mul4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.h, mul4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.s, mul4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.d, mul4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.b, mul4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.h, mul4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.s, mul4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.d, mul4");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.b, mul4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.h, mul4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.s, mul4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.d, mul4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.b, mul4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.h, mul4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.s, mul4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.d, mul4");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.b, mul3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.h, mul3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.s, mul3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.d, mul3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.b, mul3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.h, mul3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.s, mul3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.d, mul3");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.b, mul3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.h, mul3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.s, mul3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.d, mul3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.b, mul3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.h, mul3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.s, mul3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.d, mul3");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.b");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.h");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.s");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.d");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.b");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.h");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.s");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.d");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.b");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.h");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.s");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.d");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.b");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.h");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.s");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.d");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE integer compare scalar count and limit") {
|
||||
|
||||
@@ -687,6 +687,53 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point data-process
|
||||
TEST_SINGLE(frintx(HReg::h30, HReg::h29), "frintx h30, h29");
|
||||
TEST_SINGLE(frinti(HReg::h30, HReg::h29), "frinti h30, h29");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point data-processing (1 source sized)") {
|
||||
TEST_SINGLE(fmov(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "fmov s30, s29");
|
||||
TEST_SINGLE(fabs(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "fabs s30, s29");
|
||||
TEST_SINGLE(fneg(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "fneg s30, s29");
|
||||
TEST_SINGLE(fsqrt(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "fsqrt s30, s29");
|
||||
TEST_SINGLE(frintn(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintn s30, s29");
|
||||
TEST_SINGLE(frintp(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintp s30, s29");
|
||||
TEST_SINGLE(frintm(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintm s30, s29");
|
||||
TEST_SINGLE(frintz(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintz s30, s29");
|
||||
TEST_SINGLE(frinta(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frinta s30, s29");
|
||||
TEST_SINGLE(frintx(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintx s30, s29");
|
||||
TEST_SINGLE(frinti(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frinti s30, s29");
|
||||
TEST_SINGLE(frint32z(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frint32z s30, s29");
|
||||
TEST_SINGLE(frint32x(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frint32x s30, s29");
|
||||
TEST_SINGLE(frint64z(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frint64z s30, s29");
|
||||
TEST_SINGLE(frint64x(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frint64x s30, s29");
|
||||
|
||||
TEST_SINGLE(fmov(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "fmov d30, d29");
|
||||
TEST_SINGLE(fabs(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "fabs d30, d29");
|
||||
TEST_SINGLE(fneg(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "fneg d30, d29");
|
||||
TEST_SINGLE(fsqrt(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "fsqrt d30, d29");
|
||||
TEST_SINGLE(frintn(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintn d30, d29");
|
||||
TEST_SINGLE(frintp(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintp d30, d29");
|
||||
TEST_SINGLE(frintm(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintm d30, d29");
|
||||
TEST_SINGLE(frintz(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintz d30, d29");
|
||||
TEST_SINGLE(frinta(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frinta d30, d29");
|
||||
TEST_SINGLE(frintx(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintx d30, d29");
|
||||
TEST_SINGLE(frinti(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frinti d30, d29");
|
||||
TEST_SINGLE(frint32z(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frint32z d30, d29");
|
||||
TEST_SINGLE(frint32x(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frint32x d30, d29");
|
||||
TEST_SINGLE(frint64z(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frint64z d30, d29");
|
||||
TEST_SINGLE(frint64x(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frint64x d30, d29");
|
||||
|
||||
TEST_SINGLE(fmov(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "fmov h30, h29");
|
||||
TEST_SINGLE(fabs(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "fabs h30, h29");
|
||||
TEST_SINGLE(fneg(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "fneg h30, h29");
|
||||
TEST_SINGLE(fsqrt(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "fsqrt h30, h29");
|
||||
TEST_SINGLE(frintn(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintn h30, h29");
|
||||
TEST_SINGLE(frintp(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintp h30, h29");
|
||||
TEST_SINGLE(frintm(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintm h30, h29");
|
||||
TEST_SINGLE(frintz(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintz h30, h29");
|
||||
TEST_SINGLE(frinta(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frinta h30, h29");
|
||||
TEST_SINGLE(frintx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintx h30, h29");
|
||||
TEST_SINGLE(frinti(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frinti h30, h29");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point compare") {
|
||||
// Commented out lines showcase unallocated encodings.
|
||||
//TEST_SINGLE(fcmp(ScalarRegSize::i8Bit, VReg::v30, VReg::v29), "fcmp b30, b29");
|
||||
@@ -824,6 +871,39 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point data-process
|
||||
TEST_SINGLE(fminnm(HReg::h30, HReg::h29, HReg::h28), "fminnm h30, h29, h28");
|
||||
TEST_SINGLE(fnmul(HReg::h30, HReg::h29, HReg::h28), "fnmul h30, h29, h28");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point data-processing (2 source sized)") {
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fmul s30, s29, s28");
|
||||
TEST_SINGLE(fdiv(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fdiv s30, s29, s28");
|
||||
TEST_SINGLE(fadd(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fadd s30, s29, s28");
|
||||
TEST_SINGLE(fsub(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fsub s30, s29, s28");
|
||||
TEST_SINGLE(fmax(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fmax s30, s29, s28");
|
||||
TEST_SINGLE(fmin(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fmin s30, s29, s28");
|
||||
TEST_SINGLE(fmaxnm(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fmaxnm s30, s29, s28");
|
||||
TEST_SINGLE(fminnm(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fminnm s30, s29, s28");
|
||||
TEST_SINGLE(fnmul(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fnmul s30, s29, s28");
|
||||
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fmul d30, d29, d28");
|
||||
TEST_SINGLE(fdiv(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fdiv d30, d29, d28");
|
||||
TEST_SINGLE(fadd(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fadd d30, d29, d28");
|
||||
TEST_SINGLE(fsub(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fsub d30, d29, d28");
|
||||
TEST_SINGLE(fmax(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fmax d30, d29, d28");
|
||||
TEST_SINGLE(fmin(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fmin d30, d29, d28");
|
||||
TEST_SINGLE(fmaxnm(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fmaxnm d30, d29, d28");
|
||||
TEST_SINGLE(fminnm(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fminnm d30, d29, d28");
|
||||
TEST_SINGLE(fnmul(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fnmul d30, d29, d28");
|
||||
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fmul h30, h29, h28");
|
||||
TEST_SINGLE(fdiv(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fdiv h30, h29, h28");
|
||||
TEST_SINGLE(fadd(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fadd h30, h29, h28");
|
||||
TEST_SINGLE(fsub(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fsub h30, h29, h28");
|
||||
TEST_SINGLE(fmax(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fmax h30, h29, h28");
|
||||
TEST_SINGLE(fmin(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fmin h30, h29, h28");
|
||||
TEST_SINGLE(fmaxnm(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fmaxnm h30, h29, h28");
|
||||
TEST_SINGLE(fminnm(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fminnm h30, h29, h28");
|
||||
TEST_SINGLE(fnmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fnmul h30, h29, h28");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point conditional select") {
|
||||
TEST_SINGLE(fcsel(SReg::s30, SReg::s29, SReg::s28, Condition::CC_AL), "fcsel s30, s29, s28, al");
|
||||
TEST_SINGLE(fcsel(SReg::s30, SReg::s29, SReg::s28, Condition::CC_EQ), "fcsel s30, s29, s28, eq");
|
||||
|
||||
@@ -1,123 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#ifndef _WIN32
|
||||
#include <signal.h>
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FHU {
|
||||
/**
|
||||
* @brief A drop-in replacement for std::lock_guard that masks POSIX signals while the mutex is locked
|
||||
*
|
||||
* Use this class to prevent reentrancy issues of C++ mutexes with certain signal handlers.
|
||||
* Common examples of such issues are:
|
||||
* - C++ mutexes not unlocking due to a signal handler longjmping out of a scope owning the mutex
|
||||
* - The signal handler itself using a mutex that would be re-locked if the handler gets invoked
|
||||
* again before unlocking
|
||||
*
|
||||
* Ownership of this object may be moved, but it is NOT SAFE to move across threads.
|
||||
*
|
||||
* Constructor order:
|
||||
* 1) Mask signals
|
||||
* 2) Lock Mutex
|
||||
*
|
||||
* Destructor Order:
|
||||
* 1) Unlock Mutex
|
||||
* 2) Unmask signals
|
||||
*/
|
||||
#ifndef _WIN32
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedSignalMaskWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedSignalMaskWithMutexBase(MutexType &_Mutex, uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex} {
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedSignalMaskWithMutexBase(const ScopedSignalMaskWithMutexBase&) = delete;
|
||||
ScopedSignalMaskWithMutexBase& operator=(ScopedSignalMaskWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedSignalMaskWithMutexBase(ScopedSignalMaskWithMutexBase &&rhs)
|
||||
: OriginalMask {rhs.OriginalMask}, Mutex {rhs.Mutex} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedSignalMaskWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
|
||||
// Unmask back to the original signal mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
}
|
||||
}
|
||||
private:
|
||||
uint64_t OriginalMask{};
|
||||
MutexType *Mutex;
|
||||
};
|
||||
#else
|
||||
// TODO: Doesn't block signals which may or may not cause issues.
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedSignalMaskWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedSignalMaskWithMutexBase(MutexType &_Mutex, [[maybe_unused]] uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex} {
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
}
|
||||
|
||||
// No copy or assignment possible
|
||||
ScopedSignalMaskWithMutexBase(const ScopedSignalMaskWithMutexBase&) = delete;
|
||||
ScopedSignalMaskWithMutexBase& operator=(ScopedSignalMaskWithMutexBase&) = delete;
|
||||
|
||||
// Only move
|
||||
ScopedSignalMaskWithMutexBase(ScopedSignalMaskWithMutexBase &&rhs)
|
||||
: Mutex {rhs.Mutex} {
|
||||
rhs.Mutex = nullptr;
|
||||
}
|
||||
|
||||
~ScopedSignalMaskWithMutexBase() {
|
||||
if (Mutex != nullptr) {
|
||||
// Unlock the mutex
|
||||
(Mutex->*unlock_fn)();
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
using ScopedSignalMaskWithMutex = ScopedSignalMaskWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedSignalMaskWithSharedLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedSignalMaskWithUniqueLock = ScopedSignalMaskWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
using ScopedSignalMaskWithForkableMutex = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedSignalMaskWithForkableSharedLock = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedSignalMaskWithForkableUniqueLock = ScopedSignalMaskWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
}
|
||||
@@ -14,23 +14,19 @@ logger.setLevel(logging.ERROR)
|
||||
@dataclass
|
||||
class TestData:
|
||||
name: str
|
||||
optimal: int
|
||||
expectedinstructioncount: int
|
||||
code: bytes
|
||||
def __init__(self, Name, Optimal, ExpectedInstructionCount, Code):
|
||||
instructions: list
|
||||
def __init__(self, Name, ExpectedInstructionCount, Code, Instructions):
|
||||
self.name = Name
|
||||
self.expectedinstructioncount = ExpectedInstructionCount
|
||||
self.optimal = Optimal
|
||||
self.code = Code
|
||||
self.instructions = Instructions
|
||||
|
||||
@property
|
||||
def Name(self):
|
||||
return self.name
|
||||
|
||||
@property
|
||||
def Optimal(self):
|
||||
return self.optimal
|
||||
|
||||
@property
|
||||
def ExpectedInstructionCount(self):
|
||||
return self.expectedinstructioncount
|
||||
@@ -39,6 +35,10 @@ class TestData:
|
||||
def Code(self):
|
||||
return self.code
|
||||
|
||||
@property
|
||||
def Instructions(self):
|
||||
return self.instructions
|
||||
|
||||
TestDataMap = {}
|
||||
class HostFeatures(Flag) :
|
||||
FEATURE_ANY = 0
|
||||
@@ -48,7 +48,11 @@ class HostFeatures(Flag) :
|
||||
FEATURE_RNG = (1 << 3)
|
||||
FEATURE_FCMA = (1 << 4)
|
||||
FEATURE_CSSC = (1 << 5)
|
||||
|
||||
FEATURE_AFP = (1 << 6)
|
||||
FEATURE_RPRES = (1 << 7)
|
||||
FEATURE_FLAGM = (1 << 8)
|
||||
FEATURE_FLAGM2 = (1 << 9)
|
||||
FEATURE_CRYPTO = (1 << 10)
|
||||
|
||||
HostFeaturesLookup = {
|
||||
"SVE128" : HostFeatures.FEATURE_SVE128,
|
||||
@@ -57,6 +61,11 @@ HostFeaturesLookup = {
|
||||
"RNG" : HostFeatures.FEATURE_RNG,
|
||||
"FCMA" : HostFeatures.FEATURE_FCMA,
|
||||
"CSSC" : HostFeatures.FEATURE_CSSC,
|
||||
"AFP" : HostFeatures.FEATURE_AFP,
|
||||
"RPRES" : HostFeatures.FEATURE_RPRES,
|
||||
"FLAGM" : HostFeatures.FEATURE_FLAGM,
|
||||
"FLAGM2" : HostFeatures.FEATURE_FLAGM2,
|
||||
"CRYPTO" : HostFeatures.FEATURE_CRYPTO,
|
||||
}
|
||||
|
||||
def GetHostFeatures(data):
|
||||
@@ -72,7 +81,7 @@ def GetHostFeatures(data):
|
||||
HostFeaturesData |= HostFeaturesLookup[data_key]
|
||||
return HostFeaturesData
|
||||
|
||||
def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
def parse_json_data(json_filepath, json_filename, json_data, output_binary_path):
|
||||
Bitness = 64
|
||||
EnabledHostFeatures = HostFeatures.FEATURE_ANY
|
||||
DisabledHostFeatures = HostFeatures.FEATURE_ANY
|
||||
@@ -99,26 +108,31 @@ def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
|
||||
for key, items in json_data["Instructions"].items():
|
||||
ExpectedInstructionCount = 0
|
||||
Optimal = 0
|
||||
Instructions = []
|
||||
if ("ExpectedInstructionCount" in items):
|
||||
ExpectedInstructionCount = int(items["ExpectedInstructionCount"])
|
||||
|
||||
if ("Optimal" in items):
|
||||
if items["Optimal"].upper() == "YES":
|
||||
Optimal = 1
|
||||
|
||||
if ("Skip" in items):
|
||||
if items["Skip"].upper() == "YES":
|
||||
continue
|
||||
|
||||
TestName = base64.b64encode("{}.{}".format(json_filename, key).encode("ascii")).decode("ascii")
|
||||
if "x86Insts" in items:
|
||||
Instructions = items["x86Insts"]
|
||||
else:
|
||||
# No list of instructions, only one which is the key.
|
||||
Instructions.append(key)
|
||||
TestName = base64.b64encode("{}.{}.{}".format(str(hash(json_filepath)), json_filename, key).encode("ascii")).decode("ascii")
|
||||
tmp_asm = "/tmp/{}.asm".format(TestName)
|
||||
tmp_asm_out = "/tmp/{}.asm.o".format(TestName)
|
||||
logging.info("'{}' -> '{}' -> '{}'".format(key, tmp_asm, tmp_asm_out))
|
||||
|
||||
if TestName in TestDataMap:
|
||||
sys.exit("Duplicate test name {} in tests".format(TestName))
|
||||
|
||||
with open(tmp_asm, "w") as tmp_asm_file:
|
||||
tmp_asm_file.write("BITS {};\n".format(Bitness))
|
||||
tmp_asm_file.write("{}\n".format(key))
|
||||
for Inst in Instructions:
|
||||
tmp_asm_file.write("{}\n".format(Inst))
|
||||
|
||||
Process = subprocess.Popen(["nasm", tmp_asm, "-o", tmp_asm_out])
|
||||
Process.wait()
|
||||
@@ -140,7 +154,7 @@ def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
with open(tmp_asm_out, "rb") as tmp_asm_out_file:
|
||||
binary_hex = tmp_asm_out_file.read()
|
||||
|
||||
TestDataMap[TestName] = TestData(key, Optimal, ExpectedInstructionCount, binary_hex)
|
||||
TestDataMap[TestName] = TestData(key, ExpectedInstructionCount, binary_hex, Instructions)
|
||||
|
||||
os.remove(tmp_asm)
|
||||
os.remove(tmp_asm_out)
|
||||
@@ -158,9 +172,9 @@ def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
# };
|
||||
# struct TestInfo {
|
||||
# char InstName[128];
|
||||
# uint64_t Optimal;
|
||||
# int64_t ExpectedInstructionCount;
|
||||
# uint64_t CodeSize;
|
||||
# uint64_t x86InstCount;
|
||||
# uint32_t Cookie;
|
||||
# uint8_t Code[CodeSize];
|
||||
# };
|
||||
@@ -184,9 +198,9 @@ def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
# Add each test
|
||||
for key, item in TestDataMap.items():
|
||||
MemData += struct.pack('128s', item.Name.encode("ascii"))
|
||||
MemData += struct.pack('Q', item.Optimal)
|
||||
MemData += struct.pack('q', item.ExpectedInstructionCount)
|
||||
MemData += struct.pack('Q', len(item.Code))
|
||||
MemData += struct.pack('Q', len(item.Instructions))
|
||||
MemData += struct.pack('I', 0x41424344)
|
||||
MemData += item.Code
|
||||
|
||||
@@ -218,7 +232,7 @@ def main():
|
||||
if not isinstance(json_data, dict):
|
||||
raise TypeError('JSON data must be a dict')
|
||||
|
||||
return parse_json_data(os.path.basename(json_path), json_data, output_binary_path)
|
||||
return parse_json_data(json_path, os.path.basename(json_path), json_data, output_binary_path)
|
||||
|
||||
except ValueError as ve:
|
||||
logging.error(f'JSON error: {ve}')
|
||||
|
||||
@@ -40,18 +40,17 @@ class Regs(Flag):
|
||||
REG_XMM15 = (1 << 32)
|
||||
REG_GS = (1 << 33)
|
||||
REG_FS = (1 << 34)
|
||||
REG_FLAGS = (1 << 35)
|
||||
REG_MM0 = (1 << 36)
|
||||
REG_MM1 = (1 << 37)
|
||||
REG_MM2 = (1 << 38)
|
||||
REG_MM3 = (1 << 39)
|
||||
REG_MM4 = (1 << 40)
|
||||
REG_MM5 = (1 << 41)
|
||||
REG_MM6 = (1 << 42)
|
||||
REG_MM7 = (1 << 43)
|
||||
REG_MM8 = (1 << 44)
|
||||
REG_ALL = (1 << 45) - 1
|
||||
REG_INVALID = (1 << 45)
|
||||
REG_MM0 = (1 << 35)
|
||||
REG_MM1 = (1 << 36)
|
||||
REG_MM2 = (1 << 37)
|
||||
REG_MM3 = (1 << 38)
|
||||
REG_MM4 = (1 << 39)
|
||||
REG_MM5 = (1 << 40)
|
||||
REG_MM6 = (1 << 41)
|
||||
REG_MM7 = (1 << 42)
|
||||
REG_MM8 = (1 << 43)
|
||||
REG_ALL = (1 << 44) - 1
|
||||
REG_INVALID = (1 << 44)
|
||||
|
||||
class ABI(Flag) :
|
||||
ABI_SYSTEMV = 0
|
||||
@@ -113,7 +112,6 @@ RegStringLookup = {
|
||||
"XMM15": Regs.REG_XMM15,
|
||||
"GS": Regs.REG_GS,
|
||||
"FS": Regs.REG_FS,
|
||||
"FLAGS": Regs.REG_FLAGS,
|
||||
"ALL": Regs.REG_ALL,
|
||||
"MM0": Regs.REG_MM0,
|
||||
"MM1": Regs.REG_MM1,
|
||||
|
||||
@@ -104,6 +104,17 @@ namespace FEXServerClient {
|
||||
return GetServerLockFolder() + "RootFS.lock";
|
||||
}
|
||||
|
||||
fextl::string GetTempFolder() {
|
||||
auto XDGRuntimeEnv = getenv("XDG_RUNTIME_DIR");
|
||||
if (XDGRuntimeEnv) {
|
||||
// If the XDG runtime directory works then use that.
|
||||
return XDGRuntimeEnv;
|
||||
}
|
||||
// Fallback to `/tmp/` if XDG_RUNTIME_DIR doesn't exist.
|
||||
// Might not be ideal but we don't have much of a choice.
|
||||
return fextl::string{std::filesystem::temp_directory_path().string()};
|
||||
}
|
||||
|
||||
fextl::string GetServerMountFolder() {
|
||||
// We need a FEXServer mount directory that has some tricky requirements.
|
||||
// - We don't want to use `/tmp/` if possible.
|
||||
@@ -119,17 +130,7 @@ namespace FEXServerClient {
|
||||
// - If this path doesn't exist then fallback to `/tmp/` as a last resort.
|
||||
// - pressure-vessel explicitly creates an internal XDG_RUNTIME_DIR inside its chroot.
|
||||
// - This is okay since pressure-vessel rbinds the FEX rootfs from the host to `/run/pressure-vessel/interpreter-root`.
|
||||
fextl::string Folder{};
|
||||
auto XDGRuntimeEnv = getenv("XDG_RUNTIME_DIR");
|
||||
if (XDGRuntimeEnv) {
|
||||
// If the XDG runtime directory works then use that.
|
||||
Folder = XDGRuntimeEnv;
|
||||
}
|
||||
else {
|
||||
// Fallback to `/tmp/` if XDG_RUNTIME_DIR doesn't exist.
|
||||
// Might not be ideal but we don't have much of a choice.
|
||||
Folder = std::filesystem::temp_directory_path().string();
|
||||
}
|
||||
auto Folder = GetTempFolder();
|
||||
|
||||
if (FEXCore::Config::FindContainer() == "pressure-vessel") {
|
||||
// In pressure-vessel the mount point changes location.
|
||||
|
||||
@@ -50,6 +50,7 @@ namespace FEXServerClient {
|
||||
fextl::string GetServerLockFolder();
|
||||
fextl::string GetServerLockFile();
|
||||
fextl::string GetServerRootFSLockFile();
|
||||
fextl::string GetTempFolder();
|
||||
fextl::string GetServerMountFolder();
|
||||
fextl::string GetServerSocketName();
|
||||
int GetServerFD();
|
||||
|
||||
@@ -227,9 +227,9 @@ void AssertHandler(char const *Message) {
|
||||
|
||||
struct TestInfo {
|
||||
char TestInst[128];
|
||||
uint64_t Optimal;
|
||||
int64_t ExpectedInstructionCount;
|
||||
uint64_t CodeSize;
|
||||
uint64_t x86InstCount;
|
||||
uint32_t Cookie;
|
||||
uint8_t Code[];
|
||||
};
|
||||
@@ -258,7 +258,7 @@ static bool TestInstructions(FEXCore::Context::Context *CTX, FEXCore::Core::Inte
|
||||
LogMan::Msg::IFmt("Compiling instruction '{}'", CurrentTest->TestInst);
|
||||
|
||||
// Compile the INST.
|
||||
CTX->CompileRIP(Thread, CodeRIP);
|
||||
CTX->CompileRIPCount(Thread, CodeRIP, CurrentTest->x86InstCount);
|
||||
|
||||
// Go to the next test.
|
||||
CurrentTest = reinterpret_cast<TestInfo const*>(&CurrentTest->Code[CurrentTest->CodeSize]);
|
||||
@@ -275,8 +275,8 @@ static bool TestInstructions(FEXCore::Context::Context *CTX, FEXCore::Core::Inte
|
||||
|
||||
LogMan::Msg::IFmt("Testing instruction '{}': {} host instructions", CurrentTest->TestInst, INSTStats->first.HostCodeInstructions);
|
||||
|
||||
// Show the code if we know the implementation isn't optimal or if the count of instructions changed to something we didn't expect.
|
||||
bool ShouldShowCode = CurrentTest->Optimal == 0 ||
|
||||
// Show the code if the count of instructions changed to something we didn't expect.
|
||||
bool ShouldShowCode =
|
||||
INSTStats->first.HostCodeInstructions != CurrentTest->ExpectedInstructionCount;
|
||||
|
||||
if (ShouldShowCode) {
|
||||
@@ -462,6 +462,11 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEATURE_RNG = (1U << 3),
|
||||
FEATURE_FCMA = (1U << 4),
|
||||
FEATURE_CSSC = (1U << 5),
|
||||
FEATURE_AFP = (1U << 6),
|
||||
FEATURE_RPRES = (1U << 7),
|
||||
FEATURE_FLAGM = (1U << 8),
|
||||
FEATURE_FLAGM2 = (1U << 9),
|
||||
FEATURE_CRYPTO = (1U << 10),
|
||||
};
|
||||
|
||||
uint64_t SVEWidth = 0;
|
||||
@@ -486,6 +491,21 @@ int main(int argc, char **argv, char **const envp) {
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_CSSC) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLECSSC);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_AFP) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEAFP);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_RPRES) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLERPRES);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_FLAGM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEFLAGM);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_FLAGM2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEFLAGM2);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_CRYPTO) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLECRYPTO);
|
||||
}
|
||||
|
||||
// Always enable ARMv8.1 LSE atomics.
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEATOMICS);
|
||||
@@ -508,6 +528,22 @@ int main(int argc, char **argv, char **const envp) {
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_CSSC) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLECSSC);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_AFP) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEAFP);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_RPRES) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLERPRES);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_FLAGM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEFLAGM);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_FLAGM2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEFLAGM2);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_CRYPTO) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLECRYPTO);
|
||||
}
|
||||
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_HOSTFEATURES, fextl::fmt::format("{}", HostFeatureControl));
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_FORCESVEWIDTH, fextl::fmt::format("{}", SVEWidth));
|
||||
|
||||
@@ -517,7 +553,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
// Create FEXCore context.
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext();
|
||||
|
||||
CTX->InitializeContext();
|
||||
auto SignalDelegation = FEX::DummyHandlers::CreateSignalDelegator();
|
||||
auto SyscallHandler = FEX::DummyHandlers::CreateSyscallHandler();
|
||||
|
||||
|
||||
@@ -34,21 +34,11 @@ class DummySignalDelegator final : public FEXCore::SignalDelegator, public FEXCo
|
||||
}
|
||||
|
||||
protected:
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *Info, void *UContext) override {}
|
||||
|
||||
void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread() override;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void FrontendRegisterHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override {}
|
||||
void FrontendRegisterFrontendHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override {}
|
||||
private:
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread();
|
||||
};
|
||||
|
||||
fextl::unique_ptr<FEXCore::HLE::SyscallHandler> CreateSyscallHandler();
|
||||
|
||||
@@ -106,8 +106,7 @@ void AOTGenSection(FEXCore::Context::Context *CTX, ELFCodeLoader::LoadedSection
|
||||
setpriority(PRIO_PROCESS, FHU::Syscalls::gettid(), 19);
|
||||
|
||||
// Setup thread - Each compilation thread uses its own backing FEX thread
|
||||
FEXCore::Core::CPUState state;
|
||||
auto Thread = CTX->CreateThread(&state, FHU::Syscalls::gettid());
|
||||
auto Thread = CTX->CreateThread(0, 0);
|
||||
fextl::set<uint64_t> ExternalBranchesLocal;
|
||||
CTX->ConfigureAOTGen(Thread, &ExternalBranchesLocal, SectionMaxAddress);
|
||||
|
||||
|
||||
@@ -143,6 +143,10 @@ static inline uint64_t GetArmReg(void* ucontext, uint32_t id) {
|
||||
return GetMContext(ucontext)->regs[id];
|
||||
}
|
||||
|
||||
static inline uint64_t GetArmPState(void* ucontext) {
|
||||
return GetMContext(ucontext)->pstate;
|
||||
}
|
||||
|
||||
static inline uint64_t *GetArmGPRs(void* ucontext) {
|
||||
return reinterpret_cast<uint64_t*>(GetMContext(ucontext)->regs);
|
||||
}
|
||||
@@ -313,6 +317,10 @@ static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
static inline uint64_t GetArmPState(void* ucontext) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
static inline uint64_t *GetArmGPRs(void* ucontext) {
|
||||
ERROR_AND_DIE_FMT("Not implemented for x86 host");
|
||||
}
|
||||
|
||||
@@ -37,6 +37,7 @@ if (NOT MINGW_BUILD)
|
||||
${PTHREAD_LIB}
|
||||
fmt::fmt
|
||||
)
|
||||
target_compile_options(${NAME} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
target_compile_definitions(${NAME} PRIVATE -DFEXLOADER_AS_INTERPRETER=${AsInterpreter})
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "Common/Config.h"
|
||||
#include "ELFCodeLoader.h"
|
||||
#include "VDSO_Emulation.h"
|
||||
#include "LinuxSyscalls/GdbServer.h"
|
||||
#include "LinuxSyscalls/LinuxAllocator.h"
|
||||
#include "LinuxSyscalls/Syscalls.h"
|
||||
#include "LinuxSyscalls/Utils/Threads.h"
|
||||
@@ -443,7 +444,6 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Context::InitializeStaticTables(Loader.Is64BitMode() ? FEXCore::Context::MODE_64BIT : FEXCore::Context::MODE_32BIT);
|
||||
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext();
|
||||
CTX->InitializeContext();
|
||||
|
||||
// Setup TSO hardware emulation immediately after initializing the context.
|
||||
FEX::TSO::SetupTSOEmulation(CTX.get());
|
||||
@@ -474,7 +474,14 @@ int main(int argc, char **argv, char **const envp) {
|
||||
|
||||
CTX->SetSignalDelegator(SignalDelegation.get());
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
fextl::unique_ptr<FEX::GdbServer> DebugServer;
|
||||
if (GdbServer) {
|
||||
DebugServer = fextl::make_unique<FEX::GdbServer>(CTX.get(), SignalDelegation.get(), SyscallHandler.get());
|
||||
}
|
||||
|
||||
auto ParentThread = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
|
||||
// Pass in our VDSO thunks
|
||||
CTX->AppendThunkDefinitions(FEX::VDSO::GetVDSOThunkDefinitions());
|
||||
@@ -550,8 +557,9 @@ int main(int argc, char **argv, char **const envp) {
|
||||
}
|
||||
}
|
||||
|
||||
auto ProgramStatus = CTX->GetProgramStatus();
|
||||
auto ProgramStatus = ParentThread->StatusCode;
|
||||
|
||||
DebugServer.reset();
|
||||
SyscallHandler.reset();
|
||||
SignalDelegation.reset();
|
||||
|
||||
|
||||
@@ -24,6 +24,7 @@
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEX::HarnessHelper {
|
||||
inline bool CompareStates(FEXCore::Core::CPUState const& State1,
|
||||
@@ -44,54 +45,11 @@ namespace FEX::HarnessHelper {
|
||||
fextl::fmt::print("{}: 0x{:016x} {} 0x{:016x}\n", Name, A, A==B ? "==" : "!=", B);
|
||||
};
|
||||
|
||||
const auto DumpFLAGs = [OutputGPRs](const fextl::string& Name, uint64_t A, uint64_t B) {
|
||||
if (!OutputGPRs) {
|
||||
return;
|
||||
}
|
||||
if (A == B) {
|
||||
return;
|
||||
}
|
||||
|
||||
static constexpr std::array<uint32_t, 17> Flags = {
|
||||
FEXCore::X86State::RFLAG_CF_LOC,
|
||||
FEXCore::X86State::RFLAG_PF_LOC,
|
||||
FEXCore::X86State::RFLAG_AF_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_LOC,
|
||||
FEXCore::X86State::RFLAG_SF_LOC,
|
||||
FEXCore::X86State::RFLAG_TF_LOC,
|
||||
FEXCore::X86State::RFLAG_IF_LOC,
|
||||
FEXCore::X86State::RFLAG_DF_LOC,
|
||||
FEXCore::X86State::RFLAG_OF_LOC,
|
||||
FEXCore::X86State::RFLAG_IOPL_LOC,
|
||||
FEXCore::X86State::RFLAG_NT_LOC,
|
||||
FEXCore::X86State::RFLAG_RF_LOC,
|
||||
FEXCore::X86State::RFLAG_VM_LOC,
|
||||
FEXCore::X86State::RFLAG_AC_LOC,
|
||||
FEXCore::X86State::RFLAG_VIF_LOC,
|
||||
FEXCore::X86State::RFLAG_VIP_LOC,
|
||||
FEXCore::X86State::RFLAG_ID_LOC,
|
||||
};
|
||||
|
||||
fextl::fmt::print("{}: 0x{:016x} {} 0x{:016x}\n", Name, A, A==B ? "==" : "!=", B);
|
||||
for (const auto Flag : Flags) {
|
||||
const auto FlagMask = uint64_t{1} << Flag;
|
||||
if ((A & FlagMask) != (B & FlagMask)) {
|
||||
fextl::fmt::print("\t{}: {} != {}\n", FEXCore::Core::GetFlagName(Flag), (A >> Flag) & 1, (B >> Flag) & 1);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const auto CheckGPRs = [&Matches, DumpGPRs](const fextl::string& Name, uint64_t A, uint64_t B){
|
||||
DumpGPRs(Name, A, B);
|
||||
Matches &= A == B;
|
||||
};
|
||||
|
||||
const auto CheckFLAGS = [&Matches, DumpFLAGs](const fextl::string& Name, uint64_t A, uint64_t B){
|
||||
DumpFLAGs(Name, A, B);
|
||||
Matches &= A == B;
|
||||
};
|
||||
|
||||
|
||||
// RIP
|
||||
if (MatchMask & 1) {
|
||||
CheckGPRs("RIP", State1.rip, State2.rip);
|
||||
@@ -135,22 +93,6 @@ namespace FEX::HarnessHelper {
|
||||
}
|
||||
MatchMask >>= 1;
|
||||
|
||||
auto CompactRFlags = [](auto Arg) -> uint32_t {
|
||||
uint32_t Res = 2;
|
||||
for (int i = 0; i < 32; ++i) {
|
||||
Res |= Arg->flags[i] << i;
|
||||
}
|
||||
return Res;
|
||||
};
|
||||
|
||||
// FLAGS
|
||||
if (MatchMask & 1) {
|
||||
uint32_t rflags1 = CompactRFlags(&State1);
|
||||
uint32_t rflags2 = CompactRFlags(&State2);
|
||||
|
||||
CheckFLAGS("FLAGS", rflags1, rflags2);
|
||||
}
|
||||
MatchMask >>= 1;
|
||||
return Matches;
|
||||
}
|
||||
|
||||
@@ -184,7 +126,7 @@ namespace FEX::HarnessHelper {
|
||||
}
|
||||
|
||||
if (BaseConfig.OptionRegDataCount > 0) {
|
||||
static constexpr std::array<uint64_t, 45> OffsetArrayAVX = {{
|
||||
static constexpr std::array<uint64_t, 44> OffsetArrayAVX = {{
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]),
|
||||
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RBX]),
|
||||
@@ -220,7 +162,6 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[15][0]),
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, flags),
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[1][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[2][0]),
|
||||
@@ -231,7 +172,7 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, mm[7][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[8][0]),
|
||||
}};
|
||||
static constexpr std::array<uint64_t, 45> OffsetArraySSE = {{
|
||||
static constexpr std::array<uint64_t, 44> OffsetArraySSE = {{
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]),
|
||||
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RBX]),
|
||||
@@ -267,7 +208,6 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[15][0]),
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, flags),
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[1][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[2][0]),
|
||||
@@ -440,17 +380,17 @@ namespace FEX::HarnessHelper {
|
||||
}
|
||||
|
||||
uint64_t StackSize() const override {
|
||||
return STACK_SIZE;
|
||||
return sysconf(_SC_PAGESIZE);
|
||||
}
|
||||
|
||||
uint64_t GetStackPointer() override {
|
||||
if (Config.Is64BitMode()) {
|
||||
return reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(STACK_SIZE)) + STACK_SIZE;
|
||||
return reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(StackSize())) + StackSize();
|
||||
}
|
||||
else {
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(reinterpret_cast<void*>(STACK_OFFSET), STACK_SIZE));
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(reinterpret_cast<void*>(STACK_OFFSET), StackSize()));
|
||||
LOGMAN_THROW_AA_FMT(Result != ~0ULL, "Stack Pointer mmap failed");
|
||||
return Result + STACK_SIZE;
|
||||
return Result + StackSize();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -466,11 +406,7 @@ namespace FEX::HarnessHelper {
|
||||
return Result;
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
const auto AllocPageSize = FHU::FEX_PAGE_SIZE;
|
||||
#else
|
||||
const auto AllocPageSize = 64 * 1024;
|
||||
#endif
|
||||
const auto AllocPageSize = sysconf(_SC_PAGESIZE);
|
||||
if (LimitedSize) {
|
||||
DoMMap(0xe000'0000, AllocPageSize * 10);
|
||||
|
||||
@@ -538,7 +474,6 @@ namespace FEX::HarnessHelper {
|
||||
bool RequiresLinux() const { return Config.RequiresLinux(); }
|
||||
|
||||
private:
|
||||
constexpr static uint64_t STACK_SIZE = FHU::FEX_PAGE_SIZE;
|
||||
constexpr static uint64_t STACK_OFFSET = 0xc000'0000;
|
||||
// Zero is special case to know when we are done
|
||||
uint64_t Code_start_page = 0x1'0000;
|
||||
|
||||
@@ -160,7 +160,6 @@ int main(int argc, char **argv, char **const envp)
|
||||
|
||||
FEXCore::Context::InitializeStaticTables();
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext();
|
||||
CTX->InitializeContext();
|
||||
|
||||
auto SignalDelegation = FEX::HLE::CreateSignalDelegator(CTX.get(), {});
|
||||
|
||||
@@ -179,7 +178,7 @@ int main(int argc, char **argv, char **const envp)
|
||||
|
||||
if (Loader.LoadIR(CTX.get()))
|
||||
{
|
||||
CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
auto ParentThread = CTX->InitCore(Loader.DefaultRIP(), Loader.GetStackPointer());
|
||||
|
||||
auto ShutdownReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
|
||||
|
||||
@@ -211,10 +210,7 @@ int main(int argc, char **argv, char **const envp)
|
||||
LogMan::Msg::DFmt("Reason we left VM: {}", FEXCore::ToUnderlying(ShutdownReason));
|
||||
|
||||
// Just re-use compare state. It also checks against the expected values in config.
|
||||
FEXCore::Core::CPUState State;
|
||||
CTX->GetCPUState(&State);
|
||||
|
||||
const bool Passed = Loader.CompareStates(&State, SupportsAVX);
|
||||
const bool Passed = Loader.CompareStates(&ParentThread->CurrentFrame->State, SupportsAVX);
|
||||
|
||||
LogMan::Msg::IFmt("Passed? {}\n", Passed ? "Yes" : "No");
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
add_compile_options(-fno-operator-names)
|
||||
|
||||
set (SRCS
|
||||
GdbServer.cpp
|
||||
EmulatedFiles/EmulatedFiles.cpp
|
||||
FileManagement.cpp
|
||||
LinuxAllocator.cpp
|
||||
|
||||
@@ -14,6 +14,7 @@ $end_info$
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -46,9 +47,22 @@ namespace FEX::EmulatedFile {
|
||||
*
|
||||
* @return A temporary file that we can use
|
||||
*/
|
||||
static int GenTmpFD() {
|
||||
int fd = open("/tmp", O_RDWR | O_TMPFILE | O_EXCL, S_IRUSR | S_IWUSR);
|
||||
return fd;
|
||||
static int GenTmpFD(const char *pathname, int flags) {
|
||||
uint32_t memfd_flags {MFD_ALLOW_SEALING};
|
||||
if (flags & O_CLOEXEC) memfd_flags |= MFD_CLOEXEC;
|
||||
|
||||
return memfd_create(pathname, memfd_flags);
|
||||
}
|
||||
|
||||
// Seal the tmpfd features by sealing them all.
|
||||
// Makes the tmpfd read-only.
|
||||
static void SealTmpFD(int fd) {
|
||||
fcntl(fd, F_ADD_SEALS,
|
||||
F_SEAL_SEAL |
|
||||
F_SEAL_SHRINK |
|
||||
F_SEAL_GROW |
|
||||
F_SEAL_WRITE |
|
||||
F_SEAL_FUTURE_WRITE);
|
||||
}
|
||||
|
||||
fextl::string GenerateCPUInfo(FEXCore::Context::Context *ctx, uint32_t CPUCores) {
|
||||
@@ -621,21 +635,23 @@ namespace FEX::EmulatedFile {
|
||||
}
|
||||
|
||||
EmulatedFDManager::EmulatedFDManager(FEXCore::Context::Context *ctx)
|
||||
: CTX {ctx} {
|
||||
: CTX {ctx}
|
||||
, ThreadsConfig { FEXCore::CPUInfo::CalculateNumberOfCPUs() } {
|
||||
FDReadCreators["/proc/cpuinfo"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
// Only allow a single thread to initialize the cpu_info.
|
||||
// Jit in-case multiple threads try to initialize at once.
|
||||
// Check if deferred cpuinfo initialization has occured.
|
||||
std::call_once(cpu_info_initialized, [&]() { cpu_info = GenerateCPUInfo(ctx, ThreadsConfig()); });
|
||||
std::call_once(cpu_info_initialized, [&]() { cpu_info = GenerateCPUInfo(ctx, ThreadsConfig); });
|
||||
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
write(FD, (void*)&cpu_info.at(0), cpu_info.size());
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
FDReadCreators["/proc/sys/kernel/osrelease"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
uint32_t GuestVersion = FEX::HLE::_SyscallHandler->GetGuestKernelVersion();
|
||||
char Tmp[64]{};
|
||||
snprintf(Tmp, sizeof(Tmp), "%d.%d.%d\n",
|
||||
@@ -645,11 +661,12 @@ namespace FEX::EmulatedFile {
|
||||
// + 1 to ensure null at the end
|
||||
write(FD, Tmp, strlen(Tmp) + 1);
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
FDReadCreators["/proc/version"] = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
// UTS version NEEDS to be in a format that can pass to `date -d`
|
||||
// Format of this is Linux version <Release> (<Compile By>@<Compile Host>) (<Linux Compiler>) #<version> {SMP, PREEMPT, PREEMPT_RT} <UTS version>\n"
|
||||
const char kernel_version[] = "Linux version %d.%d.%d (FEX@FEX) (clang) #" GIT_DESCRIBE_STRING " SMP " __DATE__ " " __TIME__ "\n";
|
||||
@@ -662,13 +679,15 @@ namespace FEX::EmulatedFile {
|
||||
// + 1 to ensure null at the end
|
||||
write(FD, Tmp, strlen(Tmp) + 1);
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
auto NumCPUCores = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
write(FD, (void*)&cpus_online.at(0), cpus_online.size());
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
@@ -681,7 +700,7 @@ namespace FEX::EmulatedFile {
|
||||
FDReadCreators["/proc/self/auxv"] = &EmulatedFDManager::ProcAuxv;
|
||||
|
||||
auto cmdline_handler = [&](FEXCore::Context::Context *ctx, int32_t fd, const char *pathname, int32_t flags, mode_t mode) -> int32_t {
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
auto CodeLoader = FEX::HLE::_SyscallHandler->GetCodeLoader();
|
||||
auto Args = CodeLoader->GetApplicationArguments();
|
||||
char NullChar{};
|
||||
@@ -695,6 +714,7 @@ namespace FEX::EmulatedFile {
|
||||
|
||||
// One additional null terminator to finish the list
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
};
|
||||
|
||||
@@ -702,9 +722,8 @@ namespace FEX::EmulatedFile {
|
||||
fextl::string procCmdLine = fextl::fmt::format("/proc/{}/cmdline", getpid());
|
||||
FDReadCreators[procCmdLine] = cmdline_handler;
|
||||
|
||||
uint64_t CPUCores = ThreadsConfig();
|
||||
if (CPUCores > 1) {
|
||||
cpus_online = fextl::fmt::format("0-{}", CPUCores - 1);
|
||||
if (ThreadsConfig > 1) {
|
||||
cpus_online = fextl::fmt::format("0-{}", ThreadsConfig - 1);
|
||||
}
|
||||
else {
|
||||
cpus_online = "0";
|
||||
@@ -721,6 +740,7 @@ namespace FEX::EmulatedFile {
|
||||
auto Creator = FDReadCreators.end();
|
||||
if (pathname) {
|
||||
Creator = FDReadCreators.find(pathname);
|
||||
Path = pathname;
|
||||
}
|
||||
|
||||
if (Creator == FDReadCreators.end()) {
|
||||
@@ -788,9 +808,10 @@ namespace FEX::EmulatedFile {
|
||||
return -1;
|
||||
}
|
||||
|
||||
int FD = GenTmpFD();
|
||||
int FD = GenTmpFD(pathname, flags);
|
||||
write(FD, (void*)auxvBase, auxvSize);
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
SealTmpFD(FD);
|
||||
return FD;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -34,6 +34,6 @@ namespace FEX::EmulatedFile {
|
||||
fextl::unordered_map<fextl::string, FDReadStringFunc> FDReadCreators;
|
||||
|
||||
static int32_t ProcAuxv(FEXCore::Context::Context* ctx, int32_t fd, const char* pathname, int32_t flags, mode_t mode);
|
||||
FEX_CONFIG_OPT(ThreadsConfig, THREADS);
|
||||
const uint32_t ThreadsConfig;
|
||||
};
|
||||
}
|
||||
@@ -416,8 +416,17 @@ fextl::string FileManager::GetEmulatedPath(const char *pathname, bool FollowSyml
|
||||
std::pair<int, const char*> FileManager::GetEmulatedFDPath(int dirfd, const char *pathname, bool FollowSymlink, FDPathTmpData &TmpFilename) {
|
||||
constexpr auto NoEntry = std::make_pair(-1, nullptr);
|
||||
|
||||
if (!pathname || // If no pathname
|
||||
pathname[0] != '/' || // If relative
|
||||
if (!pathname) {
|
||||
// No pathname.
|
||||
return NoEntry;
|
||||
}
|
||||
|
||||
if (pathname[0] == '/') {
|
||||
// If the path is absolute then dirfd is ignored.
|
||||
dirfd = AT_FDCWD;
|
||||
}
|
||||
|
||||
if (pathname[0] != '/' || // If relative
|
||||
pathname[1] == 0 || // If we are getting root
|
||||
dirfd != AT_FDCWD) { // If dirfd isn't special FDCWD
|
||||
return NoEntry;
|
||||
|
||||
+255
-77
@@ -11,10 +11,8 @@ $end_info$
|
||||
#include <iomanip>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <Common/FEXServerClient.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -28,12 +26,14 @@ $end_info$
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/NetStream.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstring>
|
||||
@@ -44,18 +44,72 @@ $end_info$
|
||||
#endif
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <fmt/format.h>
|
||||
#include <poll.h>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <string_view>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/un.h>
|
||||
#include <sys/utsname.h>
|
||||
#include <unistd.h>
|
||||
#include <utility>
|
||||
|
||||
#include "GdbServer.h"
|
||||
#include "LinuxSyscalls/GdbServer.h"
|
||||
|
||||
namespace FEXCore
|
||||
namespace FEX
|
||||
{
|
||||
|
||||
constexpr std::array<std::string_view const, 22> FlagNames = {
|
||||
"CF",
|
||||
"",
|
||||
"PF",
|
||||
"",
|
||||
"AF",
|
||||
"",
|
||||
"ZF",
|
||||
"SF",
|
||||
"TF",
|
||||
"IF",
|
||||
"DF",
|
||||
"OF",
|
||||
"IOPL",
|
||||
"",
|
||||
"NT",
|
||||
"",
|
||||
"RF",
|
||||
"VM",
|
||||
"AC",
|
||||
"VIF",
|
||||
"VIP",
|
||||
"ID",
|
||||
};
|
||||
|
||||
static std::string_view const& GetFlagName(unsigned Flag) {
|
||||
return FlagNames[Flag];
|
||||
}
|
||||
|
||||
static std::string_view const GetGRegName(unsigned Reg) {
|
||||
switch (Reg) {
|
||||
case FEXCore::X86State::REG_RAX: return "rax";
|
||||
case FEXCore::X86State::REG_RBX: return "rbx";
|
||||
case FEXCore::X86State::REG_RCX: return "rcx";
|
||||
case FEXCore::X86State::REG_RDX: return "rdx";
|
||||
case FEXCore::X86State::REG_RSP: return "rsp";
|
||||
case FEXCore::X86State::REG_RBP: return "rbp";
|
||||
case FEXCore::X86State::REG_RSI: return "rsi";
|
||||
case FEXCore::X86State::REG_RDI: return "rdi";
|
||||
case FEXCore::X86State::REG_R8: return "r8";
|
||||
case FEXCore::X86State::REG_R9: return "r9";
|
||||
case FEXCore::X86State::REG_R10: return "r10";
|
||||
case FEXCore::X86State::REG_R11: return "r11";
|
||||
case FEXCore::X86State::REG_R12: return "r12";
|
||||
case FEXCore::X86State::REG_R13: return "r13";
|
||||
case FEXCore::X86State::REG_R14: return "r14";
|
||||
case FEXCore::X86State::REG_R15: return "r15";
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
void GdbServer::Break(int signal) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
@@ -72,7 +126,14 @@ void GdbServer::WaitForThreadWakeup() {
|
||||
ThreadBreakEvent.Wait();
|
||||
}
|
||||
|
||||
GdbServer::GdbServer(FEXCore::Context::ContextImpl *ctx) : CTX(ctx) {
|
||||
GdbServer::~GdbServer() {
|
||||
CoreShuttingDown = true;
|
||||
close(ListenSocket);
|
||||
}
|
||||
|
||||
GdbServer::GdbServer(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler)
|
||||
: CTX(ctx)
|
||||
, SyscallHandler {SyscallHandler} {
|
||||
// Pass all signals by default
|
||||
std::fill(PassSignals.begin(), PassSignals.end(), true);
|
||||
|
||||
@@ -80,19 +141,21 @@ GdbServer::GdbServer(FEXCore::Context::ContextImpl *ctx) : CTX(ctx) {
|
||||
if (ExitReason == FEXCore::Context::ExitReason::EXIT_DEBUG) {
|
||||
this->Break(SIGTRAP);
|
||||
}
|
||||
|
||||
if (ExitReason == FEXCore::Context::ExitReason::EXIT_SHUTDOWN) {
|
||||
CoreShuttingDown = true;
|
||||
}
|
||||
});
|
||||
|
||||
// This is a total hack as there is currently no way to resume once hitting a segfault
|
||||
// But it's semi-useful for debugging.
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
ctx->SignalDelegation->RegisterHostSignalHandler(Signal, [this] (FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
for (uint32_t Signal = 0; Signal <= FEX::HLE::SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
SignalDelegation->RegisterHostSignalHandler(Signal, [this] (FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
if (PassSignals[Signal]) {
|
||||
// Pass signal to the guest
|
||||
return false;
|
||||
}
|
||||
|
||||
this->CTX->Config.RunningMode = FEXCore::Context::CoreRunningMode::MODE_SINGLESTEP;
|
||||
|
||||
// Let GDB know that we have a signal
|
||||
this->Break(Signal);
|
||||
|
||||
@@ -138,9 +201,13 @@ static fextl::string encodeHex(const unsigned char *data, size_t length) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
static fextl::string encodeHex(std::string_view str) {
|
||||
return encodeHex(reinterpret_cast<const unsigned char*>(str.data()), str.size());
|
||||
}
|
||||
|
||||
static fextl::string getThreadName(uint32_t ThreadID) {
|
||||
const auto ThreadFile = fextl::fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
|
||||
fextl::string ThreadName {"<No Name>"};
|
||||
fextl::string ThreadName;
|
||||
FEXCore::FileLoading::LoadFile(ThreadName, ThreadFile);
|
||||
return ThreadName;
|
||||
}
|
||||
@@ -247,16 +314,20 @@ void GdbServer::SendACK(std::ostream &stream, bool NACK) {
|
||||
}
|
||||
}
|
||||
|
||||
struct X80Float {
|
||||
uint8_t Data[10];
|
||||
};
|
||||
|
||||
struct FEX_PACKED GDBContextDefinition {
|
||||
uint64_t gregs[Core::CPUState::NUM_GPRS];
|
||||
uint64_t gregs[FEXCore::Core::CPUState::NUM_GPRS];
|
||||
uint64_t rip;
|
||||
uint32_t eflags;
|
||||
uint32_t cs, ss, ds, es, fs, gs;
|
||||
X80SoftFloat mm[Core::CPUState::NUM_MMS];
|
||||
X80Float mm[FEXCore::Core::CPUState::NUM_MMS];
|
||||
uint32_t fctrl;
|
||||
uint32_t fstat;
|
||||
uint32_t dummies[6];
|
||||
uint64_t xmm[Core::CPUState::NUM_XMMS][4];
|
||||
uint64_t xmm[FEXCore::Core::CPUState::NUM_XMMS][4];
|
||||
uint32_t mxcsr;
|
||||
};
|
||||
|
||||
@@ -265,10 +336,10 @@ fextl::string GdbServer::readRegs() {
|
||||
FEXCore::Core::CPUState state{};
|
||||
|
||||
auto Threads = CTX->GetThreads();
|
||||
FEXCore::Core::InternalThreadState *CurrentThread { CTX->ParentThread };
|
||||
FEXCore::Core::InternalThreadState *CurrentThread { Threads.ParentThread };
|
||||
bool Found = false;
|
||||
|
||||
for (auto &Thread : *Threads) {
|
||||
for (auto &Thread : *Threads.Threads) {
|
||||
if (Thread->ThreadManager.GetTID() != CurrentDebuggingThread) {
|
||||
continue;
|
||||
}
|
||||
@@ -280,16 +351,16 @@ fextl::string GdbServer::readRegs() {
|
||||
|
||||
if (!Found) {
|
||||
// If set to an invalid thread then just get the parent thread ID
|
||||
memcpy(&state, CTX->ParentThread->CurrentFrame, sizeof(state));
|
||||
memcpy(&state, Threads.ParentThread->CurrentFrame, sizeof(state));
|
||||
}
|
||||
|
||||
// Encode the GDB context definition
|
||||
memcpy(&GDB.gregs[0], &state.gregs[0], sizeof(GDB.gregs));
|
||||
memcpy(&GDB.rip, &state.rip, sizeof(GDB.rip));
|
||||
|
||||
GDB.eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread);
|
||||
GDB.eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread, false, nullptr, 0);
|
||||
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
memcpy(&GDB.mm[i], &state.mm[i], sizeof(GDB.mm));
|
||||
}
|
||||
|
||||
@@ -316,10 +387,10 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
FEXCore::Core::CPUState state{};
|
||||
|
||||
auto Threads = CTX->GetThreads();
|
||||
FEXCore::Core::InternalThreadState *CurrentThread { CTX->ParentThread };
|
||||
FEXCore::Core::InternalThreadState *CurrentThread { Threads.ParentThread };
|
||||
bool Found = false;
|
||||
|
||||
for (auto &Thread : *Threads) {
|
||||
for (auto &Thread : *Threads.Threads) {
|
||||
if (Thread->ThreadManager.GetTID() != CurrentDebuggingThread) {
|
||||
continue;
|
||||
}
|
||||
@@ -331,7 +402,7 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
|
||||
if (!Found) {
|
||||
// If set to an invalid thread then just get the parent thread ID
|
||||
memcpy(&state, CTX->ParentThread->CurrentFrame, sizeof(state));
|
||||
memcpy(&state, Threads.ParentThread->CurrentFrame, sizeof(state));
|
||||
}
|
||||
|
||||
|
||||
@@ -343,7 +414,7 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
return {encodeHex((unsigned char *)(&state.rip), sizeof(uint64_t)), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
else if (addr == offsetof(GDBContextDefinition, eflags)) {
|
||||
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread);
|
||||
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(CurrentThread, false, nullptr, 0);
|
||||
|
||||
return {encodeHex((unsigned char *)(&eflags), sizeof(uint32_t)), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
@@ -354,7 +425,7 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
}
|
||||
else if (addr >= offsetof(GDBContextDefinition, mm[0]) &&
|
||||
addr < offsetof(GDBContextDefinition, mm[8])) {
|
||||
return {encodeHex((unsigned char *)(&state.mm[(addr - offsetof(GDBContextDefinition, mm[0])) / sizeof(X80SoftFloat)]), sizeof(X80SoftFloat)), HandledPacketType::TYPE_ACK};
|
||||
return {encodeHex((unsigned char *)(&state.mm[(addr - offsetof(GDBContextDefinition, mm[0])) / sizeof(X80Float)]), sizeof(X80Float)), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
else if (addr == offsetof(GDBContextDefinition, fctrl)) {
|
||||
// XXX: We don't support this yet
|
||||
@@ -377,9 +448,9 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
}
|
||||
else if (addr >= offsetof(GDBContextDefinition, xmm[0][0]) &&
|
||||
addr < offsetof(GDBContextDefinition, xmm[16][0])) {
|
||||
const auto XmmIndex = (addr - offsetof(GDBContextDefinition, xmm[0][0])) / Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto XmmIndex = (addr - offsetof(GDBContextDefinition, xmm[0][0])) / FEXCore::Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto *Data = (unsigned char *)&state.xmm.avx.data[XmmIndex][0];
|
||||
return {encodeHex(Data, Core::CPUState::XMM_AVX_REG_SIZE), HandledPacketType::TYPE_ACK};
|
||||
return {encodeHex(Data, FEXCore::Core::CPUState::XMM_AVX_REG_SIZE), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
else if (addr == offsetof(GDBContextDefinition, mxcsr)) {
|
||||
uint32_t Empty{};
|
||||
@@ -403,7 +474,7 @@ fextl::string buildTargetXML() {
|
||||
xml << "<flags id='fex_eflags' size='4'>\n";
|
||||
// flags register
|
||||
for(int i = 0; i < 22; i++) {
|
||||
auto name = FEXCore::Core::GetFlagName(i);
|
||||
auto name = GetFlagName(i);
|
||||
if (name.empty()) {
|
||||
continue;
|
||||
}
|
||||
@@ -421,8 +492,8 @@ fextl::string buildTargetXML() {
|
||||
// We want to just memcpy our x86 state to gdb, so we tell it the ordering.
|
||||
|
||||
// GPRs
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_GPRS; i++) {
|
||||
reg(FEXCore::Core::GetGRegName(i), "int64", 64);
|
||||
for (uint32_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; i++) {
|
||||
reg(GetGRegName(i), "int64", 64);
|
||||
}
|
||||
|
||||
reg("rip", "code_ptr", 64);
|
||||
@@ -478,7 +549,7 @@ fextl::string buildTargetXML() {
|
||||
)";
|
||||
|
||||
// SSE regs
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
reg(fextl::fmt::format("xmm{}", i), "vec128", 128);
|
||||
}
|
||||
|
||||
@@ -504,7 +575,7 @@ fextl::string buildTargetXML() {
|
||||
<field name="uint128" type="uint128"/>
|
||||
</union>
|
||||
)";
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
reg(fmt::format("ymm{}h", i), "vec128", 128);
|
||||
}
|
||||
xml << "</feature>\n";
|
||||
@@ -673,15 +744,18 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet)
|
||||
|
||||
ThreadString.clear();
|
||||
fextl::ostringstream ss;
|
||||
ss << "<?xml version=\"1.0\"?>\n";
|
||||
ss << "<threads>\n";
|
||||
for (auto &Thread : *Threads) {
|
||||
for (auto &Thread : *Threads.Threads) {
|
||||
// Thread id is in hex without 0x prefix
|
||||
ss << "\t<thread id=\"" << std::hex << Thread->ThreadManager.GetTID() << "\" name=\"" << getThreadName(Thread->ThreadManager.GetTID()) << "\">\n";
|
||||
ss << "\t</thread>\n";
|
||||
const auto ThreadName = getThreadName(Thread->ThreadManager.GetTID());
|
||||
ss << "<thread id=\"" << std::hex << Thread->ThreadManager.GetTID() << "\"";
|
||||
if (!ThreadName.empty()) {
|
||||
ss << " name=\"" << ThreadName << "\"";
|
||||
}
|
||||
ss << "/>\n";
|
||||
}
|
||||
|
||||
ss << "</threads>";
|
||||
ss << "</threads>\n";
|
||||
ss << std::flush;
|
||||
ThreadString = ss.str();
|
||||
}
|
||||
@@ -705,11 +779,11 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet)
|
||||
}
|
||||
|
||||
if (object == "auxv") {
|
||||
auto CodeLoader = CTX->SyscallHandler->GetCodeLoader();
|
||||
auto CodeLoader = SyscallHandler->GetCodeLoader();
|
||||
uint64_t auxv_ptr, auxv_size;
|
||||
CodeLoader->GetAuxv(auxv_ptr, auxv_size);
|
||||
fextl::string data;
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
if (Is64BitMode()) {
|
||||
data.resize(auxv_size);
|
||||
memcpy(data.data(), reinterpret_cast<void*>(auxv_ptr), data.size());
|
||||
}
|
||||
@@ -760,7 +834,7 @@ static size_t CheckMemMapping(uint64_t Address, size_t Size) {
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleProgramOffsets() {
|
||||
auto CodeLoader = CTX->SyscallHandler->GetCodeLoader();
|
||||
auto CodeLoader = SyscallHandler->GetCodeLoader();
|
||||
uint64_t BaseOffset = CodeLoader->GetBaseOffset();
|
||||
fextl::string str = fextl::fmt::format("Text={:x};Data={:x};Bss={:x}", BaseOffset, BaseOffset, BaseOffset);
|
||||
return {std::move(str), HandledPacketType::TYPE_ACK};
|
||||
@@ -806,7 +880,6 @@ GdbServer::HandledPacketType GdbServer::handleMemory(const fextl::string &packet
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet) {
|
||||
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
const auto MatchStr = [](const fextl::string &Str, const char *str) -> bool { return Str.rfind(str, 0) == 0; };
|
||||
@@ -858,12 +931,17 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
SupportedFeatures += "QNonStop+;";
|
||||
|
||||
SupportedFeatures += "qXfer:osdata:read+;";
|
||||
SupportedFeatures += "QStartNoAckMode+;";
|
||||
|
||||
// Causes GDB to crash?
|
||||
// SupportedFeatures += "QStartNoAckMode+;";
|
||||
// TODO: Support breakpoints
|
||||
// SupportedFeatures += "swbreak+;";
|
||||
// SupportedFeatures += "hwbreak+;";
|
||||
// SupportedFeatures += "BreakpointCommands+;";
|
||||
|
||||
// TODO: If we want to support conditional breakpoints then we need to support single stepping.
|
||||
// SupportedFeatures += "ConditionalBreakpoints+;";
|
||||
|
||||
for (auto &Feature : Features) {
|
||||
|
||||
if (MatchStr(Feature, "swbreak+")) {
|
||||
SupportedFeatures += "swbreak+;";
|
||||
}
|
||||
@@ -904,10 +982,10 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
|
||||
fextl::ostringstream ss;
|
||||
ss << "m";
|
||||
for (size_t i = 0; i < Threads->size(); ++i) {
|
||||
auto Thread = Threads->at(i);
|
||||
for (size_t i = 0; i < Threads.Threads->size(); ++i) {
|
||||
auto Thread = Threads.Threads->at(i);
|
||||
ss << std::hex << Thread->ThreadManager.TID;
|
||||
if (i != (Threads->size() - 1)) {
|
||||
if (i != (Threads.Threads->size() - 1)) {
|
||||
ss << ",";
|
||||
}
|
||||
}
|
||||
@@ -928,7 +1006,7 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
if (match("qC")) {
|
||||
// Returns the current Thread ID
|
||||
fextl::ostringstream ss;
|
||||
ss << "m" << std::hex << CTX->ParentThread->ThreadManager.TID;
|
||||
ss << "m" << std::hex << CTX->GetThreads().ParentThread->ThreadManager.TID;
|
||||
return {ss.str(), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("QStartNoAckMode")) {
|
||||
@@ -963,13 +1041,68 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
// We now have a semi-colon deliminated list of signals to pass to the guest process
|
||||
for (fextl::string tmp; std::getline(ss, tmp, ';'); ) {
|
||||
uint32_t Signal = std::stoi(tmp.c_str(), nullptr, 16);
|
||||
if (Signal < SignalDelegator::MAX_SIGNALS) {
|
||||
if (Signal < FEX::HLE::SignalDelegator::MAX_SIGNALS) {
|
||||
PassSignals[Signal] = true;
|
||||
}
|
||||
}
|
||||
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
// lldb specific queries
|
||||
if (match("qHostInfo")) {
|
||||
// Returns Key:Value pairs separated by ;
|
||||
// eg:
|
||||
// triple:7838365f36342d70632d6c696e75782d676e75;
|
||||
// ptrsize:8;
|
||||
// distribution_id:7562756e7475;
|
||||
// watchpoint_exceptions_received:after;
|
||||
// endian:little;
|
||||
// os_version:6.3.3;
|
||||
// os_build:362e332e332d3036303330332d67656e65726963;
|
||||
// os_kernel:2332303233303531373133333620534d5020505245454d50545f44594e414d494320576564204d61792031372031333a34353a3139205554432032303233;
|
||||
// hostname:7279616e682d545235303030;
|
||||
fextl::string HostFeatures{};
|
||||
|
||||
// 64-bit always returned for the host environment.
|
||||
// qProcessInfo will return i386 or not.
|
||||
HostFeatures += fextl::fmt::format("triple:{};", encodeHex("x86_64-pc-linux-gnu"));
|
||||
HostFeatures += "ptrsize:8;";
|
||||
|
||||
// Always little-endian.
|
||||
HostFeatures += "endian:little;";
|
||||
|
||||
struct utsname buf{};
|
||||
if (uname(&buf) != -1) {
|
||||
uint32_t Major{};
|
||||
uint32_t Minor{};
|
||||
uint32_t Patch{};
|
||||
|
||||
// Parse kernel version in the form of `<Major>.<Minor>.<Patch>[Optional Data]`
|
||||
const auto End = buf.release + sizeof(buf.release);
|
||||
auto Results = std::from_chars(buf.release, End, Major, 10);
|
||||
Results = std::from_chars(Results.ptr + 1, End, Minor, 10);
|
||||
Results = std::from_chars(Results.ptr + 1, End, Patch, 10);
|
||||
|
||||
HostFeatures += fextl::fmt::format("os_version:{}.{}.{};", Major, Minor, Patch);
|
||||
|
||||
// os_build returns the release untouched.
|
||||
HostFeatures += fextl::fmt::format("os_build:{};", encodeHex(buf.release));
|
||||
HostFeatures += fextl::fmt::format("os_kernel:{};", encodeHex(buf.version));
|
||||
HostFeatures += fextl::fmt::format("hostname:{};", encodeHex(buf.nodename));
|
||||
}
|
||||
|
||||
// TODO: distribution_id should be fetched with `lsb_release -i`
|
||||
// TODO: watchpoint_exceptions_received is unsupported
|
||||
return {std::move(HostFeatures), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("qGetWorkingDir")) {
|
||||
char Tmp[PATH_MAX];
|
||||
if (getcwd(Tmp, PATH_MAX)) {
|
||||
return {encodeHex(Tmp), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
return {"", HandledPacketType::TYPE_UNKNOWN};
|
||||
}
|
||||
|
||||
@@ -996,7 +1129,7 @@ GdbServer::HandledPacketType GdbServer::ThreadAction(char action, uint32_t tid)
|
||||
}
|
||||
case 't':
|
||||
// This thread isn't part of the thread pool
|
||||
CTX->Stop(false /* Ignore current thread */);
|
||||
CTX->Stop();
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
default:
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
@@ -1183,7 +1316,7 @@ GdbServer::HandledPacketType GdbServer::ProcessPacket(const fextl::string &packe
|
||||
case 'Z': // Inserts breakpoint or watchpoint
|
||||
return handleBreakpoint(packet);
|
||||
case 'k': // Kill the process
|
||||
CTX->Stop(false /* Ignore current thread */);
|
||||
CTX->Stop();
|
||||
CTX->WaitForIdle(); // Block until exit
|
||||
return {"", HandledPacketType::TYPE_NONE};
|
||||
default:
|
||||
@@ -1212,11 +1345,45 @@ void GdbServer::SendPacketPair(const HandledPacketType& response) {
|
||||
}
|
||||
}
|
||||
|
||||
GdbServer::WaitForConnectionResult GdbServer::WaitForConnection() {
|
||||
while (!CoreShuttingDown.load()) {
|
||||
struct pollfd PollFD {
|
||||
.fd = ListenSocket,
|
||||
.events = POLLIN | POLLPRI | POLLRDHUP,
|
||||
.revents = 0,
|
||||
};
|
||||
int Result = ppoll(&PollFD, 1, nullptr, nullptr);
|
||||
if (Result > 0) {
|
||||
if (PollFD.revents & POLLIN) {
|
||||
CommsStream = OpenSocket();
|
||||
return WaitForConnectionResult::CONNECTION;
|
||||
}
|
||||
else if (PollFD.revents & (POLLHUP | POLLERR | POLLNVAL)) {
|
||||
// Listen socket error or shutting down
|
||||
LogMan::Msg::EFmt("[GdbServer] gdbserver shutting down: {}");
|
||||
return WaitForConnectionResult::ERROR;
|
||||
}
|
||||
}
|
||||
else if (Result == -1) {
|
||||
LogMan::Msg::EFmt("[GdbServer] poll failure: {}", errno);
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Msg::EFmt("[GdbServer] Shutting Down");
|
||||
return WaitForConnectionResult::ERROR;
|
||||
}
|
||||
|
||||
void GdbServer::GdbServerLoop() {
|
||||
OpenListenSocket();
|
||||
if (ListenSocket == -1) {
|
||||
// Couldn't open socket, just exit.
|
||||
return;
|
||||
}
|
||||
|
||||
while (!CTX->CoreShuttingDown.load()) {
|
||||
CommsStream = OpenSocket();
|
||||
while (!CoreShuttingDown.load()) {
|
||||
if (WaitForConnection() == WaitForConnectionResult::ERROR) {
|
||||
break;
|
||||
}
|
||||
|
||||
HandledPacketType response{};
|
||||
|
||||
@@ -1266,9 +1433,11 @@ void GdbServer::GdbServerLoop() {
|
||||
}
|
||||
|
||||
close(ListenSocket);
|
||||
unlink(GdbUnixSocketPath.c_str());
|
||||
}
|
||||
static void* ThreadHandler(void *Arg) {
|
||||
FEXCore::GdbServer *This = reinterpret_cast<FEXCore::GdbServer*>(Arg);
|
||||
FEXCore::Threads::SetThreadName("FEX:gdbserver");
|
||||
auto This = reinterpret_cast<FEX::GdbServer*>(Arg);
|
||||
This->GdbServerLoop();
|
||||
return nullptr;
|
||||
}
|
||||
@@ -1280,38 +1449,48 @@ void GdbServer::StartThread() {
|
||||
}
|
||||
|
||||
void GdbServer::OpenListenSocket() {
|
||||
// getaddrinfo allocates memory that can't be removed.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
struct addrinfo hints, *res;
|
||||
|
||||
memset(&hints, 0, sizeof(hints));
|
||||
hints.ai_family = AF_UNSPEC;
|
||||
hints.ai_socktype = SOCK_STREAM;
|
||||
hints.ai_flags = AI_PASSIVE;
|
||||
|
||||
if(getaddrinfo(NULL, "8086", &hints, &res) < 0) {
|
||||
perror("getaddrinfo");
|
||||
const auto GdbUnixPath = fextl::fmt::format("{}/FEX_gdbserver/", FEXServerClient::GetTempFolder());
|
||||
if (FHU::Filesystem::CreateDirectory(GdbUnixPath) == FHU::Filesystem::CreateDirectoryResult::ERROR) {
|
||||
LogMan::Msg::EFmt("[GdbServer] Couldn't create gdbserver folder {}", GdbUnixPath);
|
||||
return;
|
||||
}
|
||||
|
||||
int on = 1;
|
||||
GdbUnixSocketPath = fextl::fmt::format("{}{}-gdb", GdbUnixPath, ::getpid());
|
||||
|
||||
ListenSocket = socket(res->ai_family, res->ai_socktype, res->ai_protocol);
|
||||
if (ListenSocket < 0) {
|
||||
perror("socket");
|
||||
ListenSocket = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (ListenSocket == -1) {
|
||||
LogMan::Msg::EFmt("[GdbServer] Couldn't open AF_UNIX socket {} {}", errno, strerror(errno));
|
||||
return;
|
||||
}
|
||||
if(setsockopt(ListenSocket, SOL_SOCKET, SO_REUSEADDR, (char*)&on, sizeof(on)) < 0) {
|
||||
perror("setsockopt");
|
||||
close(ListenSocket);
|
||||
|
||||
struct sockaddr_un addr{};
|
||||
addr.sun_family = AF_UNIX;
|
||||
strncpy(addr.sun_path, GdbUnixSocketPath.data(), sizeof(addr.sun_path));
|
||||
size_t SizeOfAddr = offsetof(sockaddr_un, sun_path) + GdbUnixSocketPath.size();
|
||||
|
||||
// Bind the socket to the path
|
||||
int Result{};
|
||||
for (int attempt = 0; attempt < 2; ++attempt) {
|
||||
Result = bind(ListenSocket, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr);
|
||||
if (Result == 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
// This can happen periodically with execve. unlink the path and try again.
|
||||
// The PID is reused but FEX likely started a gdbserver thread for the PID before execve.
|
||||
unlink(GdbUnixSocketPath.c_str());
|
||||
}
|
||||
|
||||
if (bind(ListenSocket, res->ai_addr, res->ai_addrlen) < 0) {
|
||||
perror("bind");
|
||||
if (Result != 0) {
|
||||
LogMan::Msg::EFmt("[GdbServer] Couldn't bind AF_UNIX socket '{}': {} {}\n", addr.sun_path, errno, strerror(errno));
|
||||
close(ListenSocket);
|
||||
ListenSocket = -1;
|
||||
return;
|
||||
}
|
||||
|
||||
listen(ListenSocket, 1);
|
||||
|
||||
freeaddrinfo(res);
|
||||
LogMan::Msg::IFmt("[GdbServer] Waiting for connection on {}", GdbUnixSocketPath);
|
||||
LogMan::Msg::IFmt("[GdbServer] gdb-multiarch -ex \"target extended-remote {}\"", GdbUnixSocketPath);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
@@ -1319,7 +1498,6 @@ fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
struct sockaddr_storage their_addr{};
|
||||
socklen_t addr_size{};
|
||||
|
||||
LogMan::Msg::IFmt("GdbServer, waiting for connection on localhost:8086");
|
||||
int new_fd = accept(ListenSocket, (struct sockaddr *)&their_addr, &addr_size);
|
||||
|
||||
return fextl::make_unique<FEXCore::Utils::NetStream>(new_fd);
|
||||
+16
-7
@@ -7,6 +7,7 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -18,15 +19,14 @@ $end_info$
|
||||
#include <mutex>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore {
|
||||
#include "LinuxSyscalls/SignalDelegator.h"
|
||||
|
||||
namespace Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEX {
|
||||
|
||||
class GdbServer {
|
||||
public:
|
||||
GdbServer(FEXCore::Context::ContextImpl *ctx);
|
||||
GdbServer(FEXCore::Context::Context *ctx, FEX::HLE::SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler);
|
||||
~GdbServer();
|
||||
|
||||
// Public for threading
|
||||
void GdbServerLoop();
|
||||
@@ -39,6 +39,11 @@ private:
|
||||
void Break(int signal);
|
||||
|
||||
void OpenListenSocket();
|
||||
enum class WaitForConnectionResult {
|
||||
CONNECTION,
|
||||
ERROR,
|
||||
};
|
||||
WaitForConnectionResult WaitForConnection();
|
||||
fextl::unique_ptr<std::iostream> OpenSocket();
|
||||
void StartThread();
|
||||
fextl::string ReadPacket(std::iostream &stream);
|
||||
@@ -77,7 +82,8 @@ private:
|
||||
fextl::string readRegs();
|
||||
HandledPacketType readReg(const fextl::string& packet);
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::HLE::SyscallHandler *const SyscallHandler;
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
fextl::unique_ptr<std::iostream> CommsStream;
|
||||
std::mutex sendMutex;
|
||||
@@ -88,13 +94,16 @@ private:
|
||||
fextl::string OSDataString{};
|
||||
void buildLibraryMap();
|
||||
std::atomic<bool> LibraryMapChanged = true;
|
||||
std::atomic<bool> CoreShuttingDown{};
|
||||
fextl::string LibraryMapString{};
|
||||
|
||||
// Used to keep track of which signals to pass to the guest
|
||||
std::array<bool, SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
|
||||
std::array<bool, FEX::HLE::SignalDelegator::MAX_SIGNALS + 1> PassSignals{};
|
||||
uint32_t CurrentDebuggingThread{};
|
||||
int ListenSocket{};
|
||||
fextl::string GdbUnixSocketPath{};
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
};
|
||||
|
||||
}
|
||||
@@ -150,6 +150,38 @@ namespace FEX::HLE {
|
||||
return SigInfoLayout::LAYOUT_KILL;
|
||||
}
|
||||
|
||||
void SignalDelegator::HandleSignal(int Signal, void *Info, void *UContext) {
|
||||
// Let the host take first stab at handling the signal
|
||||
auto Thread = GetTLSThread();
|
||||
|
||||
if (!Thread) {
|
||||
LogMan::Msg::AFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
|
||||
}
|
||||
else {
|
||||
SignalHandler &Handler = HostHandlers[Signal];
|
||||
for (auto &HandlerFunc : Handler.Handlers) {
|
||||
if (HandlerFunc(Thread, Signal, Info, UContext)) {
|
||||
// If the host handler handled the fault then we can continue now
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (Handler.FrontendHandler &&
|
||||
Handler.FrontendHandler(Thread, Signal, Info, UContext)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Now let the frontend handle the signal
|
||||
// It's clearly a guest signal and this ends up being an OS specific issue
|
||||
HandleGuestSignal(Thread, Signal, Info, UContext);
|
||||
}
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SetHostSignalHandler(Signal, Func, Required);
|
||||
FrontendRegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void SignalDelegator::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
|
||||
#ifdef _M_ARM_64
|
||||
for (size_t i = 0; i < Config.SRAGPRCount; i++) {
|
||||
@@ -1137,12 +1169,13 @@ namespace FEX::HLE {
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
|
||||
const bool WasInJIT = Thread->CPUBackend->IsAddressInCodeBuffer(OldPC);
|
||||
|
||||
// Spill the SRA regardless of signal handler type
|
||||
// We are going to be returning to the top of the dispatcher which will fill again
|
||||
// Otherwise we might load garbage
|
||||
if (Config.StaticRegisterAllocation) {
|
||||
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
|
||||
if (WasInJIT) {
|
||||
uint32_t IgnoreMask{};
|
||||
#ifdef _M_ARM_64
|
||||
if (Frame->InSyscallInfo != 0) {
|
||||
@@ -1207,7 +1240,7 @@ namespace FEX::HLE {
|
||||
// Backup where we think the RIP currently is
|
||||
ContextBackup->OriginalRIP = CTX->RestoreRIPFromHostPC(Thread, ArchHelpers::Context::GetPc(ucontext));
|
||||
// Calculate eflags upfront.
|
||||
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(Thread);
|
||||
uint32_t eflags = CTX->ReconstructCompactedEFLAGS(Thread, WasInJIT, ArchHelpers::Context::GetArmGPRs(ucontext), ArchHelpers::Context::GetArmPState(ucontext));
|
||||
|
||||
if (Is64BitMode) {
|
||||
NewGuestSP = SetupFrame_x64(Thread, ContextBackup, Frame, Signal, HostSigInfo, ucontext, GuestAction, GuestStack, NewGuestSP, eflags);
|
||||
@@ -1810,7 +1843,7 @@ namespace FEX::HLE {
|
||||
ThreadData.Thread = nullptr;
|
||||
}
|
||||
|
||||
void SignalDelegator::FrontendRegisterHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) {
|
||||
void SignalDelegator::FrontendRegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
// Linux signal handlers are per-process rather than per thread
|
||||
// Multiple threads could be calling in to this
|
||||
std::lock_guard lk(HostDelegatorMutex);
|
||||
@@ -1818,7 +1851,7 @@ namespace FEX::HLE {
|
||||
InstallHostThunk(Signal);
|
||||
}
|
||||
|
||||
void SignalDelegator::FrontendRegisterFrontendHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) {
|
||||
void SignalDelegator::FrontendRegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
// Linux signal handlers are per-process rather than per thread
|
||||
// Multiple threads could be calling in to this
|
||||
std::lock_guard lk(HostDelegatorMutex);
|
||||
|
||||
@@ -40,20 +40,36 @@ namespace FEX::HLE {
|
||||
|
||||
class SignalDelegator final : public FEXCore::SignalDelegator, public FEXCore::Allocator::FEXAllocOperators {
|
||||
public:
|
||||
constexpr static size_t MAX_SIGNALS {64};
|
||||
|
||||
// Use the last signal just so we are less likely to ever conflict with something that the guest application is using
|
||||
// 64 is used internally by Valgrind
|
||||
constexpr static size_t SIGNAL_FOR_PAUSE {63};
|
||||
|
||||
// Returns true if the host handled the signal
|
||||
// Arguments are the same as sigaction handler
|
||||
SignalDelegator(FEXCore::Context::Context *_CTX, const std::string_view ApplicationName);
|
||||
~SignalDelegator() override;
|
||||
|
||||
// Called from the signal trampoline function.
|
||||
void HandleSignal(int Signal, void *Info, void *UContext);
|
||||
|
||||
void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal specifically for guest handling
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void RegisterHostSignalHandlerForGuest(int Signal, FEX::HLE::HostSignalDelegatorFunctionForGuest Func);
|
||||
void RegisterHostSignalHandlerForGuest(int Signal, HostSignalDelegatorFunctionForGuest Func);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
/**
|
||||
@@ -107,21 +123,27 @@ namespace FEX::HLE {
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
|
||||
void SaveTelemetry();
|
||||
protected:
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *Info, void *UContext) override;
|
||||
private:
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread();
|
||||
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread() override;
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *Info, void *UContext);
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
void FrontendRegisterHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override;
|
||||
void FrontendRegisterFrontendHostSignalHandler(int Signal, FEXCore::HostSignalDelegatorFunction Func, bool Required) override;
|
||||
void FrontendRegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void FrontendRegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
void SetHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
HostHandlers[Signal].Handlers.push_back(std::move(Func));
|
||||
}
|
||||
void SetFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
HostHandlers[Signal].FrontendHandler = std::move(Func);
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
fextl::string const ApplicationName;
|
||||
@@ -155,6 +177,10 @@ namespace FEX::HLE {
|
||||
FEX::HLE::HostSignalDelegatorFunctionForGuest GuestHandler{};
|
||||
GuestSigAction GuestAction{};
|
||||
DefaultBehaviour DefaultBehaviour {DEFAULT_TERM};
|
||||
|
||||
// Callbacks
|
||||
fextl::vector<HostSignalDelegatorFunction> Handlers{};
|
||||
HostSignalDelegatorFunction FrontendHandler{};
|
||||
};
|
||||
|
||||
std::array<SignalHandler, MAX_SIGNALS + 1> HostHandlers{};
|
||||
|
||||
@@ -525,6 +525,14 @@ static uint64_t Clone3Handler(FEXCore::Core::CpuStateFrame *Frame, FEX::HLE::clo
|
||||
uint64_t CloneHandler(FEXCore::Core::CpuStateFrame *Frame, FEX::HLE::clone3_args *args) {
|
||||
uint64_t flags = args->args.flags;
|
||||
|
||||
if (flags & CLONE_CLEAR_SIGHAND) {
|
||||
// CLONE_CLEAR_SIGHAND was added in kernel 5.5. FEX doesn't properly support this.
|
||||
// glibc started using this flag in 2.38 as an optimization for posix_spawn.
|
||||
// If clone returns EINVAL or ENOSYS then it will fallback to the non-optimized path.
|
||||
LogMan::Msg::IFmt("CLONE_CLEAR_SIGHAND passed to clone3. Returning EINVAL.");
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
auto HasUnhandledFlags = [](FEX::HLE::clone3_args *args) -> bool {
|
||||
constexpr uint64_t UNHANDLED_FLAGS =
|
||||
CLONE_NEWNS |
|
||||
|
||||
@@ -16,7 +16,7 @@ $end_info$
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -175,7 +175,6 @@ public:
|
||||
FEX_CONFIG_OPT(IsInterpreterInstalled, INTERPRETER_INSTALLED);
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
FEX_CONFIG_OPT(ThreadsConfig, THREADS);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
|
||||
|
||||
@@ -89,21 +89,8 @@ namespace FEX::HLE {
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(getcpu, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame *Frame, unsigned *cpu, unsigned *node, struct getcpu_cache *tcache) -> uint64_t {
|
||||
uint32_t LocalCPU{};
|
||||
uint32_t LocalNode{};
|
||||
// tcache is ignored
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(getcpu), cpu ? &LocalCPU : nullptr, node ? &LocalNode : nullptr, nullptr);
|
||||
if (Result == 0) {
|
||||
if (cpu) {
|
||||
// Ensure we don't return a number over our number of emulated cores
|
||||
*cpu = LocalCPU % FEX::HLE::_SyscallHandler->ThreadsConfig();
|
||||
}
|
||||
|
||||
if (node) {
|
||||
// Just claim we are part of node zero
|
||||
*node = 0;
|
||||
}
|
||||
}
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(getcpu), cpu, node, nullptr);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
|
||||
Loaded 100 of 261 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user