mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 12:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d5db8948db | ||
|
|
5013b8a0db | ||
|
|
ac65deed6c | ||
|
|
d1d4d2d876 | ||
|
|
e234e118e0 | ||
|
|
1cc54312fa | ||
|
|
de9ab6a023 | ||
|
|
463a0b5ed4 | ||
|
|
7b0365b377 | ||
|
|
803501526e | ||
|
|
a63a3a47e3 | ||
|
|
a66fac614b | ||
|
|
6d4693cbc1 | ||
|
|
46a2a0608b | ||
|
|
49b40b71a6 | ||
|
|
ca8f347902 | ||
|
|
e415b9443f | ||
|
|
fd4f6b8020 | ||
|
|
7a9eb01573 | ||
|
|
b0474816c7 | ||
|
|
941b065da6 | ||
|
|
ccba993ca9 | ||
|
|
e664f61da8 | ||
|
|
c74df6a59b | ||
|
|
fc76d4e8c3 | ||
|
|
099211f39c | ||
|
|
b327d7ae19 | ||
|
|
0e6a22febe | ||
|
|
b368223d50 | ||
|
|
8fe1e9562d | ||
|
|
ae545b8cf3 | ||
|
|
ac32876e4e | ||
|
|
9336e35052 | ||
|
|
0754affb98 | ||
|
|
c413d7950b | ||
|
|
f8eaf9c14f | ||
|
|
fc677eaabf | ||
|
|
114112a716 | ||
|
|
9056d9b9de | ||
|
|
74e95df661 | ||
|
|
d9544e7e02 | ||
|
|
62e1767ee0 | ||
|
|
4baeffe84f | ||
|
|
b4a67a6178 | ||
|
|
8296bfc7de | ||
|
|
f588304b12 | ||
|
|
c748dbf0e3 | ||
|
|
06497fdfad | ||
|
|
e2a7fef742 | ||
|
|
17692d6eef | ||
|
|
3020a0db2b | ||
|
|
92ddc0041b | ||
|
|
8a5388f514 | ||
|
|
a444db7dac | ||
|
|
74da2eb5fc | ||
|
|
539d5ed26f | ||
|
|
cf7ee98831 | ||
|
|
5529948479 | ||
|
|
91b6aeffe1 | ||
|
|
3a79b61f58 | ||
|
|
e92b24302f | ||
|
|
cc9ccf881b | ||
|
|
72d74ae482 | ||
|
|
e88c57bec7 | ||
|
|
ac814ac015 | ||
|
|
90f7cc925d | ||
|
|
2039950762 | ||
|
|
b526c60c73 | ||
|
|
1bfdd031d3 | ||
|
|
49ee8ba3be | ||
|
|
335cd9180e | ||
|
|
ca9a94d572 | ||
|
|
03832b2523 | ||
|
|
812224a0ef | ||
|
|
42f2851575 | ||
|
|
9c7f44f8d4 | ||
|
|
5117ba351e | ||
|
|
441fdb689d | ||
|
|
e786dfc998 | ||
|
|
a1d3183c14 | ||
|
|
c5ef0910c5 | ||
|
|
affa1d0efc | ||
|
|
a83dc27a42 | ||
|
|
129ec63e92 | ||
|
|
3459369c6e | ||
|
|
7a490a3811 | ||
|
|
ee17fe239a | ||
|
|
0416950aaa | ||
|
|
6e46383cdb | ||
|
|
faf1b85904 | ||
|
|
bec5f4fe2a | ||
|
|
fe6dbbfa63 | ||
|
|
b19440c78f | ||
|
|
abf9700475 | ||
|
|
8d6b454455 | ||
|
|
205ec3e14d | ||
|
|
f897579593 | ||
|
|
2478abba29 | ||
|
|
a4db585664 | ||
|
|
8bf4a124c8 | ||
|
|
13c3b65732 | ||
|
|
7e6ba184f9 | ||
|
|
f6cb914a2a | ||
|
|
6c92f94ee8 | ||
|
|
bebf4209d6 | ||
|
|
c82a683987 | ||
|
|
886d40ccfe | ||
|
|
8f33e56e21 | ||
|
|
b1634680fe | ||
|
|
54d332935e | ||
|
|
2829ad56a1 | ||
|
|
fbf62f1296 | ||
|
|
e6aa268093 | ||
|
|
cc589ba7e6 | ||
|
|
f009a00986 | ||
|
|
8617150a42 | ||
|
|
ef823ce82b | ||
|
|
fc2917117e | ||
|
|
68abe400a7 | ||
|
|
df86d80a85 | ||
|
|
58cff72d38 | ||
|
|
fd1f5643c7 | ||
|
|
0ea29dfbdf | ||
|
|
f4fca4482f | ||
|
|
86d1ce7f00 | ||
|
|
15fef1794c | ||
|
|
8b5d9d8fdf | ||
|
|
0ec724cf1f | ||
|
|
0dc0117e86 | ||
|
|
1d00ad6030 | ||
|
|
23a076c313 | ||
|
|
57eacab654 | ||
|
|
b4093a8888 | ||
|
|
7c5a9b5d6a | ||
|
|
12b3c82d83 | ||
|
|
66520bce0a | ||
|
|
25cc2bdcb8 | ||
|
|
f98b18800c | ||
|
|
0dd687a7a1 | ||
|
|
1aff3acbb9 | ||
|
|
a6ab2ca30d | ||
|
|
ca43e2a61c | ||
|
|
4f404160d0 | ||
|
|
5edc69b692 | ||
|
|
b6cb897896 | ||
|
|
47b3637452 | ||
|
|
5b65f30c8f | ||
|
|
23572539f8 | ||
|
|
7176c717e5 | ||
|
|
3a05b760d4 | ||
|
|
1e0273d570 | ||
|
|
8232be6302 | ||
|
|
43cd897cf4 | ||
|
|
e593807856 | ||
|
|
ce8e6e5c0c | ||
|
|
5b9a7c7845 | ||
|
|
0e5f5a2db9 | ||
|
|
50a9cea16e | ||
|
|
6aa95d82f2 | ||
|
|
7c34f449d1 | ||
|
|
e9435203a0 | ||
|
|
8d9ba0e102 | ||
|
|
3fcfd2ccde | ||
|
|
aa4205c2e8 | ||
|
|
5d5b411611 | ||
|
|
5f5a06fb3b | ||
|
|
3077addcf8 | ||
|
|
8b6fe0c4ff | ||
|
|
29fd62e9ba | ||
|
|
2631b113da | ||
|
|
da152031b3 | ||
|
|
be9c5678c0 | ||
|
|
fdd370dd1a | ||
|
|
6951284924 | ||
|
|
b03c613fbf | ||
|
|
b893bdb8df | ||
|
|
db45f6eec8 | ||
|
|
d53e689e22 | ||
|
|
f55378257a | ||
|
|
6a7914ac56 | ||
|
|
d935d25f0c | ||
|
|
84cf1d2fd2 | ||
|
|
ed0c045c17 | ||
|
|
d585063e60 | ||
|
|
3077fec9ab | ||
|
|
9500842efc | ||
|
|
a6f9c51317 | ||
|
|
cfa2ad8423 | ||
|
|
19010491da | ||
|
|
5d613e8716 | ||
|
|
894aaa980f | ||
|
|
877b2f4fef | ||
|
|
2a170cfdec | ||
|
|
a299d6b1a5 | ||
|
|
ffb85e6305 | ||
|
|
d7a20fa28f | ||
|
|
5ac7d5dfcd | ||
|
|
75644b33df | ||
|
|
4c4c6e7807 | ||
|
|
a8c9c71ce3 | ||
|
|
8a4bd5f22c | ||
|
|
138a36c69a | ||
|
|
3400ca5d42 | ||
|
|
5d8164da4e | ||
|
|
8b89b30a6f | ||
|
|
e66d7cfd6c | ||
|
|
63afa29dae | ||
|
|
926a9b40e2 | ||
|
|
9bb43264b5 | ||
|
|
32d6daf558 | ||
|
|
bebcb73c68 | ||
|
|
d0e040514f | ||
|
|
850c027d52 | ||
|
|
5df90563e2 | ||
|
|
7a489d18c6 | ||
|
|
1db092e96f | ||
|
|
65a4de221b | ||
|
|
9c8df79dfb | ||
|
|
689b461d7b | ||
|
|
92c951c81f | ||
|
|
9c8438f264 | ||
|
|
caf7ad53e6 | ||
|
|
47f0fec2f2 | ||
|
|
eadb502059 | ||
|
|
8e1695afa3 | ||
|
|
a7138f26b6 | ||
|
|
f2a9ce9d4c | ||
|
|
84e4960e52 | ||
|
|
99afd876ba | ||
|
|
1caa31c5cb | ||
|
|
d2c82ba707 | ||
|
|
9067f3513a | ||
|
|
409d691877 | ||
|
|
1762ff0393 | ||
|
|
4d26178f25 | ||
|
|
c6582a1ce5 | ||
|
|
a51be56c9e | ||
|
|
e5cb583cc0 | ||
|
|
3ef83a9e23 | ||
|
|
7de6a5cb07 | ||
|
|
e06b3a4186 | ||
|
|
c495f82f4f | ||
|
|
49e4426ed5 | ||
|
|
c4e2436885 | ||
|
|
f078b25c7d | ||
|
|
e5149fba57 | ||
|
|
96055cbde7 | ||
|
|
4abac0cac7 | ||
|
|
1e1bcc4af2 | ||
|
|
f1d7879365 | ||
|
|
0ea3de95e0 | ||
|
|
33cef7c8cd | ||
|
|
b0fd220f3e | ||
|
|
1e3539537d | ||
|
|
1b13a3dd4e | ||
|
|
d1ec242e4f | ||
|
|
21f9841e9d | ||
|
|
ea6c05a46a | ||
|
|
226f5e2f23 | ||
|
|
a82fcdecd7 | ||
|
|
27acbe305d | ||
|
|
cd0739a534 | ||
|
|
f1055d0713 | ||
|
|
7875b20594 | ||
|
|
25679cd319 | ||
|
|
fa45ec32db | ||
|
|
40d12c4d0e | ||
|
|
5e706dff64 | ||
|
|
a4860d6f87 | ||
|
|
ef4c4f6e9b | ||
|
|
abbd65549a | ||
|
|
00ef1baead | ||
|
|
75687070a5 | ||
|
|
9b1496c327 | ||
|
|
27cce9d9b7 | ||
|
|
08b66bf827 | ||
|
|
0aa5807482 | ||
|
|
7a2d8c5c01 | ||
|
|
1b8b1a24b0 | ||
|
|
86eb2b13b9 | ||
|
|
c634c53434 | ||
|
|
2eb7a9ff28 | ||
|
|
933c65d805 | ||
|
|
df0ecad15b | ||
|
|
ce88f5f948 | ||
|
|
27cd399d6f | ||
|
|
7b70925acf | ||
|
|
2a3e0b93e8 | ||
|
|
2b26d0fff5 | ||
|
|
129f676610 | ||
|
|
2dc92c122e | ||
|
|
0db17bd58a | ||
|
|
aac16493f1 | ||
|
|
86e5e1a15e | ||
|
|
aa5d2ff31c | ||
|
|
30dd5f5c75 | ||
|
|
f5bc06476a | ||
|
|
c70e44cf05 | ||
|
|
ede13e37d9 | ||
|
|
a7dada457a | ||
|
|
6367554d30 | ||
|
|
81969a684e | ||
|
|
ee339b5960 | ||
|
|
881c940693 | ||
|
|
900c62fa7b | ||
|
|
200c6c054f | ||
|
|
bc0927b7b1 | ||
|
|
231c28395c | ||
|
|
64a45c0d29 | ||
|
|
cab02be637 | ||
|
|
74f341bc0e | ||
|
|
feaa1af1a8 | ||
|
|
d59d040b4e | ||
|
|
a9c26cbf71 | ||
|
|
f4b5c4e69a | ||
|
|
b8cac9f7d5 | ||
|
|
13974df204 | ||
|
|
fa6fe9bf06 | ||
|
|
746be0824e | ||
|
|
93120cabbb | ||
|
|
04ae05f4ce | ||
|
|
83773dddc7 | ||
|
|
9813553f02 | ||
|
|
924723d433 | ||
|
|
33558e63c4 | ||
|
|
97c229d5eb | ||
|
|
16b007df33 | ||
|
|
3cbc421c7e | ||
|
|
890e5e1f0f | ||
|
|
6b98454f03 | ||
|
|
b05329c8df | ||
|
|
6f43c8ffac | ||
|
|
ffac98051f | ||
|
|
40812efaae | ||
|
|
3429321d59 | ||
|
|
6cddd6cbe7 | ||
|
|
23d07d7d0c | ||
|
|
1351575713 | ||
|
|
8aa7d1a278 | ||
|
|
3383786205 | ||
|
|
91f4c54768 | ||
|
|
d9c779289c | ||
|
|
5631ff4fd5 | ||
|
|
34301319bf | ||
|
|
8eac3198b6 | ||
|
|
832edd4da3 | ||
|
|
5823e74bcd | ||
|
|
a4545f493e | ||
|
|
3b8cd44ca4 | ||
|
|
f138d7d9b8 | ||
|
|
3bb9d44bf5 | ||
|
|
06e6a1b19e | ||
|
|
256d166126 | ||
|
|
630285c589 | ||
|
|
9621eca677 | ||
|
|
7eee50d929 | ||
|
|
1dc22e33ae | ||
|
|
0ef8aaebeb | ||
|
|
d17c427e47 | ||
|
|
f73fb62c6e | ||
|
|
5976b712ca | ||
|
|
63f5e64adb | ||
|
|
34ee3bb8aa | ||
|
|
79c745929a | ||
|
|
8330cc6876 | ||
|
|
16e6163677 | ||
|
|
bbbc0dc9dc | ||
|
|
ea4004ec9d | ||
|
|
a469047e7a | ||
|
|
2cc0a867a9 |
No files matched your search
@@ -8,12 +8,6 @@
|
||||
[submodule "External/imgui"]
|
||||
path = External/imgui
|
||||
url = https://github.com/Sonicadvance1/imgui.git
|
||||
[submodule "External/json-maker"]
|
||||
path = External/json-maker
|
||||
url = https://github.com/Sonicadvance1/json-maker.git
|
||||
[submodule "External/tiny-json"]
|
||||
path = External/tiny-json
|
||||
url = https://github.com/Sonicadvance1/tiny-json.git
|
||||
[submodule "External/xbyak"]
|
||||
shallow = true
|
||||
path = External/xbyak
|
||||
|
||||
+11
-9
@@ -7,7 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
|
||||
set(USE_FEXCONFIG_TOOLKIT "imgui" CACHE STRING "If set, build FEXConfig (qt or imgui)")
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
@@ -28,6 +28,7 @@ option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
|
||||
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
@@ -48,7 +49,7 @@ endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
message (STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
set (CLANG_MINIMUM_VERSION 12.0)
|
||||
set (CLANG_MINIMUM_VERSION 13.0)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_LESS ${CLANG_MINIMUM_VERSION})
|
||||
message (FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
endif()
|
||||
@@ -303,11 +304,10 @@ endif()
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
include_directories(External/imgui/)
|
||||
|
||||
add_subdirectory(External/json-maker/)
|
||||
include_directories(External/json-maker/)
|
||||
if (USE_FEXCONFIG_TOOLKIT STREQUAL "imgui")
|
||||
add_subdirectory(External/imgui/)
|
||||
include_directories(External/imgui/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/tiny-json/)
|
||||
include_directories(External/tiny-json/)
|
||||
@@ -423,8 +423,10 @@ add_subdirectory(FEXHeaderUtils/)
|
||||
add_subdirectory(CodeEmitter/)
|
||||
add_subdirectory(FEXCore/)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
if (NOT MINGW_BUILD)
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
# This is a reference AArch64 cross compile script
|
||||
# Pass in to cmake when building:
|
||||
# eg: cmake -DCMAKE_TOOLCHAIN_FILE=../CMakeToolchains/AArch64.cmake ..
|
||||
if (NOT DEFINED ENV{SYSROOT})
|
||||
message(FATAL_ERROR "Need to have SYSROOT environment variable set")
|
||||
endif()
|
||||
|
||||
set(CMAKE_SYSTEM_NAME Linux)
|
||||
set(CMAKE_SYSTEM_PROCESSOR aarch64)
|
||||
set(CMAKE_CROSSCOMPILING TRUE)
|
||||
|
||||
# Target triple needs to match the binutils exactly
|
||||
set(TARGET_TRIPLE aarch64-linux-gnu)
|
||||
set(CMAKE_C_COMPILER "clang")
|
||||
set(CMAKE_CXX_COMPILER "clang++")
|
||||
set(CMAKE_C_COMPILER_AR "llvm-ar")
|
||||
set(CMAKE_CXX_COMPILER_AR "llvm-ar")
|
||||
set(CMAKE_C_COMPILER_RANLIB "llvm-ranlib")
|
||||
set(CMAKE_CXX_COMPILER_RANLIB "llvm-ranlib")
|
||||
set(CMAKE_LINKER "ld.lld")
|
||||
|
||||
set(CMAKE_C_COMPILER_TARGET ${TARGET_TRIPLE})
|
||||
set(CMAKE_CXX_COMPILER_TARGET ${TARGET_TRIPLE})
|
||||
|
||||
# Set the environment variable SYSROOT to the aarch64 rootfs
|
||||
set(CMAKE_FIND_ROOT_PATH "$ENV{SYSROOT}")
|
||||
set(CMAKE_SYSROOT "$ENV{SYSROOT}")
|
||||
|
||||
list(APPEND CMAKE_PREFIX_PATH "$ENV{SYSROOT}/usr/lib/${TARGET_TRIPLE}/cmake/")
|
||||
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_PROGRAM NEVER)
|
||||
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_LIBRARY ONLY)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_INCLUDE ONLY)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_PACKAGE ONLY)
|
||||
@@ -13,5 +13,14 @@ function(GenBinFmt Name)
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
|
||||
endfunction()
|
||||
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
if (NOT USE_LEGACY_BINFMTMISC)
|
||||
configure_file(FEX-x86.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf)
|
||||
configure_file(FEX-x86_64.conf.in ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf)
|
||||
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
|
||||
else()
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
endif()
|
||||
@@ -0,0 +1 @@
|
||||
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
@@ -0,0 +1 @@
|
||||
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
Vendored
+1
-1
Submodule External/fmt updated: f5e54359df...0c9fce2ffe.
Vendored
-1
Submodule External/json-maker deleted from 8ecb8ecc34.
Vendored
+1
-1
Submodule External/robin-map updated: f1ab690046...d5683d9f18.
Vendored
-1
Submodule External/tiny-json deleted from 9d09127f87.
Vendored
+3
@@ -0,0 +1,3 @@
|
||||
set(NAME tiny-json)
|
||||
set(SRCS tiny-json.c)
|
||||
add_library(${NAME} ${SRCS})
|
||||
Vendored
+21
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2018 Rafa Garcia
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
Vendored
+647
@@ -0,0 +1,647 @@
|
||||
|
||||
/*
|
||||
|
||||
<https://github.com/rafagafe/tiny-json>
|
||||
|
||||
Licensed under the MIT License <http://opensource.org/licenses/MIT>.
|
||||
SPDX-License-Identifier: MIT
|
||||
Copyright (c) 2016-2018 Rafa Garcia <rafagarcia77@gmail.com>.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <ctype.h>
|
||||
#include <stddef.h> // For NULL
|
||||
#include "tiny-json.h"
|
||||
|
||||
/** Structure to handle a heap of JSON properties. */
|
||||
typedef struct jsonStaticPool_s {
|
||||
json_t* const mem; /**< Pointer to array of json properties. */
|
||||
unsigned int const qty; /**< Length of the array of json properties. */
|
||||
unsigned int nextFree; /**< The index of the next free json property. */
|
||||
jsonPool_t pool;
|
||||
} jsonStaticPool_t;
|
||||
|
||||
/* Search a property by its name in a JSON object. */
|
||||
json_t const* json_getProperty( json_t const* obj, char const* property ) {
|
||||
json_t const* sibling;
|
||||
for( sibling = obj->u.c.child; sibling; sibling = sibling->sibling )
|
||||
if ( sibling->name && !strcmp( sibling->name, property ) )
|
||||
return sibling;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Search a property by its name in a JSON object and return its value. */
|
||||
char const* json_getPropertyValue( json_t const* obj, char const* property ) {
|
||||
json_t const* field = json_getProperty( obj, property );
|
||||
if ( !field ) return 0;
|
||||
jsonType_t type = json_getType( field );
|
||||
if ( JSON_ARRAY >= type ) return 0;
|
||||
return json_getValue( field );
|
||||
}
|
||||
|
||||
/* Internal prototypes: */
|
||||
static char* goBlank( char* str );
|
||||
static char* goNum( char* str );
|
||||
static json_t* poolInit( jsonPool_t* pool );
|
||||
static json_t* poolAlloc( jsonPool_t* pool );
|
||||
static char* objValue( char* ptr, json_t* obj, jsonPool_t* pool );
|
||||
static char* setToNull( char* ch );
|
||||
static bool isEndOfPrimitive( char ch );
|
||||
|
||||
/* Parse a string to get a json. */
|
||||
json_t const* json_createWithPool( char *str, jsonPool_t *pool ) {
|
||||
char* ptr = goBlank( str );
|
||||
if ( !ptr || *ptr != '{' ) return 0;
|
||||
json_t* obj = pool->init( pool );
|
||||
obj->name = 0;
|
||||
obj->sibling = 0;
|
||||
obj->u.c.child = 0;
|
||||
ptr = objValue( ptr, obj, pool );
|
||||
if ( !ptr ) return 0;
|
||||
return obj;
|
||||
}
|
||||
|
||||
/* Parse a string to get a json. */
|
||||
json_t const* json_create( char* str, json_t mem[], unsigned int qty ) {
|
||||
jsonStaticPool_t spool = {
|
||||
.mem = mem,
|
||||
.qty = qty,
|
||||
.pool = {
|
||||
.init = poolInit,
|
||||
.alloc = poolAlloc
|
||||
}
|
||||
};
|
||||
return json_createWithPool( str, &spool.pool );
|
||||
}
|
||||
|
||||
/** Get a special character with its escape character. Examples:
|
||||
* 'b' -> '\b', 'n' -> '\n', 't' -> '\t'
|
||||
* @param ch The escape character.
|
||||
* @return The character code. */
|
||||
static char getEscape( char ch ) {
|
||||
static struct { char ch; char code; } const pair[] = {
|
||||
{ '\"', '\"' }, { '\\', '\\' },
|
||||
{ '/', '/' }, { 'b', '\b' },
|
||||
{ 'f', '\f' }, { 'n', '\n' },
|
||||
{ 'r', '\r' }, { 't', '\t' },
|
||||
};
|
||||
unsigned int i;
|
||||
for( i = 0; i < sizeof pair / sizeof *pair; ++i )
|
||||
if ( pair[i].ch == ch )
|
||||
return pair[i].code;
|
||||
return '\0';
|
||||
}
|
||||
|
||||
/** Parse 4 characters.
|
||||
* @Param str Pointer to first digit.
|
||||
* @retval '?' If the four characters are hexadecimal digits.
|
||||
* @retcal '\0' In other cases. */
|
||||
static unsigned char getCharFromUnicode( unsigned char const* str ) {
|
||||
unsigned int i;
|
||||
for( i = 0; i < 4; ++i )
|
||||
if ( !isxdigit( str[i] ) )
|
||||
return '\0';
|
||||
return '?';
|
||||
}
|
||||
|
||||
/** Parse a string and replace the scape characters by their meaning characters.
|
||||
* This parser stops when finds the character '\"'. Then replaces '\"' by '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* parseString( char* str ) {
|
||||
unsigned char* head = (unsigned char*)str;
|
||||
unsigned char* tail = (unsigned char*)str;
|
||||
for( ; *head >= ' '; ++head, ++tail ) {
|
||||
if ( *head == '\"' ) {
|
||||
*tail = '\0';
|
||||
return (char*)++head;
|
||||
}
|
||||
if ( *head == '\\' ) {
|
||||
if ( *++head == 'u' ) {
|
||||
char const ch = getCharFromUnicode( ++head );
|
||||
if ( ch == '\0' ) return 0;
|
||||
*tail = ch;
|
||||
head += 3;
|
||||
}
|
||||
else {
|
||||
char const esc = getEscape( *head );
|
||||
if ( esc == '\0' ) return 0;
|
||||
*tail = esc;
|
||||
}
|
||||
}
|
||||
else *tail = *head;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Parse a string to get the name of a property.
|
||||
* @param str Pointer to first character.
|
||||
* @param property The property to assign the name.
|
||||
* @retval Pointer to first of property value. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* propertyName( char* ptr, json_t* property ) {
|
||||
property->name = ++ptr;
|
||||
ptr = parseString( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
ptr = goBlank( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
if ( *ptr++ != ':' ) return 0;
|
||||
return goBlank( ptr );
|
||||
}
|
||||
|
||||
/** Parse a string to get the value of a property when its type is JSON_TEXT.
|
||||
* @param str Pointer to first character ('\"').
|
||||
* @param property The property to assign the name.
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* textValue( char* ptr, json_t* property ) {
|
||||
++property->u.value;
|
||||
ptr = parseString( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
property->type = JSON_TEXT;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Compare two strings until get the null character in the second one.
|
||||
* @param ptr sub string
|
||||
* @param str main string
|
||||
* @retval Pointer to next character.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* checkStr( char* ptr, char const* str ) {
|
||||
while( *str )
|
||||
if ( *ptr++ != *str++ )
|
||||
return 0;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Parser a string to get a primitive value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @param value String with the primitive literal.
|
||||
* @param type The code of the type. ( JSON_BOOLEAN or JSON_NULL )
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* primitiveValue( char* ptr, json_t* property, char const* value, jsonType_t type ) {
|
||||
ptr = checkStr( ptr, value );
|
||||
if ( !ptr || !isEndOfPrimitive( *ptr ) ) return 0;
|
||||
ptr = setToNull( ptr );
|
||||
property->type = type;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Parser a string to get a true value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* trueValue( char* ptr, json_t* property ) {
|
||||
return primitiveValue( ptr, property, "true", JSON_BOOLEAN );
|
||||
}
|
||||
|
||||
/** Parser a string to get a false value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* falseValue( char* ptr, json_t* property ) {
|
||||
return primitiveValue( ptr, property, "false", JSON_BOOLEAN );
|
||||
}
|
||||
|
||||
/** Parser a string to get a null value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type, (true, false or null).
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* nullValue( char* ptr, json_t* property ) {
|
||||
return primitiveValue( ptr, property, "null", JSON_NULL );
|
||||
}
|
||||
|
||||
/** Analyze the exponential part of a real number.
|
||||
* @param str Pointer to first character.
|
||||
* @retval Pointer to first non numerical after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* expValue( char* ptr ) {
|
||||
if ( *ptr == '-' || *ptr == '+' ) ++ptr;
|
||||
if ( !isdigit( *ptr ) ) return 0;
|
||||
ptr = goNum( ++ptr );
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Analyze the decimal part of a real number.
|
||||
* @param str Pointer to first character.
|
||||
* @retval Pointer to first non numerical after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* fraqValue( char* ptr ) {
|
||||
if ( !isdigit( *ptr ) ) return 0;
|
||||
ptr = goNum( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Parser a string to get a numerical value.
|
||||
* If the first character after the value is different of '}' or ']' is set to '\0'.
|
||||
* @param str Pointer to first character.
|
||||
* @param property Property handler to set the value and the type: JSON_REAL or JSON_INTEGER.
|
||||
* @retval Pointer to first non white space after the string. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* numValue( char* ptr, json_t* property ) {
|
||||
if ( *ptr == '-' ) ++ptr;
|
||||
if ( !isdigit( *ptr ) ) return 0;
|
||||
if ( *ptr != '0' ) {
|
||||
ptr = goNum( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
}
|
||||
else if ( isdigit( *++ptr ) ) return 0;
|
||||
property->type = JSON_INTEGER;
|
||||
if ( *ptr == '.' ) {
|
||||
ptr = fraqValue( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
property->type = JSON_REAL;
|
||||
}
|
||||
if ( *ptr == 'e' || *ptr == 'E' ) {
|
||||
ptr = expValue( ++ptr );
|
||||
if ( !ptr ) return 0;
|
||||
property->type = JSON_REAL;
|
||||
}
|
||||
if ( !isEndOfPrimitive( *ptr ) ) return 0;
|
||||
if ( JSON_INTEGER == property->type ) {
|
||||
char const* value = property->u.value;
|
||||
bool const negative = *value == '-';
|
||||
static char const min[] = "-9223372036854775808";
|
||||
static char const max[] = "9223372036854775807";
|
||||
unsigned int const maxdigits = ( negative? sizeof min: sizeof max ) - 1;
|
||||
unsigned int const len = ptr - value;
|
||||
if ( len > maxdigits ) return 0;
|
||||
if ( len == maxdigits ) {
|
||||
char const tmp = *ptr;
|
||||
*ptr = '\0';
|
||||
char const* const threshold = negative ? min: max;
|
||||
if ( 0 > strcmp( threshold, value ) ) return 0;
|
||||
*ptr = tmp;
|
||||
}
|
||||
}
|
||||
ptr = setToNull( ptr );
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/** Add a property to a JSON object or array.
|
||||
* @param obj The handler of the JSON object or array.
|
||||
* @param property The handler of the property to be added. */
|
||||
static void add( json_t* obj, json_t* property ) {
|
||||
property->sibling = 0;
|
||||
if ( !obj->u.c.child ){
|
||||
obj->u.c.child = property;
|
||||
obj->u.c.last_child = property;
|
||||
} else {
|
||||
obj->u.c.last_child->sibling = property;
|
||||
obj->u.c.last_child = property;
|
||||
}
|
||||
}
|
||||
|
||||
/** Parser a string to get a json object value.
|
||||
* @param str Pointer to first character.
|
||||
* @param pool The handler of a json pool for creating json instances.
|
||||
* @retval Pointer to first character after the value. If success.
|
||||
* @retval Null pointer if any error occur. */
|
||||
static char* objValue( char* ptr, json_t* obj, jsonPool_t* pool ) {
|
||||
obj->type = JSON_OBJ;
|
||||
obj->u.c.child = 0;
|
||||
obj->sibling = 0;
|
||||
ptr++;
|
||||
for(;;) {
|
||||
ptr = goBlank( ptr );
|
||||
if ( !ptr ) return 0;
|
||||
if ( *ptr == ',' ) {
|
||||
++ptr;
|
||||
continue;
|
||||
}
|
||||
char const endchar = ( obj->type == JSON_OBJ )? '}': ']';
|
||||
if ( *ptr == endchar ) {
|
||||
*ptr = '\0';
|
||||
json_t* parentObj = obj->sibling;
|
||||
if ( !parentObj ) return ++ptr;
|
||||
obj->sibling = 0;
|
||||
obj = parentObj;
|
||||
++ptr;
|
||||
continue;
|
||||
}
|
||||
json_t* property = pool->alloc( pool );
|
||||
if ( !property ) return 0;
|
||||
if( obj->type != JSON_ARRAY ) {
|
||||
if ( *ptr != '\"' ) return 0;
|
||||
ptr = propertyName( ptr, property );
|
||||
if ( !ptr ) return 0;
|
||||
}
|
||||
else property->name = 0;
|
||||
add( obj, property );
|
||||
property->u.value = ptr;
|
||||
switch( *ptr ) {
|
||||
case '{':
|
||||
property->type = JSON_OBJ;
|
||||
property->u.c.child = 0;
|
||||
property->sibling = obj;
|
||||
obj = property;
|
||||
++ptr;
|
||||
break;
|
||||
case '[':
|
||||
property->type = JSON_ARRAY;
|
||||
property->u.c.child = 0;
|
||||
property->sibling = obj;
|
||||
obj = property;
|
||||
++ptr;
|
||||
break;
|
||||
case '\"': ptr = textValue( ptr, property ); break;
|
||||
case 't': ptr = trueValue( ptr, property ); break;
|
||||
case 'f': ptr = falseValue( ptr, property ); break;
|
||||
case 'n': ptr = nullValue( ptr, property ); break;
|
||||
default: ptr = numValue( ptr, property ); break;
|
||||
}
|
||||
if ( !ptr ) return 0;
|
||||
}
|
||||
}
|
||||
|
||||
/** Initialize a json pool.
|
||||
* @param pool The handler of the pool.
|
||||
* @return a instance of a json. */
|
||||
static json_t* poolInit( jsonPool_t* pool ) {
|
||||
jsonStaticPool_t *spool = json_containerOf( pool, jsonStaticPool_t, pool );
|
||||
spool->nextFree = 1;
|
||||
return spool->mem;
|
||||
}
|
||||
|
||||
/** Create an instance of a json from a pool.
|
||||
* @param pool The handler of the pool.
|
||||
* @retval The handler of the new instance if success.
|
||||
* @retval Null pointer if the pool was empty. */
|
||||
static json_t* poolAlloc( jsonPool_t* pool ) {
|
||||
jsonStaticPool_t *spool = json_containerOf( pool, jsonStaticPool_t, pool );
|
||||
if ( spool->nextFree >= spool->qty ) return 0;
|
||||
return spool->mem + spool->nextFree++;
|
||||
}
|
||||
|
||||
/** Checks whether an character belongs to set.
|
||||
* @param ch Character value to be checked.
|
||||
* @param set Set of characters. It is just a null-terminated string.
|
||||
* @return true or false there is membership or not. */
|
||||
static bool isOneOfThem( char ch, char const* set ) {
|
||||
while( *set != '\0' )
|
||||
if ( ch == *set++ )
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Increases a pointer while it points to a character that belongs to a set.
|
||||
* @param str The initial pointer value.
|
||||
* @param set Set of characters. It is just a null-terminated string.
|
||||
* @return The final pointer value or null pointer if the null character was found. */
|
||||
static char* goWhile( char* str, char const* set ) {
|
||||
for(; *str != '\0'; ++str ) {
|
||||
if ( !isOneOfThem( *str, set ) )
|
||||
return str;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Set of characters that defines a blank. */
|
||||
static char const* const blank = " \n\r\t\f";
|
||||
|
||||
/** Increases a pointer while it points to a white space character.
|
||||
* @param str The initial pointer value.
|
||||
* @return The final pointer value or null pointer if the null character was found. */
|
||||
static char* goBlank( char* str ) {
|
||||
return goWhile( str, blank );
|
||||
}
|
||||
|
||||
/** Increases a pointer while it points to a decimal digit character.
|
||||
* @param str The initial pointer value.
|
||||
* @return The final pointer value or null pointer if the null character was found. */
|
||||
static char* goNum( char* str ) {
|
||||
for( ; *str != '\0'; ++str ) {
|
||||
if ( !isdigit( *str ) )
|
||||
return str;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Set of characters that defines the end of an array or a JSON object. */
|
||||
static char const* const endofblock = "}]";
|
||||
|
||||
/** Set a char to '\0' and increase its pointer if the char is different to '}' or ']'.
|
||||
* @param ch Pointer to character.
|
||||
* @return Final value pointer. */
|
||||
static char* setToNull( char* ch ) {
|
||||
if ( !isOneOfThem( *ch, endofblock ) ) *ch++ = '\0';
|
||||
return ch;
|
||||
}
|
||||
|
||||
/** Indicate if a character is the end of a primitive value. */
|
||||
static bool isEndOfPrimitive( char ch ) {
|
||||
return ch == ',' || isOneOfThem( ch, blank ) || isOneOfThem( ch, endofblock );
|
||||
}
|
||||
|
||||
/** Add a character at the end of a string.
|
||||
* @param dest Pointer to the null character of the string
|
||||
* @param ch Value to be added.
|
||||
* @return Pointer to the null character of the destination string. */
|
||||
static char* chtoa( char* dest, char ch ) {
|
||||
*dest = ch;
|
||||
*++dest = '\0';
|
||||
return dest;
|
||||
}
|
||||
|
||||
/** Copy a null-terminated string.
|
||||
* @param dest Destination memory block.
|
||||
* @param src Source string.
|
||||
* @return Pointer to the null character of the destination string. */
|
||||
static char* atoa( char* dest, char const* src ) {
|
||||
for( ; *src != '\0'; ++dest, ++src )
|
||||
*dest = *src;
|
||||
*dest = '\0';
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Open a JSON object in a JSON string. */
|
||||
char* json_objOpen( char* dest, char const* name ) {
|
||||
if ( NULL == name )
|
||||
dest = chtoa( dest, '{' );
|
||||
else {
|
||||
dest = chtoa( dest, '\"' );
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":{" );
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Close a JSON object in a JSON string. */
|
||||
char* json_objClose( char* dest ) {
|
||||
if ( dest[-1] == ',' )
|
||||
--dest;
|
||||
return atoa( dest, "}," );
|
||||
}
|
||||
|
||||
/* Open an array in a JSON string. */
|
||||
char* json_arrOpen( char* dest, char const* name ) {
|
||||
if ( NULL == name )
|
||||
dest = chtoa( dest, '[' );
|
||||
else {
|
||||
dest = chtoa( dest, '\"' );
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":[" );
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Close an array in a JSON string. */
|
||||
char* json_arrClose( char* dest ) {
|
||||
if ( dest[-1] == ',' )
|
||||
--dest;
|
||||
return atoa( dest, "]," );
|
||||
}
|
||||
|
||||
/** Add the name of a text property.
|
||||
* @param dest Destination memory.
|
||||
* @param name The name of the property.
|
||||
* @return Pointer to the next char. */
|
||||
static char* strname( char* dest, char const* name ) {
|
||||
dest = chtoa( dest, '\"' );
|
||||
if ( NULL != name ) {
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":\"" );
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
/** Get the hexadecimal digit of the least significant nibble of a integer. */
|
||||
static int nibbletoch( int nibble ) {
|
||||
return "0123456789ABCDEF"[ nibble % 16u ];
|
||||
}
|
||||
|
||||
/** Get the escape character of a non-printable.
|
||||
* @param ch Character source.
|
||||
* @return The escape character or null character if error. */
|
||||
static int escape( int ch ) {
|
||||
static struct { char code; char ch; } const pair[] = {
|
||||
{ '\"', '\"' }, { '\\', '\\' }, { '/', '/' }, { 'b', '\b' },
|
||||
{ 'f', '\f' }, { 'n', '\n' }, { 'r', '\r' }, { 't', '\t' },
|
||||
};
|
||||
for( int i = 0; i < sizeof pair / sizeof *pair; ++i )
|
||||
if ( ch == pair[i].ch )
|
||||
return pair[i].code;
|
||||
return '\0';
|
||||
}
|
||||
|
||||
/** Copy a null-terminated string inserting escape characters if needed.
|
||||
* @param dest Destination memory block.
|
||||
* @param src Source string.
|
||||
* @return Pointer to the null character of the destination string. */
|
||||
static char* atoesc( char* dest, char const* src ) {
|
||||
for( ; *src != '\0'; ++dest, ++src ) {
|
||||
if ( *src >= ' ' && *src != '\"' && *src != '\\' && *src != '/' )
|
||||
*dest = *src;
|
||||
else {
|
||||
*dest++ = '\\';
|
||||
int const esc = escape( *src );
|
||||
if ( esc )
|
||||
*dest = esc;
|
||||
else {
|
||||
*dest++ = 'u';
|
||||
*dest++ = '0';
|
||||
*dest++ = '0';
|
||||
*dest++ = nibbletoch( *src / 16 );
|
||||
*dest++ = nibbletoch( *src );
|
||||
}
|
||||
}
|
||||
}
|
||||
*dest = '\0';
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Add a text property in a JSON string. */
|
||||
char* json_str( char* dest, char const* name, char const* value ) {
|
||||
dest = strname( dest, name );
|
||||
dest = atoesc( dest, value );
|
||||
dest = atoa( dest, "\"," );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/** Add the name of a primitive property.
|
||||
* @param dest Destination memory.
|
||||
* @param name The name of the property.
|
||||
* @return Pointer to the next char. */
|
||||
static char* primitivename( char* dest, char const* name ) {
|
||||
if( NULL == name )
|
||||
return dest;
|
||||
dest = chtoa( dest, '\"' );
|
||||
dest = atoa( dest, name );
|
||||
dest = atoa( dest, "\":" );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Add a boolean property in a JSON string. */
|
||||
char* json_bool( char* dest, char const* name, int value ) {
|
||||
dest = primitivename( dest, name );
|
||||
dest = atoa( dest, value ? "true," : "false," );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Add a null property in a JSON string. */
|
||||
char* json_null( char* dest, char const* name ) {
|
||||
dest = primitivename( dest, name );
|
||||
dest = atoa( dest, "null," );
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* Used to finish the root JSON object. After call json_objClose(). */
|
||||
char* json_end( char* dest ) {
|
||||
if ( ',' == dest[-1] ) {
|
||||
dest[-1] = '\0';
|
||||
--dest;
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
#define ALL_TYPES \
|
||||
X( json_int, int, "%d" ) \
|
||||
X( json_long, long, "%ld" ) \
|
||||
X( json_uint, unsigned int, "%u" ) \
|
||||
X( json_ulong, unsigned long, "%lu" ) \
|
||||
X( json_verylong, long long, "%lld" ) \
|
||||
X( json_double, double, "%g" ) \
|
||||
|
||||
|
||||
#define json_num( funcname, type, fmt ) \
|
||||
char* funcname( char* dest, char const* name, type value ) { \
|
||||
dest = primitivename( dest, name ); \
|
||||
dest += sprintf( dest, fmt, value ); \
|
||||
dest = chtoa( dest, ',' ); \
|
||||
return dest; \
|
||||
}
|
||||
|
||||
#define X( name, type, fmt ) json_num( name, type, fmt )
|
||||
ALL_TYPES
|
||||
#undef X
|
||||
Vendored
+270
@@ -0,0 +1,270 @@
|
||||
|
||||
/*
|
||||
|
||||
<https://github.com/rafagafe/tiny-json>
|
||||
|
||||
Licensed under the MIT License <http://opensource.org/licenses/MIT>.
|
||||
SPDX-License-Identifier: MIT
|
||||
Copyright (c) 2016-2018 Rafa Garcia <rafagarcia77@gmail.com>.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
*/
|
||||
|
||||
#ifndef _TINY_JSON_H_
|
||||
#define _TINY_JSON_H_
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdlib.h>
|
||||
#include <stdbool.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#define json_containerOf( ptr, type, member ) \
|
||||
((type*)( (char*)ptr - offsetof( type, member ) ))
|
||||
|
||||
/** @defgroup tinyJson Tiny JSON parser.
|
||||
* @{ */
|
||||
|
||||
/** Enumeration of codes of supported JSON properties types. */
|
||||
typedef enum {
|
||||
JSON_OBJ, JSON_ARRAY, JSON_TEXT, JSON_BOOLEAN,
|
||||
JSON_INTEGER, JSON_REAL, JSON_NULL
|
||||
} jsonType_t;
|
||||
|
||||
/** Structure to handle JSON properties. */
|
||||
typedef struct json_s {
|
||||
struct json_s* sibling;
|
||||
const char* name;
|
||||
union {
|
||||
const char* value;
|
||||
struct {
|
||||
struct json_s* child;
|
||||
struct json_s* last_child;
|
||||
} c;
|
||||
} u;
|
||||
jsonType_t type;
|
||||
} json_t;
|
||||
|
||||
/** Parse a string to get a json.
|
||||
* @param str String pointer with a JSON object. It will be modified.
|
||||
* @param mem Array of json properties to allocate.
|
||||
* @param qty Number of elements of mem.
|
||||
* @retval Null pointer if any was wrong in the parse process.
|
||||
* @retval If the parser process was successfully a valid handler of a json.
|
||||
* This property is always unnamed and its type is JSON_OBJ. */
|
||||
const json_t* json_create(char* str, json_t mem[], unsigned int qty);
|
||||
|
||||
/** Get the name of a json property.
|
||||
* @param json A valid handler of a json property.
|
||||
* @retval Pointer to null-terminated if property has name.
|
||||
* @retval Null pointer if the property is unnamed. */
|
||||
static inline const char* json_getName(const json_t* json) {
|
||||
return json->name;
|
||||
}
|
||||
|
||||
/** Get the value of a json property.
|
||||
* The type of property cannot be JSON_OBJ or JSON_ARRAY.
|
||||
* @param json A valid handler of a json property.
|
||||
* @return Pointer to null-terminated string with the value. */
|
||||
static inline const char* json_getValue(const json_t* property) {
|
||||
return property->u.value;
|
||||
}
|
||||
|
||||
/** Get the type of a json property.
|
||||
* @param json A valid handler of a json property.
|
||||
* @return The code of type.*/
|
||||
static inline jsonType_t json_getType(const json_t* json) {
|
||||
return json->type;
|
||||
}
|
||||
|
||||
/** Get the next sibling of a JSON property that is within a JSON object or array.
|
||||
* @param json A valid handler of a json property.
|
||||
* @retval The handler of the next sibling if found.
|
||||
* @retval Null pointer if the json property is the last one. */
|
||||
static inline const json_t* json_getSibling(const json_t* json) {
|
||||
return json->sibling;
|
||||
}
|
||||
|
||||
/** Search a property by its name in a JSON object.
|
||||
* @param obj A valid handler of a json object. Its type must be JSON_OBJ.
|
||||
* @param property The name of property to get.
|
||||
* @retval The handler of the json property if found.
|
||||
* @retval Null pointer if not found. */
|
||||
const json_t* json_getProperty(const json_t* obj, const char* property);
|
||||
|
||||
|
||||
/** Search a property by its name in a JSON object and return its value.
|
||||
* @param obj A valid handler of a json object. Its type must be JSON_OBJ.
|
||||
* @param property The name of property to get.
|
||||
* @retval If found a pointer to null-terminated string with the value.
|
||||
* @retval Null pointer if not found or it is an array or an object. */
|
||||
const char* json_getPropertyValue(const json_t* obj, const char* property);
|
||||
|
||||
/** Get the first property of a JSON object or array.
|
||||
* @param json A valid handler of a json property.
|
||||
* Its type must be JSON_OBJ or JSON_ARRAY.
|
||||
* @retval The handler of the first property if there is.
|
||||
* @retval Null pointer if the json object has not properties. */
|
||||
static inline const json_t* json_getChild(const json_t* json) {
|
||||
return json->u.c.child;
|
||||
}
|
||||
|
||||
/** Get the value of a json boolean property.
|
||||
* @param property A valid handler of a json object. Its type must be JSON_BOOLEAN.
|
||||
* @return The value stdbool. */
|
||||
static inline bool json_getBoolean(const json_t* property) {
|
||||
return *property->u.value == 't';
|
||||
}
|
||||
|
||||
/** Get the value of a json integer property.
|
||||
* @param property A valid handler of a json object. Its type must be JSON_INTEGER.
|
||||
* @return The value stdint. */
|
||||
static inline int64_t json_getInteger(const json_t* property) {
|
||||
return atoll( property->u.value );
|
||||
}
|
||||
|
||||
/** Get the value of a json real property.
|
||||
* @param property A valid handler of a json object. Its type must be JSON_REAL.
|
||||
* @return The value. */
|
||||
static inline double json_getReal(const json_t* property) {
|
||||
return atof( property->u.value );
|
||||
}
|
||||
|
||||
|
||||
/** Structure to handle a heap of JSON properties. */
|
||||
typedef struct jsonPool_s jsonPool_t;
|
||||
struct jsonPool_s {
|
||||
json_t* (*init)( jsonPool_t* pool );
|
||||
json_t* (*alloc)( jsonPool_t* pool );
|
||||
};
|
||||
|
||||
/** Parse a string to get a json.
|
||||
* @param str String pointer with a JSON object. It will be modified.
|
||||
* @param pool Custom json pool pointer.
|
||||
* @retval Null pointer if any was wrong in the parse process.
|
||||
* @retval If the parser process was successfully a valid handler of a json.
|
||||
* This property is always unnamed and its type is JSON_OBJ. */
|
||||
const json_t* json_createWithPool(char* str, jsonPool_t* pool);
|
||||
|
||||
/** @ } */
|
||||
|
||||
/** @defgroup makejoson Make JSON.
|
||||
* @{ */
|
||||
|
||||
/** Open a JSON object in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_objOpen(char* dest, const char* name);
|
||||
|
||||
/** Close a JSON object in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_objClose(char* dest);
|
||||
|
||||
/** Used to finish the root JSON object. After call json_objClose().
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_end(char* dest);
|
||||
|
||||
/** Open an array in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_arrOpen(char* dest, const char* name);
|
||||
|
||||
/** Close an array in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_arrClose(char* dest);
|
||||
|
||||
/** Add a text property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value A valid null-terminated string with the value.
|
||||
* Backslash escapes will be added for special characters.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_str(char* dest, const char* name, const char* value);
|
||||
|
||||
/** Add a boolean property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Zero for false. Non zero for true.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_bool(char* dest, const char* name, int value);
|
||||
|
||||
/** Add a null property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_null(char* dest, const char* name);
|
||||
|
||||
/** Add an integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_int(char* dest, const char* name, int value);
|
||||
|
||||
/** Add an unsigned integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_uint(char* dest, const char* name, unsigned int value);
|
||||
|
||||
/** Add a long integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_long(char* dest, const char* name, long int value);
|
||||
|
||||
/** Add an unsigned long integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_ulong(char* dest, const char* name, unsigned long int value);
|
||||
|
||||
/** Add a long long integer property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_verylong(char* dest, const char* name, long long int value);
|
||||
|
||||
/** Add a double precision number property in a JSON string.
|
||||
* @param dest Pointer to the end of JSON under construction.
|
||||
* @param name Pointer to null-terminated string or null for unnamed.
|
||||
* @param value Value of the property.
|
||||
* @return Pointer to the new end of JSON under construction. */
|
||||
char* json_double(char* dest, const char* name, double value);
|
||||
|
||||
/** @ } */
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* _TINY_JSON_H_ */
|
||||
@@ -217,6 +217,14 @@ def print_man_environment_tail():
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored. These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
|
||||
],
|
||||
"''", True)
|
||||
|
||||
def print_man_header():
|
||||
header ='''.Dd {0}
|
||||
.Dt FEX
|
||||
|
||||
@@ -125,21 +125,36 @@ def parse_ops(ops):
|
||||
|
||||
RHS = EqualSplit[0].strip()
|
||||
if len(EqualSplit) > 1:
|
||||
OpDef.HasDest = True
|
||||
LHS = EqualSplit[0].strip()
|
||||
RHS = EqualSplit[1].strip()
|
||||
|
||||
# Parse the destination, must be one type of SSA, GPR, or FPR
|
||||
ResultType = EqualSplit[0].strip()
|
||||
if ResultType == "SSA":
|
||||
OpDef.DestType = "SSA" # We don't know this type right now
|
||||
elif ResultType == "GPR":
|
||||
OpDef.DestType = "GPR"
|
||||
elif ResultType == "GPRPair":
|
||||
OpDef.DestType = "GPRPair"
|
||||
elif ResultType == "FPR":
|
||||
OpDef.DestType = "FPR"
|
||||
if ":" in LHS:
|
||||
# Named destinations. This is a hack, but so is the entire
|
||||
# multi-destination support bolten onto the old IR...
|
||||
#
|
||||
# Named destinations require side effects because they break
|
||||
# SSA hard. Validate that.
|
||||
assert("HasSideEffects" in op_val and op_val["HasSideEffects"])
|
||||
|
||||
for Dest in LHS.split(","):
|
||||
Dest = Dest.strip()
|
||||
DType, Name = Dest.split(":$")
|
||||
|
||||
# If the destination appears also as a source, it is
|
||||
# read-modify-write.
|
||||
if Dest in RHS:
|
||||
# Turn RMW into an in/out source
|
||||
RHS = RHS.replace(Dest.strip(), f"{DType}:$Inout{Name}")
|
||||
else:
|
||||
# Turn named destinations into an out source.
|
||||
RHS += f", {DType}:$Out{Name}"
|
||||
else:
|
||||
ExitError("Unknown destination class type {}. Needs to be one of {SSA, GPR, GPRPair, FPR}".format(ResultType))
|
||||
# Single anonymous destination
|
||||
if LHS not in ["SSA", "GPR", "GPRPair", "FPR"]:
|
||||
ExitError(f"Unknown destination class type {LHS}. Needs to be one of SSA, GPR, GPRPair, FPR")
|
||||
|
||||
OpDef.HasDest = True
|
||||
OpDef.DestType = LHS
|
||||
|
||||
# IR Op needs to start with a name
|
||||
RHS = RHS.split(" ", 1)
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
include(GNUInstallDirs)
|
||||
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (FEXCORE_BASE_SRCS
|
||||
@@ -181,7 +182,11 @@ endif()
|
||||
# Some defines for the softfloat library
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
|
||||
|
||||
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
set (LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
|
||||
if (ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
list (APPEND LIBS vixl)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
@@ -369,11 +374,8 @@ AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
install(TARGETS ${PROJECT_NAME} ${PROJECT_NAME}_shared
|
||||
install(TARGETS ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION lib
|
||||
COMPONENT Libraries
|
||||
ARCHIVE
|
||||
DESTINATION lib
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
COMPONENT Libraries)
|
||||
endif()
|
||||
@@ -142,7 +142,7 @@
|
||||
},
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks/",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks/",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
@@ -158,7 +158,7 @@
|
||||
},
|
||||
"ThunkHostLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/lib/fex-emu/HostThunks_32/",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit host-side thunking libraries."
|
||||
]
|
||||
@@ -422,6 +422,14 @@
|
||||
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
|
||||
]
|
||||
},
|
||||
"StrictInProcessSplitLocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Strict global lock when handling an unaligned atomic that crosses a 16-byte or cacheline granularity",
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
@@ -509,6 +517,13 @@
|
||||
"Desc": [
|
||||
"Override for a FEXServer socket path. Only useful for chroots."
|
||||
]
|
||||
},
|
||||
"NeedsSeccomp": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Disables inline syscalls in order to support seccomp handling"
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
@@ -76,7 +76,6 @@ struct ExitFunctionLinkData {
|
||||
};
|
||||
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE = 128;
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
@@ -206,6 +205,7 @@ public:
|
||||
struct {
|
||||
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
uint64_t TSCScale = 0;
|
||||
|
||||
// Used if the JIT needs to have its interrupt fault code emitted.
|
||||
bool NeedsPendingInterruptFaultCheck {false};
|
||||
@@ -216,6 +216,8 @@ public:
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
|
||||
@@ -237,13 +239,15 @@ public:
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
} Config;
|
||||
|
||||
|
||||
std::atomic_bool CoreShuttingDown {false};
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
uint32_t StrictSplitLockMutex {};
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
// CPUID depends on HostFeatures so needs to be initialized after that.
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
@@ -326,16 +330,6 @@ public:
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
void GetVDSOSigReturn(VDSOSigReturn* VDSOPointers) override {
|
||||
if (VDSOPointers->VDSO_kernel_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_sigreturn = reinterpret_cast<void*>(X86CodeGen.sigreturn_32);
|
||||
}
|
||||
|
||||
if (VDSOPointers->VDSO_kernel_rt_sigreturn == nullptr) {
|
||||
VDSOPointers->VDSO_kernel_rt_sigreturn = reinterpret_cast<void*>(X86CodeGen.rt_sigreturn_32);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
@@ -344,27 +338,21 @@ public:
|
||||
return AtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
// If atomic-based TSO emulation is enabled for vector operations.
|
||||
bool IsVectorAtomicTSOEnabled() const {
|
||||
return VectorAtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
// If atomic-based TSO emulation is enabled for memcpy operations.
|
||||
bool IsMemcpyAtomicTSOEnabled() const {
|
||||
return MemcpyAtomicTSOEmulationEnabled;
|
||||
}
|
||||
|
||||
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
|
||||
SupportsHardwareTSO = HardwareTSOSupported;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
// Returns if Software TSO emulation is required.
|
||||
// NOTE: This doesn't necessary return if Atomic-based TSO is currently enabled.
|
||||
// This will still return true if on a single thread and TSO is currently disabled.
|
||||
//
|
||||
// This is to ensure that if early initialization checks CPU features and TSO /could/ be enabled, that
|
||||
// we return consistent results.
|
||||
//
|
||||
// To check if Atomic TSO is currently enabled in the JIT, use `IsAtomicTSOEnabled` instead.
|
||||
bool SoftwareTSORequired() const {
|
||||
if (SupportsHardwareTSO) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return Config.TSOEnabled;
|
||||
}
|
||||
|
||||
void EnableExitOnHLT() override {
|
||||
ExitOnHLT = true;
|
||||
}
|
||||
@@ -378,9 +366,15 @@ protected:
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
VectorAtomicTSOEmulationEnabled = false;
|
||||
MemcpyAtomicTSOEmulationEnabled = false;
|
||||
} else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
|
||||
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
|
||||
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -403,6 +397,9 @@ private:
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool VectorAtomicTSOEmulationEnabled = false;
|
||||
bool MemcpyAtomicTSOEmulationEnabled = false;
|
||||
|
||||
bool ExitOnHLT = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
|
||||
@@ -29,11 +29,12 @@ namespace FEXCore::CPU {
|
||||
namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
// Note that rax/rdx are rearranged here so we can coalesce cmpxchg.
|
||||
constexpr std::array<ARMEmitter::Register, 18> SRA = {
|
||||
ARMEmitter::Reg::r4,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r5,
|
||||
ARMEmitter::Reg::r6,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8,
|
||||
ARMEmitter::Reg::r9,
|
||||
ARMEmitter::Reg::r10,
|
||||
@@ -194,12 +195,12 @@ namespace x64 {
|
||||
} // namespace x64
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
// All but x19 and x29 are caller saved. eax/edx rearranged for cmpxchg.
|
||||
constexpr std::array<ARMEmitter::Register, 10> SRA = {
|
||||
ARMEmitter::Reg::r4,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r5,
|
||||
ARMEmitter::Reg::r6,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8,
|
||||
ARMEmitter::Reg::r9,
|
||||
ARMEmitter::Reg::r10,
|
||||
@@ -373,6 +374,20 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::X86State::X86Reg Arm64Emitter::GetX86RegRelationToARMReg(ARMEmitter::Register Reg) {
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
const auto& RegI = StaticRegisters[i];
|
||||
if (RegI == Reg) {
|
||||
// X86 Registers are mapped linerally from the StaticRegisters span.
|
||||
// Directly correlating Enum index to span index.
|
||||
return static_cast<FEXCore::X86State::X86Reg>(FEXCore::ToUnderlying(FEXCore::X86State::X86Reg::REG_RAX) + i);
|
||||
}
|
||||
}
|
||||
|
||||
// Unmapped register.
|
||||
return FEXCore::X86State::X86Reg::REG_INVALID;
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
|
||||
@@ -4,11 +4,6 @@
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
#endif
|
||||
@@ -17,6 +12,7 @@
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
@@ -106,6 +102,10 @@ protected:
|
||||
|
||||
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
|
||||
|
||||
// Correlate an ARM register back to an x86 register index.
|
||||
// Returning REG_INVALID if there was no mapping.
|
||||
FEXCore::X86State::X86Reg GetX86RegRelationToARMReg(ARMEmitter::Register Reg);
|
||||
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
|
||||
@@ -72,8 +72,18 @@ namespace ProductNames {
|
||||
static const char ARM_Denver[] = "Nvidia Denver";
|
||||
static const char ARM_Carmel[] = "Nvidia Carmel";
|
||||
|
||||
static const char ARM_Firestorm[] = "Apple Firestorm";
|
||||
static const char ARM_Icestorm[] = "Apple Icestorm";
|
||||
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
|
||||
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
|
||||
static const char ARM_Firestorm_M1Pro[] = "Apple Firestorm (M1 Pro)";
|
||||
static const char ARM_Icestorm_M1Pro[] = "Apple Icestorm (M1 Pro)";
|
||||
static const char ARM_Firestorm_M1Max[] = "Apple Firestorm (M1 Max)";
|
||||
static const char ARM_Icestorm_M1Max[] = "Apple Icestorm (M1 Max)";
|
||||
static const char ARM_Avalanche_M2[] = "Apple Avalanche (M2)";
|
||||
static const char ARM_Blizzard_M2[] = "Apple Blizzard (M2)";
|
||||
static const char ARM_Avalanche_M2Pro[] = "Apple Avalanche (M2 Pro)";
|
||||
static const char ARM_Blizzard_M2Pro[] = "Apple Blizzard (M2 Pro)";
|
||||
static const char ARM_Avalanche_M2Max[] = "Apple Avalanche (M2 Max)";
|
||||
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
#else
|
||||
@@ -114,27 +124,16 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
uint64_t MIDR {};
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
std::error_code ec {};
|
||||
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
|
||||
std::array<char, 18> Data;
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFileToBuffer(MIDRPath, Data) == sizeof(Data)) {
|
||||
uint64_t NewMIDR {};
|
||||
std::string_view MIDRView(Data.data(), sizeof(Data));
|
||||
if (FEXCore::StrConv::Conv(MIDRView, &NewMIDR)) {
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
}
|
||||
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
auto NewMIDR = CTX->HostFeatures.CPUMIDRs[i];
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
}
|
||||
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
|
||||
struct CPUMIDR {
|
||||
@@ -147,11 +146,16 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 48> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
|
||||
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
{0x61, 0x039, 1, ProductNames::ARM_Avalanche_M2Max}, // Apple Avalanche (M2 Max)
|
||||
{0x61, 0x035, 1, ProductNames::ARM_Avalanche_M2Pro}, // Apple Avalanche (M2 Pro)
|
||||
{0x61, 0x033, 1, ProductNames::ARM_Avalanche_M2}, // Apple Avalanche (M2)
|
||||
{0x61, 0x029, 1, ProductNames::ARM_Firestorm_M1Max}, // Apple Firestorm (M1 Max)
|
||||
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
|
||||
|
||||
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
|
||||
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
|
||||
@@ -193,7 +197,13 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x41, 0xd07, 1, ProductNames::ARM_A57}, // A57
|
||||
|
||||
// Typically Little CPU cores
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm}, // Apple M1 Icestorm
|
||||
{0x61, 0x038, 0, ProductNames::ARM_Blizzard_M2Max}, // Apple Blizzard (M2 Max)
|
||||
{0x61, 0x034, 0, ProductNames::ARM_Blizzard_M2Pro}, // Apple Blizzard (M2 Pro)
|
||||
{0x61, 0x032, 0, ProductNames::ARM_Blizzard_M2}, // Apple Blizzard (M2)
|
||||
{0x61, 0x028, 0, ProductNames::ARM_Icestorm_M1Max}, // Apple Icestorm (M1 Max)
|
||||
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
|
||||
|
||||
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
@@ -608,8 +618,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
// Disable Enhanced REP MOVS when TSO is enabled.
|
||||
// vcruntime140 memmove will use `rep movsb` in this case which completely destroys perf in Hades(appId 1145360)
|
||||
// This is due to LRCPC performance on Cortex being abysmal.
|
||||
// Only enable EnhancedREPMOVS if SoftwareTSO isn't required OR if MemcpySetTSO is not enabled.
|
||||
const uint32_t SupportsEnhancedREPMOVS = CTX->SoftwareTSORequired() == false || MemcpySetTSOEnabled() == false;
|
||||
// Only enable EnhancedREPMOVS if atomic memcpy tso emulation isn't enabled.
|
||||
const uint32_t SupportsEnhancedREPMOVS = CTX->IsMemcpyAtomicTSOEnabled() == false;
|
||||
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
|
||||
|
||||
// Number of subfunctions
|
||||
@@ -772,7 +782,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) const {
|
||||
uint32_t FrequencyHz = GetCycleCounterFrequency();
|
||||
if (FrequencyHz) {
|
||||
Res.eax = 1;
|
||||
Res.ebx = CTX->Config.SmallTSCScale() ? FEXCore::Context::TSC_SCALE : 1;
|
||||
Res.ebx = 1U << CTX->Config.TSCScale;
|
||||
Res.ecx = FrequencyHz;
|
||||
}
|
||||
return Res;
|
||||
@@ -1185,7 +1195,7 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
|
||||
CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx} {
|
||||
Cores = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
Cores = CTX->HostFeatures.CPUMIDRs.size();
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
@@ -118,8 +118,6 @@ private:
|
||||
bool Hybrid {};
|
||||
uint32_t Cores {};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
// Mask that configures what features are enabled on the CPU.
|
||||
|
||||
@@ -29,6 +29,7 @@ $end_info$
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -94,8 +95,13 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
Symbols.InitFile();
|
||||
}
|
||||
|
||||
if (FEXCore::GetCycleCounterFrequency() >= FEXCore::Context::TSC_SCALE_MAXIMUM) {
|
||||
Config.SmallTSCScale = false;
|
||||
uint64_t FrequencyCounter = FEXCore::GetCycleCounterFrequency();
|
||||
if (FrequencyCounter && FrequencyCounter < FEXCore::Context::TSC_SCALE_MAXIMUM && Config.SmallTSCScale()) {
|
||||
// Scale TSC until it is at the minimum required.
|
||||
while (FrequencyCounter < FEXCore::Context::TSC_SCALE_MAXIMUM) {
|
||||
FrequencyCounter <<= 1;
|
||||
++Config.TSCScale;
|
||||
}
|
||||
}
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
@@ -190,6 +196,9 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC)) & 1;
|
||||
uint32_t SF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC)) & 1;
|
||||
|
||||
// CF is inverted in our representation, undo the invert here.
|
||||
CF ^= 1;
|
||||
|
||||
// Pack in to EFLAGS
|
||||
EFLAGS |= OF << X86State::RFLAG_OF_RAW_LOC;
|
||||
EFLAGS |= CF << X86State::RFLAG_CF_RAW_LOC;
|
||||
@@ -293,10 +302,10 @@ void ContextImpl::SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate packed NZCV
|
||||
// Calculate packed NZCV. Note CF is inverted.
|
||||
uint32_t Packed_NZCV {};
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_OF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_RAW_LOC)) ? 0 : 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC);
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_ZF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_SF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC) : 0;
|
||||
memcpy(&Frame->State.flags[X86State::RFLAG_NZCV_LOC], &Packed_NZCV, sizeof(Packed_NZCV));
|
||||
@@ -344,8 +353,8 @@ bool ContextImpl::InitCore() {
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#else
|
||||
// WIN32 always needs the interrupt fault check to be enabled.
|
||||
#elif !defined(_M_ARM64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
|
||||
@@ -468,8 +477,14 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
StrictSplitLockMutex = 0;
|
||||
}
|
||||
} else {
|
||||
CodeInvalidationMutex.unlock();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
FEXCore::Utils::SpinWaitLock::unlock(&StrictSplitLockMutex);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
@@ -477,6 +492,9 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
CodeInvalidationMutex.lock();
|
||||
Allocator::LockBeforeFork(Thread);
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
FEXCore::Utils::SpinWaitLock::lock(&StrictSplitLockMutex);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -595,6 +613,11 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
|
||||
const FEXCore::X86Tables::DecodedInst* DecodedInfo {nullptr};
|
||||
|
||||
@@ -274,12 +274,10 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -557,12 +555,10 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -967,8 +963,14 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
// If the target RIP is within the symbol ranges then we are golden
|
||||
if (TargetRIP >= SymbolMinAddress && TargetRIP < SymbolMaxAddress) {
|
||||
// If the target RIP is x86 code within the symbol ranges then we are golden
|
||||
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < SymbolMaxAddress;
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
|
||||
#endif
|
||||
|
||||
if (ValidMultiblockMember) {
|
||||
// Update our conditional branch ranges before we return
|
||||
if (Conditional) {
|
||||
MaxCondBranchForward = std::max(MaxCondBranchForward, TargetRIP);
|
||||
@@ -977,12 +979,12 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
uint64_t FallthroughRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
if (!HasBlocks.contains(FallthroughRIP)) {
|
||||
BlocksToDecode.insert(FallthroughRIP);
|
||||
CurrentBlockTargets.insert(FallthroughRIP);
|
||||
}
|
||||
}
|
||||
|
||||
if (!HasBlocks.contains(TargetRIP)) {
|
||||
BlocksToDecode.insert(TargetRIP);
|
||||
CurrentBlockTargets.insert(TargetRIP);
|
||||
}
|
||||
} else {
|
||||
if (ExternalBranches) {
|
||||
@@ -1079,6 +1081,8 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
|
||||
bool EntryBlock {true};
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1115,7 +1119,6 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) [[unlikely]] {
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", RIPToDecode + PCOffset, PC);
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
@@ -1123,6 +1126,14 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
DecodeInst->InstSize = 0;
|
||||
}
|
||||
|
||||
if (!ErrorDuringDecoding) {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
auto TableInfo = DecodedBuffer[BlockStartOffset + BlockNumberOfInstructions].TableInfo;
|
||||
if (!TableInfo || !TableInfo->OpcodeDispatcher) {
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
}
|
||||
}
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
|
||||
DecodedMaxAddress = std::max(DecodedMaxAddress, RIPToDecode + PCOffset + DecodeInst->InstSize);
|
||||
++TotalInstructions;
|
||||
@@ -1130,7 +1141,17 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
++DecodedSize;
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) {
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) [[unlikely]] {
|
||||
if (!EntryBlock) {
|
||||
// In multiblock configurations, we can early terminate any non-entrypoint blocks with the expectation that this won't get hit.
|
||||
// Improves compile-times.
|
||||
// Just need to undo additions that this block decoding has caused.
|
||||
TotalInstructions -= CurrentBlockDecoding.NumInstructions;
|
||||
DecodedSize = BlockStartOffset;
|
||||
BlockNumberOfInstructions = 0;
|
||||
InstStream -= PCOffset;
|
||||
CurrentBlockTargets.clear();
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1160,6 +1181,9 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
InstStream += DecodeInst->InstSize;
|
||||
}
|
||||
|
||||
BlocksToDecode.merge(CurrentBlockTargets);
|
||||
CurrentBlockTargets.clear();
|
||||
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
HasBlocks.emplace(RIPToDecode);
|
||||
|
||||
@@ -1167,6 +1191,8 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
CurrentBlockDecoding.NumInstructions = BlockNumberOfInstructions;
|
||||
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockInfo.TotalInstructionCount += BlockNumberOfInstructions;
|
||||
|
||||
EntryBlock = false;
|
||||
}
|
||||
|
||||
for (auto CodePage : CodePages) {
|
||||
|
||||
@@ -104,6 +104,7 @@ private:
|
||||
uint64_t SectionMaxAddress {~0ULL};
|
||||
|
||||
DecodedBlockInformation BlockInfo;
|
||||
fextl::set<uint64_t> CurrentBlockTargets;
|
||||
fextl::set<uint64_t> BlocksToDecode;
|
||||
fextl::set<uint64_t> HasBlocks;
|
||||
fextl::set<uint64_t>* ExternalBranches {nullptr};
|
||||
|
||||
@@ -41,21 +41,6 @@ DEF_BINOP_WITH_CONSTANT(Lshl, lslv, lsl)
|
||||
DEF_BINOP_WITH_CONSTANT(Lshr, lsrv, lsr)
|
||||
DEF_BINOP_WITH_CONSTANT(Ror, rorv, ror)
|
||||
|
||||
DEF_OP(TruncElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_TruncElementPair>();
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: {
|
||||
auto Dst = GetRegPair(Node);
|
||||
auto Src = GetRegPair(Op->Pair.ID());
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, Src.first);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.second, Src.second);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", IROp->Size); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Constant) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
auto Dst = GetReg(Node);
|
||||
@@ -130,6 +115,21 @@ DEF_OP(AdcWithFlags) {
|
||||
adcs(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AdcZeroWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AdcZeroWithFlags>();
|
||||
auto Size = ConvertSize48(IROp);
|
||||
|
||||
cset(Size, TMP1, ARMEmitter::Condition::CC_CC);
|
||||
adds(Size, GetReg(Node), GetReg(Op->Src1.ID()), TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(AdcZero) {
|
||||
auto Op = IROp->C<IR::IROp_AdcZero>();
|
||||
auto Size = ConvertSize48(IROp);
|
||||
|
||||
cinc(Size, GetReg(Node), GetReg(Op->Src1.ID()), ARMEmitter::Condition::CC_CC);
|
||||
}
|
||||
|
||||
DEF_OP(Adc) {
|
||||
auto Op = IROp->C<IR::IROp_Adc>();
|
||||
|
||||
@@ -232,10 +232,8 @@ DEF_OP(CmpPairZ) {
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// Compare, setting Z and clobbering NzCV
|
||||
const auto Src1 = GetRegPair(Op->Src1.ID());
|
||||
const auto Src2 = GetRegPair(Op->Src2.ID());
|
||||
cmp(EmitSize, Src1.first, Src2.first);
|
||||
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
cmp(EmitSize, GetReg(Op->Src1Lo.ID()), GetReg(Op->Src2Lo.ID()));
|
||||
ccmp(EmitSize, GetReg(Op->Src1Hi.ID()), GetReg(Op->Src2Hi.ID()), ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
// Restore NzCV
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
@@ -663,6 +661,11 @@ DEF_OP(ShiftFlags) {
|
||||
lsrv(EmitSize, CFWord, Src1, CFWord);
|
||||
}
|
||||
|
||||
if (Op->InvertCF) {
|
||||
mvn(ARMEmitter::Size::i64Bit, TMP1, CFWord);
|
||||
CFWord = TMP1;
|
||||
}
|
||||
|
||||
bool SetOF = Op->Shift != IR::ShiftType::ASR;
|
||||
if (SetOF) {
|
||||
// Only defined when Shift is 1 else undefined
|
||||
@@ -702,6 +705,50 @@ DEF_OP(ShiftFlags) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RotateFlags) {
|
||||
auto Op = IROp->C<IR::IROp_RotateFlags>();
|
||||
const auto Result = GetReg(Op->Result.ID());
|
||||
const auto Shift = GetReg(Op->Shift.ID());
|
||||
const bool Left = Op->Left;
|
||||
const auto EmitSize = Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto BitSize = Op->Size * 8;
|
||||
unsigned CFBit = Left ? 0 : BitSize - 1;
|
||||
|
||||
// For ROR, OF is the XOR of the new CF bit and the most significant bit of the result.
|
||||
// For ROL, OF is the LSB and MSB XOR'd together.
|
||||
// OF is architecturally only defined for 1-bit rotate.
|
||||
eor(ARMEmitter::Size::i64Bit, TMP1, Result, Result, ARMEmitter::ShiftType::LSR, Left ? BitSize - 1 : 1);
|
||||
unsigned OFBit = Left ? 0 : BitSize - 2;
|
||||
|
||||
// Invert result so we get inverted carry.
|
||||
mvn(ARMEmitter::Size::i64Bit, TMP2, Result);
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
rmif(TMP2, (CFBit - 1) % 64, 1 << 1 /* nzCv */);
|
||||
rmif(TMP1, OFBit, 1 << 0 /* nzcV */);
|
||||
} else {
|
||||
if (OFBit != 0) {
|
||||
lsr(EmitSize, TMP1, TMP1, OFBit);
|
||||
}
|
||||
if (CFBit != 0) {
|
||||
lsr(EmitSize, TMP2, TMP2, CFBit);
|
||||
}
|
||||
|
||||
mrs(TMP3, ARMEmitter::SystemRegister::NZCV);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP3, TMP1, 28 /* V */, 1);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP3, TMP2, 29 /* C */, 1);
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(Extr) {
|
||||
auto Op = IROp->C<IR::IROp_Extr>();
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1368,11 +1415,6 @@ DEF_OP(Select) {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetRegPair(Op->Cmp1.ID());
|
||||
const auto Src2 = GetRegPair(Op->Cmp2.ID());
|
||||
cmp(EmitSize, Src1.first, Src2.first);
|
||||
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, cc);
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1.ID());
|
||||
const auto Src2 = GetVReg(Op->Cmp2.ID());
|
||||
@@ -1433,6 +1475,12 @@ DEF_OP(NZCVSelect) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelectIncrement) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelectIncrement>();
|
||||
|
||||
csinc(ConvertSize(IROp), GetReg(Node), GetReg(Op->TrueVal.ID()), GetZeroableReg(Op->FalseVal), MapCC(Op->Cond));
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1443,6 +1491,7 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -1463,7 +1512,6 @@ DEF_OP(VExtractToGPR) {
|
||||
// when acting on larger register sizes.
|
||||
PerformMove(Vector, Op->Index);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE256, "Host doesn't support SVE. Cannot perform 256-bit operation.");
|
||||
LOGMAN_THROW_AA_FMT(Is256Bit, "Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(Offset < AVXRegBitSize, "Trying to extract element outside bounds of register. Offset={}, Index={}", Offset, Op->Index);
|
||||
|
||||
|
||||
@@ -15,19 +15,43 @@ DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Wrong element size");
|
||||
// Size is the size of each pair element
|
||||
auto Dst = GetRegPair(Node);
|
||||
auto Expected = GetRegPair(Op->Expected.ID());
|
||||
auto Desired = GetRegPair(Op->Desired.ID());
|
||||
auto Dst0 = GetReg(Op->OutLo.ID());
|
||||
auto Dst1 = GetReg(Op->OutHi.ID());
|
||||
auto Expected0 = GetReg(Op->ExpectedLo.ID());
|
||||
auto Expected1 = GetReg(Op->ExpectedHi.ID());
|
||||
auto Desired0 = GetReg(Op->DesiredLo.ID());
|
||||
auto Desired1 = GetReg(Op->DesiredHi.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = IROp->ElementSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(EmitSize, TMP3, Expected.first);
|
||||
mov(EmitSize, TMP4, Expected.second);
|
||||
// RA has heuristics to try to pair sources, but we need to handle the cases
|
||||
// where they fail. We do so by moving to temporaries. Note we use 64-bit
|
||||
// moves here even for 32-bit cmpxchg, for the Firestorm register renamer.
|
||||
if (Desired1.Idx() != (Desired0.Idx() + 1) || Desired0.Idx() & 1) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, Desired0);
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, Desired1);
|
||||
Desired0 = TMP1;
|
||||
Desired1 = TMP2;
|
||||
}
|
||||
|
||||
caspal(EmitSize, TMP3, TMP4, Desired.first, Desired.second, MemSrc);
|
||||
mov(EmitSize, Dst.first, TMP3.R());
|
||||
mov(EmitSize, Dst.second, TMP4.R());
|
||||
auto CaspalDst0 = Dst0;
|
||||
auto CaspalDst1 = Dst1;
|
||||
if (CaspalDst1.Idx() != (CaspalDst0.Idx() + 1) || CaspalDst0.Idx() & 1) {
|
||||
CaspalDst0 = TMP3;
|
||||
CaspalDst1 = TMP4;
|
||||
}
|
||||
|
||||
// We can't clobber the source, these moves are inherently required due to
|
||||
// ISA limitations. But by making them 64-bit, Firestorm can rename.
|
||||
mov(ARMEmitter::Size::i64Bit, CaspalDst0, Expected0);
|
||||
mov(ARMEmitter::Size::i64Bit, CaspalDst1, Expected1);
|
||||
caspal(EmitSize, CaspalDst0, CaspalDst1, Desired0, Desired1, MemSrc);
|
||||
|
||||
if (CaspalDst0 != Dst0) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst0, CaspalDst0);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst1, CaspalDst1);
|
||||
}
|
||||
} else {
|
||||
// Save NZCV so we don't have to mark this op as clobbering NZCV (the
|
||||
// SupportsAtomics does not clobber atomics and this !SupportsAtomics path
|
||||
@@ -43,19 +67,19 @@ DEF_OP(CASPair) {
|
||||
|
||||
// This instruction sequence must be synced with HandleCASPAL_Armv8.
|
||||
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
|
||||
cmp(EmitSize, TMP2, Expected.first);
|
||||
ccmp(EmitSize, TMP3, Expected.second, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
cmp(EmitSize, TMP2, Expected0);
|
||||
ccmp(EmitSize, TMP3, Expected1, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxp(EmitSize, TMP2, Desired.first, Desired.second, MemSrc);
|
||||
stlxp(EmitSize, TMP2, Desired0, Desired1, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
mov(EmitSize, Dst.first, Expected.first);
|
||||
mov(EmitSize, Dst.second, Expected.second);
|
||||
mov(EmitSize, Dst0, Expected0);
|
||||
mov(EmitSize, Dst1, Expected1);
|
||||
|
||||
b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst.first, TMP2.R());
|
||||
mov(EmitSize, Dst.second, TMP3.R());
|
||||
mov(EmitSize, Dst0, TMP2.R());
|
||||
mov(EmitSize, Dst1, TMP3.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
|
||||
@@ -78,8 +78,12 @@ DEF_OP(ExitFunction) {
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// arithmetic. ubfiz+add is marginally faster on Firestorm than
|
||||
// and+add(shift). Same performance on Cortex.
|
||||
static_assert(LookupCache::L1_ENTRIES_MASK == ((1u << 20) - 1));
|
||||
ubfiz(ARMEmitter::Size::i64Bit, TMP4, RipReg, 4, 20);
|
||||
add(TMP1, TMP1, TMP4);
|
||||
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
|
||||
@@ -112,20 +116,27 @@ DEF_OP(CondJump) {
|
||||
[[maybe_unused]] uint64_t Const;
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
auto Reg = GetReg(Op->Cmp1.ID());
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst && Const == 0, "CondJump: Expected 0 source");
|
||||
LOGMAN_THROW_A_FMT(Op->Cond.Val == FEXCore::IR::COND_EQ || Op->Cond.Val == FEXCore::IR::COND_NEQ, "CondJump: Expected simple "
|
||||
"condition");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
cbz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbnz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbz(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbnz(Reg, Const, TrueTargetLabel);
|
||||
} else {
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
LOGMAN_THROW_A_FMT(false, "CondJump expected simple condition");
|
||||
}
|
||||
|
||||
// TODO: Wire up tbz/tbnz
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
@@ -255,15 +266,13 @@ DEF_OP(InlineSyscall) {
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RBX, and RSI. Which have just been spilled
|
||||
// Just load back from the context. Could be slightly smarter but this is fairly uncommon
|
||||
if (Reg == ARMEmitter::Reg::r8) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
} else if (Reg == ARMEmitter::Reg::r4) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RAX]));
|
||||
} else if (Reg == ARMEmitter::Reg::r5) {
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RCX]));
|
||||
if (SpillMask & (1U << Reg.Idx())) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RDX, and RSP. Which have just been spilled
|
||||
// Just load back from the context.
|
||||
auto Correlation = GetX86RegRelationToARMReg(Reg);
|
||||
LOGMAN_THROW_A_FMT(Correlation != X86State::REG_INVALID, "Invalid register mapping");
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[Correlation]));
|
||||
} else {
|
||||
mov(EmitSize, RegArgs[i].R(), Reg);
|
||||
}
|
||||
@@ -427,10 +436,11 @@ DEF_OP(CPUID) {
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.first, TMP1);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, TMP2);
|
||||
// Results want to be 4xi32 scalars
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutECX.ID()), TMP2);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEBX.ID()), TMP1, 32, 32);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP2, 32, 32);
|
||||
}
|
||||
|
||||
DEF_OP(XGetBV) {
|
||||
@@ -459,11 +469,9 @@ DEF_OP(XGetBV) {
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0
|
||||
// Results want to be in a i32v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, TMP1);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, TMP1, 32);
|
||||
// Results are in x0, need to split into i32 parts
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP1, 32, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
@@ -17,6 +17,7 @@ DEF_OP(VInsGPR) {
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementsPer128Bit = 16 / ElementSize;
|
||||
@@ -112,6 +113,8 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -204,6 +207,7 @@ DEF_OP(Vector_SToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -236,6 +240,7 @@ DEF_OP(Vector_FToZS) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -266,6 +271,8 @@ DEF_OP(Vector_FToS) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -295,6 +302,8 @@ DEF_OP(Vector_FToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -396,6 +405,7 @@ DEF_OP(Vector_FToI) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -456,6 +466,7 @@ DEF_OP(Vector_F64ToI32) {
|
||||
const auto Round = Op->Round;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
@@ -654,12 +654,6 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
@@ -724,6 +718,16 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::SingleUseForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
#endif
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
|
||||
@@ -14,9 +14,6 @@ $end_info$
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
@@ -67,8 +64,6 @@ public:
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
@@ -114,15 +109,6 @@ private:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
std::pair<ARMEmitter::Register, ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return std::make_pair(GeneralRegisters[Reg.Reg], GeneralRegisters[Reg.Reg + 1]);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
@@ -253,8 +239,6 @@ private:
|
||||
bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
@@ -348,7 +332,8 @@ private:
|
||||
using ScalarFMAOpCaller =
|
||||
std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2, ARMEmitter::VRegister Src3)>;
|
||||
void VFScalarFMAOperation(uint8_t OpSize, uint8_t ElementSize, ScalarFMAOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2, ARMEmitter::VRegister Addend);
|
||||
ARMEmitter::VRegister Upper, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2,
|
||||
ARMEmitter::VRegister Addend);
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -47,6 +48,31 @@ DEF_OP(LoadContext) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetReg(Op->OutValue2.ID());
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.W(), Dst2.W(), STATE, Op->Offset); break;
|
||||
case 8: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.X(), Dst2.X(), STATE, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst1 = GetVReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetVReg(Op->OutValue2.ID());
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.S(), Dst2.S(), STATE, Op->Offset); break;
|
||||
case 8: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.D(), Dst2.D(), STATE, Op->Offset); break;
|
||||
case 16: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.Q(), Dst2.Q(), STATE, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -79,6 +105,32 @@ DEF_OP(StoreContext) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto Src1 = GetZeroableReg(Op->Value1);
|
||||
auto Src2 = GetZeroableReg(Op->Value2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: stp<ARMEmitter::IndexType::OFFSET>(Src1.W(), Src2.W(), STATE, Op->Offset); break;
|
||||
case 8: stp<ARMEmitter::IndexType::OFFSET>(Src1.X(), Src2.X(), STATE, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src1 = GetVReg(Op->Value1.ID());
|
||||
const auto Src2 = GetVReg(Op->Value2.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: stp<ARMEmitter::IndexType::OFFSET>(Src1.S(), Src2.S(), STATE, Op->Offset); break;
|
||||
case 8: stp<ARMEmitter::IndexType::OFFSET>(Src1.D(), Src2.D(), STATE, Op->Offset); break;
|
||||
case 16: stp<ARMEmitter::IndexType::OFFSET>(Src1.Q(), Src2.Q(), STATE, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContextPair size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -367,30 +419,6 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRPairClass) {
|
||||
const auto Src = GetRegPair(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 8: {
|
||||
if (SlotOffset <= 252 && (SlotOffset & 0b11) == 0) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Src.first.W(), Src.second.W(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Src.first.W(), Src.second.W(), TMP1, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
if (SlotOffset <= 504 && (SlotOffset & 0b111) == 0) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Src.first.X(), Src.second.X(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Src.first.X(), Src.second.X(), TMP1, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister(GPRPair) size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
}
|
||||
@@ -480,30 +508,6 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRPairClass) {
|
||||
const auto Src = GetRegPair(Node);
|
||||
switch (OpSize) {
|
||||
case 8: {
|
||||
if (SlotOffset <= 252 && (SlotOffset & 0b11) == 0) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Src.first.W(), Src.second.W(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Src.first.W(), Src.second.W(), TMP1, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
if (SlotOffset <= 504 && (SlotOffset & 0b111) == 0) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Src.first.X(), Src.second.X(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Src.first.X(), Src.second.X(), TMP1, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister(GPRPair) size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
}
|
||||
@@ -635,6 +639,7 @@ DEF_OP(LoadMem) {
|
||||
case 8: ldr(Dst.D(), MemSrc); break;
|
||||
case 16: ldr(Dst.Q(), MemSrc); break;
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Operand = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), Operand);
|
||||
break;
|
||||
@@ -644,6 +649,32 @@ DEF_OP(LoadMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetReg(Op->OutValue2.ID());
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.W(), Dst2.W(), Addr, Op->Offset); break;
|
||||
case 8: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.X(), Dst2.X(), Addr, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst1 = GetVReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetVReg(Op->OutValue2.ID());
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.S(), Dst2.S(), Addr, Op->Offset); break;
|
||||
case 8: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.D(), Dst2.D(), Addr, Op->Offset); break;
|
||||
case 16: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.Q(), Dst2.Q(), Addr, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -717,13 +748,14 @@ DEF_OP(LoadMemTSO) {
|
||||
case 8: ldr(Dst.D(), MemSrc); break;
|
||||
case 16: ldr(Dst.Q(), MemSrc); break;
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
if (VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
// Half-barrier.
|
||||
dmb(ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
@@ -736,9 +768,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VLoadVectorMasked with 256-bit operation");
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -833,9 +863,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreVectorMasked with 256-bit operation");
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -1052,9 +1080,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
/// - AddrBase also doesn't need to exist
|
||||
/// - If the instruction is using 64-bit vector indexing or 32-bit addresses where the top-bit isn't set then this is valid!
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreVectorMasked with 256-bit operation");
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
@@ -1238,7 +1264,7 @@ DEF_OP(VLoadVectorElement) {
|
||||
}
|
||||
|
||||
// Emit a half-barrier if TSO is enabled.
|
||||
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
dmb(ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
}
|
||||
@@ -1257,7 +1283,7 @@ DEF_OP(VStoreVectorElement) {
|
||||
"size");
|
||||
|
||||
// Emit a half-barrier if TSO is enabled.
|
||||
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
|
||||
@@ -1280,6 +1306,7 @@ DEF_OP(VBroadcastFromMem) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1288,11 +1315,6 @@ DEF_OP(VBroadcastFromMem) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid element "
|
||||
"size");
|
||||
|
||||
if (Is256Bit && !HostSupportsSVE256) {
|
||||
LOGMAN_MSG_A_FMT("{}: 256-bit vectors must support SVE256", __func__);
|
||||
return;
|
||||
}
|
||||
|
||||
if (Is256Bit && HostSupportsSVE256) {
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
|
||||
@@ -1319,7 +1341,7 @@ DEF_OP(VBroadcastFromMem) {
|
||||
}
|
||||
|
||||
// Emit a half-barrier if TSO is enabled.
|
||||
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
dmb(ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
}
|
||||
@@ -1409,6 +1431,37 @@ DEF_OP(Push) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Pop) {
|
||||
const auto Op = IROp->C<IR::IROp_Pop>();
|
||||
const auto Addr = GetReg(Op->InoutAddr.ID());
|
||||
const auto Dst = GetReg(Op->OutValue.ID());
|
||||
|
||||
LOGMAN_THROW_A_FMT(Dst != Addr, "Invalid");
|
||||
|
||||
switch (Op->Size) {
|
||||
case 1: {
|
||||
ldrb<ARMEmitter::IndexType::POST>(Dst.W(), Addr, Op->Size);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ldrh<ARMEmitter::IndexType::POST>(Dst.W(), Addr, Op->Size);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ldr<ARMEmitter::IndexType::POST>(Dst.W(), Addr, Op->Size);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ldr<ARMEmitter::IndexType::POST>(Dst.X(), Addr, Op->Size);
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Op->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1450,6 +1503,7 @@ DEF_OP(StoreMem) {
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemSrc);
|
||||
break;
|
||||
@@ -1459,6 +1513,32 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src1 = GetReg(Op->Value1.ID());
|
||||
const auto Src2 = GetReg(Op->Value2.ID());
|
||||
switch (OpSize) {
|
||||
case 4: stp<ARMEmitter::IndexType::OFFSET>(Src1.W(), Src2.W(), Addr, Op->Offset); break;
|
||||
case 8: stp<ARMEmitter::IndexType::OFFSET>(Src1.X(), Src2.X(), Addr, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src1 = GetVReg(Op->Value1.ID());
|
||||
const auto Src2 = GetVReg(Op->Value2.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: stp<ARMEmitter::IndexType::OFFSET>(Src1.S(), Src2.S(), Addr, Op->Offset); break;
|
||||
case 8: stp<ARMEmitter::IndexType::OFFSET>(Src1.D(), Src2.D(), Addr, Op->Offset); break;
|
||||
case 16: stp<ARMEmitter::IndexType::OFFSET>(Src1.Q(), Src2.Q(), Addr, Op->Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemPair size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -1508,7 +1588,7 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (VectorTSOEnabled()) {
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
// Half-Barrier.
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
@@ -1521,6 +1601,7 @@ DEF_OP(StoreMemTSO) {
|
||||
case 8: str(Src.D(), MemSrc); break;
|
||||
case 16: str(Src.Q(), MemSrc); break;
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Operand = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, Operand);
|
||||
break;
|
||||
@@ -1540,7 +1621,7 @@ DEF_OP(MemSet) {
|
||||
// that the value is zero, we can optimize any operation larger than 8-bit down to 8-bit to use the MOPS implementation.
|
||||
const auto Op = IROp->C<IR::IROp_MemSet>();
|
||||
|
||||
const bool IsAtomic = Op->IsAtomic && MemcpySetTSOEnabled();
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const int32_t Size = Op->Size;
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
@@ -1730,7 +1811,7 @@ DEF_OP(MemCpy) {
|
||||
// Assuming non-atomicity and non-faulting behaviour, this can accelerate this implementation.
|
||||
const auto Op = IROp->C<IR::IROp_MemCpy>();
|
||||
|
||||
const bool IsAtomic = Op->IsAtomic && MemcpySetTSOEnabled();
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const int32_t Size = Op->Size;
|
||||
const auto MemRegDest = GetReg(Op->Dest.ID());
|
||||
const auto MemRegSrc = GetReg(Op->Src.ID());
|
||||
@@ -1743,7 +1824,8 @@ DEF_OP(MemCpy) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
|
||||
auto Dst = GetRegPair(Node);
|
||||
auto Dst0 = GetReg(Op->OutDstAddress.ID());
|
||||
auto Dst1 = GetReg(Op->OutSrcAddress.ID());
|
||||
// If Direction > 0 then:
|
||||
// MemRegDest is incremented (by size)
|
||||
// MemRegSrc is incremented (by size)
|
||||
@@ -1940,40 +2022,40 @@ DEF_OP(MemCpy) {
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
add(Dst.first.X(), TMP1, TMP3);
|
||||
add(Dst.second.X(), TMP2, TMP3);
|
||||
add(Dst0.X(), TMP1, TMP3);
|
||||
add(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
add(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
add(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
add(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
add(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
add(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
add(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
sub(Dst.first.X(), TMP1, TMP3);
|
||||
sub(Dst.second.X(), TMP2, TMP3);
|
||||
sub(Dst0.X(), TMP1, TMP3);
|
||||
sub(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
sub(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
sub(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
sub(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
sub(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
sub(Dst.first.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
sub(Dst.second.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
@@ -2070,6 +2152,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case 32:
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
@@ -2146,6 +2229,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
@@ -2157,6 +2241,11 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
|
||||
dmb(ARMEmitter::BarrierScope::SY);
|
||||
return;
|
||||
}
|
||||
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
@@ -2181,6 +2270,11 @@ DEF_OP(CacheLineClear) {
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClean) {
|
||||
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
|
||||
dmb(ARMEmitter::BarrierScope::ST);
|
||||
return;
|
||||
}
|
||||
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClean>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
@@ -2262,6 +2356,7 @@ DEF_OP(VStoreNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
@@ -2269,7 +2364,6 @@ DEF_OP(VStoreNonTemporal) {
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreNonTemporal with 256-bit operation");
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
const auto OffsetScaled = Offset / 32;
|
||||
stnt1b(Value.Z(), GoverningPredicate, MemReg, OffsetScaled);
|
||||
@@ -2304,6 +2398,7 @@ DEF_OP(VLoadNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -2311,7 +2406,6 @@ DEF_OP(VLoadNonTemporal) {
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreNonTemporal with 256-bit operation");
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
const auto OffsetScaled = Offset / 32;
|
||||
ldnt1b(Dst.Z(), GoverningPredicate, MemReg, OffsetScaled);
|
||||
|
||||
@@ -18,6 +18,10 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AllocateGPR) {}
|
||||
DEF_OP(AllocateGPRAfter) {}
|
||||
DEF_OP(AllocateFPR) {}
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
@@ -254,18 +258,7 @@ DEF_OP(ProcessorID) {
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be in a i64v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
|
||||
if (Op->GetReseeded) {
|
||||
mrs(Dst.first, ARMEmitter::SystemRegister::RNDRRS);
|
||||
} else {
|
||||
mrs(Dst.first, ARMEmitter::SystemRegister::RNDR);
|
||||
}
|
||||
|
||||
// If the rng number is valid then NZCV is 0b0000, otherwise NZCV is 0b0100
|
||||
cset(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Condition::CC_NE);
|
||||
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
|
||||
@@ -9,47 +9,22 @@ $end_info$
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Pair = GetRegPair(Op->Pair.ID());
|
||||
const auto Src = Op->Element == 0 ? Pair.first : Pair.second;
|
||||
|
||||
if (Dst != Src) {
|
||||
mov(ConvertSize48(IROp), Dst, Src);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CreateElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_CreateElementPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Invalid size");
|
||||
std::pair<ARMEmitter::Register, ARMEmitter::Register> Dst = GetRegPair(Node);
|
||||
ARMEmitter::Register RegFirst = GetReg(Op->Lower.ID());
|
||||
ARMEmitter::Register RegSecond = GetReg(Op->Upper.ID());
|
||||
ARMEmitter::Register RegTmp = TMP1.R();
|
||||
|
||||
const auto EmitSize = IROp->ElementSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
if (Dst.first.Idx() != RegSecond.Idx()) {
|
||||
mov(EmitSize, Dst.first, RegFirst);
|
||||
mov(EmitSize, Dst.second, RegSecond);
|
||||
} else if (Dst.second.Idx() != RegFirst.Idx()) {
|
||||
mov(EmitSize, Dst.second, RegSecond);
|
||||
mov(EmitSize, Dst.first, RegFirst);
|
||||
} else {
|
||||
mov(EmitSize, RegTmp, RegFirst);
|
||||
mov(EmitSize, Dst.second, RegSecond);
|
||||
mov(EmitSize, Dst.first, RegTmp);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Copy) {
|
||||
auto Op = IROp->C<IR::IROp_Copy>();
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Source.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(RMWHandle) {
|
||||
auto Op = IROp->C<IR::IROp_RMWHandle>();
|
||||
auto Dest = GetReg(Node);
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
if (Dest != Src) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dest, Src);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Swap1) {
|
||||
auto Op = IROp->C<IR::IROp_Swap1>();
|
||||
auto A = GetReg(Op->A.ID()), B = GetReg(Op->B.ID());
|
||||
|
||||
@@ -13,162 +13,168 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
#define DEF_UNOP(FEXOp, ARMOp, ScalarCase) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto ElementSize = Op->Header.ElementSize; \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Src = GetVReg(Op->Vector.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), PRED_TMP_32B.Merging(), Src.Z()); \
|
||||
} else { \
|
||||
if (ElementSize == OpSize && ScalarCase) { \
|
||||
ARMOp(SubRegSize, Dst.D(), Src.D()); \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), Src.Q()); \
|
||||
} \
|
||||
} \
|
||||
#define DEF_UNOP(FEXOp, ARMOp, ScalarCase) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto ElementSize = Op->Header.ElementSize; \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Src = GetVReg(Op->Vector.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), PRED_TMP_32B.Merging(), Src.Z()); \
|
||||
} else { \
|
||||
if (ElementSize == OpSize && ScalarCase) { \
|
||||
ARMOp(SubRegSize, Dst.D(), Src.D()); \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), Src.Q()); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
#define DEF_BITOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else { \
|
||||
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} \
|
||||
#define DEF_BITOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else { \
|
||||
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define DEF_BINOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} \
|
||||
#define DEF_BINOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define DEF_ZIPOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID()); \
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), VectorLower.Z(), VectorUpper.Z()); \
|
||||
} else { \
|
||||
if (OpSize == 8) { \
|
||||
ARMOp(SubRegSize, Dst.D(), VectorLower.D(), VectorUpper.D()); \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), VectorLower.Q(), VectorUpper.Q()); \
|
||||
} \
|
||||
} \
|
||||
#define DEF_ZIPOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID()); \
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), VectorLower.Z(), VectorUpper.Z()); \
|
||||
} else { \
|
||||
if (OpSize == 8) { \
|
||||
ARMOp(SubRegSize, Dst.D(), VectorLower.D(), VectorUpper.D()); \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), VectorLower.Q(), VectorUpper.Q()); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
#define DEF_FUNOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto ElementSize = Op->Header.ElementSize; \
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp); \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Src = GetVReg(Op->Vector.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), PRED_TMP_32B.Merging(), Src.Z()); \
|
||||
} else { \
|
||||
if (ElementSize == OpSize) { \
|
||||
switch (ElementSize) { \
|
||||
case 2: { \
|
||||
ARMOp(Dst.H(), Src.H()); \
|
||||
break; \
|
||||
} \
|
||||
case 4: { \
|
||||
ARMOp(Dst.S(), Src.S()); \
|
||||
break; \
|
||||
} \
|
||||
case 8: { \
|
||||
ARMOp(Dst.D(), Src.D()); \
|
||||
break; \
|
||||
} \
|
||||
default: break; \
|
||||
} \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), Src.Q()); \
|
||||
} \
|
||||
} \
|
||||
#define DEF_FUNOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto ElementSize = Op->Header.ElementSize; \
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp); \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Src = GetVReg(Op->Vector.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), PRED_TMP_32B.Merging(), Src.Z()); \
|
||||
} else { \
|
||||
if (ElementSize == OpSize) { \
|
||||
switch (ElementSize) { \
|
||||
case 2: { \
|
||||
ARMOp(Dst.H(), Src.H()); \
|
||||
break; \
|
||||
} \
|
||||
case 4: { \
|
||||
ARMOp(Dst.S(), Src.S()); \
|
||||
break; \
|
||||
} \
|
||||
case 8: { \
|
||||
ARMOp(Dst.D(), Src.D()); \
|
||||
break; \
|
||||
} \
|
||||
default: break; \
|
||||
} \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), Src.Q()); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
#define DEF_FBINOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto ElementSize = Op->Header.ElementSize; \
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp); \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
const auto IsScalar = ElementSize == OpSize; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else { \
|
||||
if (IsScalar) { \
|
||||
switch (ElementSize) { \
|
||||
case 2: { \
|
||||
ARMOp(Dst.H(), Vector1.H(), Vector2.H()); \
|
||||
break; \
|
||||
} \
|
||||
case 4: { \
|
||||
ARMOp(Dst.S(), Vector1.S(), Vector2.S()); \
|
||||
break; \
|
||||
} \
|
||||
case 8: { \
|
||||
ARMOp(Dst.D(), Vector1.D(), Vector2.D()); \
|
||||
break; \
|
||||
} \
|
||||
default: break; \
|
||||
} \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} \
|
||||
} \
|
||||
#define DEF_FBINOP(FEXOp, ARMOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
\
|
||||
const auto ElementSize = Op->Header.ElementSize; \
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp); \
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
const auto IsScalar = ElementSize == OpSize; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else { \
|
||||
if (IsScalar) { \
|
||||
switch (ElementSize) { \
|
||||
case 2: { \
|
||||
ARMOp(Dst.H(), Vector1.H(), Vector2.H()); \
|
||||
break; \
|
||||
} \
|
||||
case 4: { \
|
||||
ARMOp(Dst.S(), Vector1.S(), Vector2.S()); \
|
||||
break; \
|
||||
} \
|
||||
case 8: { \
|
||||
ARMOp(Dst.D(), Vector1.D(), Vector2.D()); \
|
||||
break; \
|
||||
} \
|
||||
default: break; \
|
||||
} \
|
||||
} else { \
|
||||
ARMOp(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
#define DEF_FBINOP_SCALAR_INSERT(FEXOp, ARMOp) \
|
||||
@@ -205,11 +211,12 @@ namespace FEXCore::CPU {
|
||||
}; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Upper = GetVReg(Op->Upper.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Addend = GetVReg(Op->Addend.ID()); \
|
||||
\
|
||||
VFScalarFMAOperation(IROp->Size, ElementSize, ScalarEmit, Dst, Vector1, Vector2, Addend); \
|
||||
VFScalarFMAOperation(IROp->Size, ElementSize, ScalarEmit, Dst, Upper, Vector1, Vector2, Addend); \
|
||||
}
|
||||
|
||||
DEF_UNOP(VAbs, abs, true)
|
||||
@@ -254,17 +261,17 @@ DEF_FMAOP_SCALAR_INSERT(VFNMLAScalarInsert, fmsub)
|
||||
DEF_FMAOP_SCALAR_INSERT(VFNMLSScalarInsert, fnmadd)
|
||||
|
||||
void Arm64JITCore::VFScalarFMAOperation(uint8_t OpSize, uint8_t ElementSize, ScalarFMAOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2, ARMEmitter::VRegister Addend) {
|
||||
ARMEmitter::VRegister Upper, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2,
|
||||
ARMEmitter::VRegister Addend) {
|
||||
LOGMAN_THROW_A_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "256-bit unsupported", __func__);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
if (Dst != Vector1 && Dst != Vector2 && Dst != Addend && HostSupportsAFP) {
|
||||
// If destination doesnt overlap any incoming register then move the adder to the destination first.
|
||||
mov(Dst.Q(), Addend.Q());
|
||||
Dst = Addend;
|
||||
if (Dst != Upper) {
|
||||
// If destination is not tied, move the upper bits to the destination first.
|
||||
mov(Dst.Q(), Upper.Q());
|
||||
}
|
||||
|
||||
if (HostSupportsAFP && Dst == Addend) {
|
||||
@@ -272,7 +279,7 @@ void Arm64JITCore::VFScalarFMAOperation(uint8_t OpSize, uint8_t ElementSize, Sca
|
||||
// If the host CPU supports AFP then scalar does an insert without modifying upper bits.
|
||||
ScalarEmit(Dst, Vector1, Vector2, Addend);
|
||||
} else {
|
||||
// No overlap between addr and destination or host doesn't support AFP, need to emit in to a temporary then insert.
|
||||
// Host doesn't support AFP, need to emit in to a temporary then insert.
|
||||
ScalarEmit(VTMP1, Vector1, Vector2, Addend);
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
@@ -285,6 +292,7 @@ void Arm64JITCore::VFScalarFMAOperation(uint8_t OpSize, uint8_t ElementSize, Sca
|
||||
void Arm64JITCore::VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2) {
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(Is256Bit || !ZeroUpperBits, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
|
||||
|
||||
// Bit of a tricky detail.
|
||||
@@ -357,6 +365,7 @@ void Arm64JITCore::VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, b
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1,
|
||||
std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2) {
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(Is256Bit || !ZeroUpperBits, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
@@ -647,6 +656,7 @@ DEF_OP(VSToFVectorInsert) {
|
||||
|
||||
// Dealing with the odd case of this being actually a vector operation rather than scalar.
|
||||
const auto Is256Bit = IROp->Size == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
constexpr auto Predicate = ARMEmitter::PReg::p0;
|
||||
|
||||
ScalarEmit(VTMP1, Vector2);
|
||||
@@ -746,6 +756,7 @@ DEF_OP(VFCMPScalarInsert) {
|
||||
|
||||
const auto ZeroUpperBits = Op->ZeroUpperBits;
|
||||
const auto Is256Bit = IROp->Size == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
auto ScalarEmitEQ = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
switch (SubRegSize.Scalar) {
|
||||
@@ -903,6 +914,7 @@ DEF_OP(VectorImm) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
@@ -1057,6 +1069,7 @@ DEF_OP(VAddP) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto IsScalar = OpSize == 8;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1092,6 +1105,7 @@ DEF_OP(VFAddV) {
|
||||
const auto Op = IROp->C<IR::IROp_VAddV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
@@ -1126,6 +1140,7 @@ DEF_OP(VAddV) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSizePair8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -1159,6 +1174,8 @@ DEF_OP(VUMinV) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1178,6 +1195,8 @@ DEF_OP(VUMaxV) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1198,6 +1217,7 @@ DEF_OP(VURAvg) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -1229,6 +1249,8 @@ DEF_OP(VFAddP) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1264,6 +1286,7 @@ DEF_OP(VFDiv) {
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -1319,6 +1342,7 @@ DEF_OP(VFMin) {
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -1409,6 +1433,7 @@ DEF_OP(VFMax) {
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -1485,6 +1510,7 @@ DEF_OP(VFRecp) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = Op->Header.ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -1550,6 +1576,7 @@ DEF_OP(VFRSqrt) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -1614,6 +1641,7 @@ DEF_OP(VNot) {
|
||||
const auto Op = IROp->C<IR::IROp_VNot>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -1632,6 +1660,7 @@ DEF_OP(VUMin) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -1680,6 +1709,7 @@ DEF_OP(VSMin) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -1728,6 +1758,7 @@ DEF_OP(VUMax) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -1776,6 +1807,7 @@ DEF_OP(VSMax) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -1821,6 +1853,8 @@ DEF_OP(VBSL) {
|
||||
const auto Op = IROp->C<IR::IROp_VBSL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1890,6 +1924,7 @@ DEF_OP(VCMPEQ) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -1929,6 +1964,7 @@ DEF_OP(VCMPEQZ) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -1970,6 +2006,7 @@ DEF_OP(VCMPGT) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -2009,6 +2046,7 @@ DEF_OP(VCMPGTZ) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -2047,6 +2085,7 @@ DEF_OP(VCMPLTZ) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -2085,6 +2124,7 @@ DEF_OP(VFCMPEQ) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -2123,6 +2163,7 @@ DEF_OP(VFCMPNEQ) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -2163,6 +2204,7 @@ DEF_OP(VFCMPLT) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -2201,6 +2243,7 @@ DEF_OP(VFCMPGT) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -2239,6 +2282,7 @@ DEF_OP(VFCMPLE) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -2277,6 +2321,7 @@ DEF_OP(VFCMPORD) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -2327,6 +2372,7 @@ DEF_OP(VFCMPUNO) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -2375,6 +2421,8 @@ DEF_OP(VUShl) {
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto MaxShift = ElementSize * 8;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -2429,6 +2477,8 @@ DEF_OP(VUShr) {
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto MaxShift = ElementSize * 8;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -2486,6 +2536,8 @@ DEF_OP(VSShr) {
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto MaxShift = (ElementSize * 8) - 1;
|
||||
const auto RangeCheck = Op->RangeCheck;
|
||||
|
||||
@@ -2542,6 +2594,7 @@ DEF_OP(VUShlS) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
@@ -2570,6 +2623,7 @@ DEF_OP(VUShrS) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
@@ -2600,6 +2654,7 @@ DEF_OP(VUShrSWide) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
@@ -2665,6 +2720,7 @@ DEF_OP(VSShrSWide) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
@@ -2730,6 +2786,7 @@ DEF_OP(VUShlSWide) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
@@ -2793,6 +2850,7 @@ DEF_OP(VSShrS) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
@@ -2820,6 +2878,7 @@ DEF_OP(VInsElement) {
|
||||
const auto Op = IROp->C<IR::IROp_VInsElement>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const uint32_t ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
@@ -2902,6 +2961,8 @@ DEF_OP(VDupElement) {
|
||||
const auto Index = Op->Index;
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -2922,6 +2983,7 @@ DEF_OP(VExtr) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtr>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
// AArch64 ext op has bit arrangement as [Vm:Vn] so arguments need to be swapped
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -2973,6 +3035,7 @@ DEF_OP(VUShrI) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3014,6 +3077,7 @@ DEF_OP(VUShraI) {
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector.ID());
|
||||
@@ -3056,6 +3120,7 @@ DEF_OP(VSShrI) {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Shift = std::min(uint8_t(ElementSize * 8 - 1), Op->BitShift);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3094,6 +3159,7 @@ DEF_OP(VShlI) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3135,6 +3201,7 @@ DEF_OP(VUShrNI) {
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3154,6 +3221,7 @@ DEF_OP(VUShrNI2) {
|
||||
const auto BitShift = Op->BitShift;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
@@ -3190,6 +3258,7 @@ DEF_OP(VSXTL) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3207,6 +3276,7 @@ DEF_OP(VSXTL2) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3224,6 +3294,7 @@ DEF_OP(VSSHLL) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3245,6 +3316,7 @@ DEF_OP(VSSHLL2) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3266,6 +3338,7 @@ DEF_OP(VUXTL) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3283,6 +3356,7 @@ DEF_OP(VUXTL2) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3300,6 +3374,7 @@ DEF_OP(VSQXTN) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3351,6 +3426,7 @@ DEF_OP(VSQXTN2) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
@@ -3394,6 +3470,7 @@ DEF_OP(VSQXTNPair) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
@@ -3437,6 +3514,7 @@ DEF_OP(VSQXTUN) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -3455,6 +3533,7 @@ DEF_OP(VSQXTUN2) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
@@ -3500,6 +3579,7 @@ DEF_OP(VSQXTUNPair) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
@@ -3542,6 +3622,8 @@ DEF_OP(VSRSHR) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -3570,6 +3652,8 @@ DEF_OP(VSQSHL) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -3598,6 +3682,8 @@ DEF_OP(VMul) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -3617,6 +3703,7 @@ DEF_OP(VUMull) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -3637,6 +3724,7 @@ DEF_OP(VSMull) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -3657,6 +3745,7 @@ DEF_OP(VUMull2) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -3677,6 +3766,7 @@ DEF_OP(VSMull2) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -3698,6 +3788,8 @@ DEF_OP(VUMulH) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -3747,6 +3839,8 @@ DEF_OP(VSMulH) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -3795,6 +3889,7 @@ DEF_OP(VUABDL) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -3820,6 +3915,7 @@ DEF_OP(VUABDL2) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -3969,6 +4065,7 @@ DEF_OP(VRev32) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -4007,6 +4104,7 @@ DEF_OP(VRev64) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize4(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -4044,6 +4142,7 @@ DEF_OP(VFCADD) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -4090,6 +4189,7 @@ DEF_OP(VFMLA) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -4156,6 +4256,7 @@ DEF_OP(VFMLS) {
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -4245,6 +4346,7 @@ DEF_OP(VFNMLA) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
@@ -4312,6 +4414,8 @@ DEF_OP(VFNMLS) {
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -4,6 +4,7 @@
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -117,6 +118,7 @@ public:
|
||||
// Changes get stored out by CalculateDeferredFlags.
|
||||
CachedNZCV = nullptr;
|
||||
PossiblySetNZCVBits = ~0U;
|
||||
CFInverted = CFInvertedABI;
|
||||
FlushRegisterCache();
|
||||
|
||||
// New block needs to reset segment telemetry.
|
||||
@@ -155,6 +157,11 @@ public:
|
||||
auto Placeholder = _InlineConstant(0);
|
||||
return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJumpBit(Ref Src, unsigned Bit, bool Set) {
|
||||
FlushRegisterCache();
|
||||
auto InlineConst = _InlineConstant(Bit);
|
||||
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, {Set ? COND_TSTNZ : COND_TSTZ}, 0, false);
|
||||
}
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP) {
|
||||
FlushRegisterCache();
|
||||
return _ExitFunction(NewRIP);
|
||||
@@ -454,32 +461,22 @@ public:
|
||||
|
||||
void MOVQOp(OpcodeArgs, VectorOpType VectorType);
|
||||
void MOVQMMXOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void MOVMSKOp(OpcodeArgs);
|
||||
void MOVMSKOp(OpcodeArgs, size_t ElementSize);
|
||||
void MOVMSKOpOne(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PUNPCKLOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PUNPCKHOp(OpcodeArgs);
|
||||
void PUNPCKLOp(OpcodeArgs, size_t ElementSize);
|
||||
void PUNPCKHOp(OpcodeArgs, size_t ElementSize);
|
||||
void PSHUFBOp(OpcodeArgs);
|
||||
template<bool Low>
|
||||
void PSHUFWOp(OpcodeArgs);
|
||||
void PSHUFWOp(OpcodeArgs, bool Low);
|
||||
void PSHUFW8ByteOp(OpcodeArgs);
|
||||
void PSHUFDOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSRLDOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSRLI(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSLLI(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSLL(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSRAOp(OpcodeArgs);
|
||||
void PSRLDOp(OpcodeArgs, size_t ElementSize);
|
||||
void PSRLI(OpcodeArgs, size_t ElementSize);
|
||||
void PSLLI(OpcodeArgs, size_t ElementSize);
|
||||
void PSLL(OpcodeArgs, size_t ElementSize);
|
||||
void PSRAOp(OpcodeArgs, size_t ElementSize);
|
||||
void PSRLDQ(OpcodeArgs);
|
||||
void PSLLDQ(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PSRAIOp(OpcodeArgs);
|
||||
void PSRAIOp(OpcodeArgs, size_t ElementSize);
|
||||
void MOVDDUPOp(OpcodeArgs);
|
||||
template<size_t DstElementSize>
|
||||
void CVTGPR_To_FPR(OpcodeArgs);
|
||||
@@ -501,13 +498,11 @@ public:
|
||||
void LZCNT(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void VFCMPOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void SHUFOp(OpcodeArgs);
|
||||
void SHUFOp(OpcodeArgs, size_t ElementSize);
|
||||
template<size_t ElementSize>
|
||||
void PINSROp(OpcodeArgs);
|
||||
void InsertPSOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PExtrOp(OpcodeArgs);
|
||||
void PExtrOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void PSIGN(OpcodeArgs);
|
||||
@@ -601,8 +596,7 @@ public:
|
||||
void VPBLENDDOp(OpcodeArgs);
|
||||
void VPBLENDWOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VBROADCASTOp(OpcodeArgs);
|
||||
void VBROADCASTOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VDPPOp(OpcodeArgs);
|
||||
@@ -611,8 +605,7 @@ public:
|
||||
|
||||
template<IROps IROp, size_t ElementSize>
|
||||
void VHADDPOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void VHSUBPOp(OpcodeArgs);
|
||||
void VHSUBPOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VINSERTOp(OpcodeArgs);
|
||||
void VINSERTPSOp(OpcodeArgs);
|
||||
@@ -635,11 +628,9 @@ public:
|
||||
|
||||
void VMPSADBWOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPACKSSOp(OpcodeArgs);
|
||||
void VPACKSSOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPACKUSOp(OpcodeArgs);
|
||||
void VPACKUSOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VPALIGNROp(OpcodeArgs);
|
||||
|
||||
@@ -656,8 +647,7 @@ public:
|
||||
void VPERMDOp(OpcodeArgs);
|
||||
void VPERMQOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPERMILImmOp(OpcodeArgs);
|
||||
void VPERMILImmOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
Ref VPERMILRegOpImpl(OpSize DstSize, size_t ElementSize, Ref Src, Ref Indices);
|
||||
template<size_t ElementSize>
|
||||
@@ -665,8 +655,7 @@ public:
|
||||
|
||||
void VPHADDSWOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPHSUBOp(OpcodeArgs);
|
||||
void VPHSUBOp(OpcodeArgs, size_t ElementSize);
|
||||
void VPHSUBSWOp(OpcodeArgs);
|
||||
|
||||
void VPINSRBOp(OpcodeArgs);
|
||||
@@ -691,40 +680,30 @@ public:
|
||||
|
||||
void VPSHUFBOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool Low>
|
||||
void VPSHUFWOp(OpcodeArgs);
|
||||
void VPSHUFWOp(OpcodeArgs, size_t ElementSize, bool Low);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSLLOp(OpcodeArgs);
|
||||
void VPSLLOp(OpcodeArgs, size_t ElementSize);
|
||||
void VPSLLDQOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void VPSLLIOp(OpcodeArgs);
|
||||
void VPSLLIOp(OpcodeArgs, size_t ElementSize);
|
||||
void VPSLLVOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSRAOp(OpcodeArgs);
|
||||
void VPSRAOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSRAIOp(OpcodeArgs);
|
||||
void VPSRAIOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VPSRAVDOp(OpcodeArgs);
|
||||
void VPSRLVOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSRLDOp(OpcodeArgs);
|
||||
void VPSRLDOp(OpcodeArgs, size_t ElementSize);
|
||||
void VPSRLDQOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPUNPCKHOp(OpcodeArgs);
|
||||
void VPUNPCKHOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPUNPCKLOp(OpcodeArgs);
|
||||
void VPUNPCKLOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPSRLIOp(OpcodeArgs);
|
||||
void VPSRLIOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VSHUFOp(OpcodeArgs);
|
||||
void VSHUFOp(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VTESTPOp(OpcodeArgs);
|
||||
@@ -937,8 +916,7 @@ public:
|
||||
template<size_t ElementSize>
|
||||
void VectorBlend(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VectorVariableBlend(OpcodeArgs);
|
||||
void VectorVariableBlend(OpcodeArgs, size_t ElementSize);
|
||||
void PTestOpImpl(OpSize Size, Ref Dest, Ref Src);
|
||||
void PTestOp(OpcodeArgs);
|
||||
void PHMINPOSUWOp(OpcodeArgs);
|
||||
@@ -1227,6 +1205,11 @@ public:
|
||||
}
|
||||
|
||||
void FlushRegisterCache(bool SRAOnly = false) {
|
||||
// At block boundaries, fix up the carry flag.
|
||||
if (!SRAOnly) {
|
||||
RectifyCarryInvert(CFInvertedABI);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
@@ -1262,7 +1245,23 @@ public:
|
||||
} else {
|
||||
bool Partial = RegCache.Partial & (1ull << Index);
|
||||
unsigned Size = Partial ? 8 : CacheIndexToSize(Index);
|
||||
_StoreContext(Size, CacheIndexClass(Index), Value, CacheIndexToContextOffset(Index));
|
||||
uint64_t NextBit = (1ull << (Index - 1));
|
||||
uint32_t Offset = CacheIndexToContextOffset(Index);
|
||||
auto Class = CacheIndexClass(Index);
|
||||
|
||||
// Use stp where possible to store multiple values at a time. This accelerates AVX.
|
||||
// TODO: this is all really confusing because of backwards iteration,
|
||||
// can we peel back that hack?
|
||||
if ((Bits & NextBit) && !Partial && Size >= 4 && CacheIndexToContextOffset(Index - 1) == Offset - Size && (Offset - Size) / Size < 64) {
|
||||
LOGMAN_THROW_A_FMT(CacheIndexClass(Index - 1) == Class, "construction");
|
||||
LOGMAN_THROW_A_FMT((Offset % Size) == 0, "construction");
|
||||
Ref ValueNext = RegCache.Value[Index - 1];
|
||||
|
||||
_StoreContextPair(Size, Class, ValueNext, Value, Offset - Size);
|
||||
Bits &= ~NextBit;
|
||||
} else {
|
||||
_StoreContext(Size, Class, Value, Offset);
|
||||
}
|
||||
}
|
||||
|
||||
Bits &= ~(1ull << Index);
|
||||
@@ -1345,6 +1344,18 @@ private:
|
||||
bool NZCVDirty {};
|
||||
uint32_t PossiblySetNZCVBits {};
|
||||
|
||||
// Set if the host carry is inverted from the guest carry. This is set after
|
||||
// subtraction, because arm64 and x86 have inverted borrow flags, but clear
|
||||
// after addition.
|
||||
//
|
||||
// All CF access needs to maintain this flag. cfinv may be inserted at the end
|
||||
// of a block to rectify to the FEX convention (current convention: NOT
|
||||
// INVERTED).
|
||||
bool CFInverted {};
|
||||
|
||||
// FEX convention for CF at the end of blocks: INVERTED.
|
||||
const bool CFInvertedABI {true};
|
||||
|
||||
fextl::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
bool HandledLock {false};
|
||||
bool DecodeFailure {false};
|
||||
@@ -1365,8 +1376,7 @@ private:
|
||||
void AVXVectorALUOp(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorUnaryOp(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVXVectorVariableBlend(OpcodeArgs);
|
||||
void AVXVectorVariableBlend(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void AVXVariableShiftImpl(OpcodeArgs, IROps IROp);
|
||||
|
||||
@@ -1436,7 +1446,7 @@ private:
|
||||
void MOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
void VMOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
Ref VFCMPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
Ref VFCMPOpImpl(OpSize Size, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
|
||||
void VTESTOpImpl(OpSize SrcSize, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
@@ -1582,23 +1592,6 @@ private:
|
||||
return IR::SizeToOpSize(GetSrcSize(Op));
|
||||
}
|
||||
|
||||
static inline constexpr unsigned NZCVIndexMask(unsigned BitMask) {
|
||||
unsigned NZCVMask {};
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_OF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_SF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
}
|
||||
return NZCVMask;
|
||||
}
|
||||
|
||||
// Set flag tracking to prepare for an operation that directly writes NZCV. If
|
||||
// some bits are known to be zeroed, the PossiblySetNZCVBits mask can be
|
||||
// passed. Otherwise, it defaults to assuming all bits may be set after
|
||||
@@ -1624,6 +1617,10 @@ private:
|
||||
// Special case of the above where we are known to zero C/V
|
||||
void HandleNZ00Write() {
|
||||
HandleNZCVWrite((1u << 31) | (1u << 30));
|
||||
|
||||
// Host carry will be implicitly zeroed, and we want guest carry zeroed as
|
||||
// well. So do not invert.
|
||||
CFInverted = false;
|
||||
}
|
||||
|
||||
Ref GetNZCV() {
|
||||
@@ -1645,9 +1642,34 @@ private:
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, Ref Res) {
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, Ref Res, bool SetPF = false) {
|
||||
HandleNZ00Write();
|
||||
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
|
||||
|
||||
// x - 0 = x. NZ set according to Res. C always set. V always unset. This
|
||||
// matches what we want since we want carry inverted.
|
||||
//
|
||||
// This is currently worse for 8/16-bit, but that should be optimized. TODO
|
||||
if (SrcSize >= 4) {
|
||||
if (SetPF) {
|
||||
CalculatePF(_SubWithFlags(IR::SizeToOpSize(SrcSize), Res, _Constant(0)));
|
||||
} else {
|
||||
_SubNZCV(IR::SizeToOpSize(SrcSize), Res, _Constant(0));
|
||||
}
|
||||
|
||||
PossiblySetNZCVBits |= 1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
CFInverted = true;
|
||||
} else {
|
||||
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
|
||||
CFInverted = false;
|
||||
|
||||
if (SetPF) {
|
||||
CalculatePF(Res);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void SetNZP_ZeroCV(unsigned SrcSize, Ref Res) {
|
||||
SetNZ_ZeroCV(SrcSize, Res, true);
|
||||
}
|
||||
|
||||
void InsertNZCV(unsigned BitOffset, Ref Value, signed FlagOffset, bool MustMask) {
|
||||
@@ -1692,19 +1714,30 @@ private:
|
||||
PossiblySetNZCVBits |= (1u << Bit);
|
||||
}
|
||||
|
||||
void CarryInvert() {
|
||||
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
// Ensure the carry invert flag matches the desired form. Used before an
|
||||
// operation reading carry or at the end of a block.
|
||||
void RectifyCarryInvert(bool RequiredInvert) {
|
||||
if (CFInverted != RequiredInvert) {
|
||||
if (CTX->HostFeatures.SupportsFlagM && !NZCVDirty) {
|
||||
// Invert as NZCV.
|
||||
_CarryInvert();
|
||||
CachedNZCV = nullptr;
|
||||
} else {
|
||||
// Invert as a GPR
|
||||
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), _Constant(1u << Bit)));
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM && !NZCVDirty) {
|
||||
// Invert as NZCV.
|
||||
_CarryInvert();
|
||||
CachedNZCV = nullptr;
|
||||
} else {
|
||||
// Invert as a GPR
|
||||
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), _Constant(1u << Bit)));
|
||||
CFInverted ^= true;
|
||||
}
|
||||
|
||||
PossiblySetNZCVBits |= 1u << Bit;
|
||||
LOGMAN_THROW_AA_FMT(CFInverted == RequiredInvert, "post condition");
|
||||
}
|
||||
|
||||
void CarryInvert() {
|
||||
CFInverted ^= true;
|
||||
PossiblySetNZCVBits |= 1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
}
|
||||
|
||||
template<unsigned BitOffset>
|
||||
@@ -1712,24 +1745,36 @@ private:
|
||||
SetRFLAG(Value, BitOffset, ValueOffset, MustMask);
|
||||
}
|
||||
|
||||
void SetCFDirect(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
Value = _Xor(OpSize::i64Bit, Value, _InlineConstant(1ull << ValueOffset));
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ValueOffset, MustMask);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void SetCFInverted(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ValueOffset, MustMask);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void SetRFLAG(Ref Value, unsigned BitOffset, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
InsertNZCV(BitOffset, Value, ValueOffset, MustMask);
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (ValueOffset || MustMask) {
|
||||
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
|
||||
}
|
||||
|
||||
if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
StoreRegister(Core::CPUState::PF_AS_GREG, false, Value);
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
StoreRegister(Core::CPUState::AF_AS_GREG, false, Value);
|
||||
} else {
|
||||
if (ValueOffset || MustMask) {
|
||||
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
|
||||
}
|
||||
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
|
||||
// For DF, we need to transform 0/1 into 1/-1
|
||||
if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
|
||||
StoreDF(_SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else {
|
||||
_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
StoreDF(_SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else {
|
||||
_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1756,14 +1801,10 @@ private:
|
||||
|
||||
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_SF_RAW_LOC: return Invert ? CondClassType {COND_PL} : CondClassType {COND_MI};
|
||||
|
||||
case FEXCore::X86State::RFLAG_ZF_RAW_LOC: return Invert ? CondClassType {COND_NEQ} : CondClassType {COND_EQ};
|
||||
|
||||
case FEXCore::X86State::RFLAG_CF_RAW_LOC: return Invert ? CondClassType {COND_ULT} : CondClassType {COND_UGE};
|
||||
|
||||
case FEXCore::X86State::RFLAG_OF_RAW_LOC: return Invert ? CondClassType {COND_FNU} : CondClassType {COND_FU};
|
||||
|
||||
case X86State::RFLAG_SF_RAW_LOC: return {Invert ? COND_PL : COND_MI};
|
||||
case X86State::RFLAG_ZF_RAW_LOC: return {Invert ? COND_NEQ : COND_EQ};
|
||||
case X86State::RFLAG_CF_RAW_LOC: return {Invert ? COND_ULT : COND_UGE};
|
||||
case X86State::RFLAG_OF_RAW_LOC: return {Invert ? COND_FNU : COND_FU};
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
@@ -1865,6 +1906,43 @@ private:
|
||||
return RegCache.Value[Index];
|
||||
}
|
||||
|
||||
RefPair AllocatePair(FEXCore::IR::RegisterClassType Class, uint8_t Size) {
|
||||
if (Class == FPRClass) {
|
||||
return {_AllocateFPR(Size, Size), _AllocateFPR(Size, Size)};
|
||||
} else {
|
||||
return {_AllocateGPR(false), _AllocateGPR(false)};
|
||||
}
|
||||
}
|
||||
|
||||
RefPair LoadContextPair_Uncached(FEXCore::IR::RegisterClassType Class, uint8_t Size, unsigned Offset) {
|
||||
RefPair Values = AllocatePair(Class, Size);
|
||||
_LoadContextPair(Size, Class, Offset, Values.Low, Values.High);
|
||||
return Values;
|
||||
}
|
||||
|
||||
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, uint8_t Size) {
|
||||
LOGMAN_THROW_AA_FMT(Index != DFIndex, "must be pairable");
|
||||
|
||||
// Try to load a pair into the cache
|
||||
uint64_t Bits = (3ull << (uint64_t)Index);
|
||||
if (((RegCache.Partial | RegCache.Cached) & Bits) == 0 && ((Offset / Size) < 64)) {
|
||||
auto Values = LoadContextPair_Uncached(RegClass, Size, Offset);
|
||||
RegCache.Value[Index] = Values.Low;
|
||||
RegCache.Value[Index + 1] = Values.High;
|
||||
RegCache.Cached |= Bits;
|
||||
if (Size == 8) {
|
||||
RegCache.Partial |= Bits;
|
||||
}
|
||||
return Values;
|
||||
}
|
||||
|
||||
// Fallback on a pair of loads
|
||||
return {
|
||||
.Low = LoadRegCache(Offset, Index, RegClass, Size),
|
||||
.High = LoadRegCache(Offset + Size, Index + 1, RegClass, Size),
|
||||
};
|
||||
}
|
||||
|
||||
Ref LoadGPR(uint8_t Reg) {
|
||||
return LoadRegCache(Reg, GPR0Index + Reg, GPRClass, CTX->GetGPRSize());
|
||||
}
|
||||
@@ -1873,6 +1951,10 @@ private:
|
||||
return LoadRegCache(CacheIndexToContextOffset(Index), Index, CacheIndexClass(Index), Size);
|
||||
}
|
||||
|
||||
RefPair LoadContextPair(uint8_t Size, uint8_t Index) {
|
||||
return LoadRegCachePair(CacheIndexToContextOffset(Index), Index, CacheIndexClass(Index), Size);
|
||||
}
|
||||
|
||||
Ref LoadContext(uint8_t Index) {
|
||||
return LoadContext(CacheIndexToSize(Index), Index);
|
||||
}
|
||||
@@ -1912,6 +1994,12 @@ private:
|
||||
|
||||
Ref GetRFLAG(unsigned BitOffset, bool Invert = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
// Handle the CFInverted state internally so GetRFLAG is safe regardless
|
||||
// of the invert state. This simplifies the call sites.
|
||||
if (BitOffset == X86State::RFLAG_CF_RAW_LOC) {
|
||||
Invert ^= CFInverted;
|
||||
}
|
||||
|
||||
if (!(PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset)))) {
|
||||
return _Constant(Invert ? 1 : 0);
|
||||
} else if (NZCVDirty) {
|
||||
@@ -1923,6 +2011,8 @@ private:
|
||||
return Value;
|
||||
}
|
||||
} else {
|
||||
// Because we explicitly inverted for CF above, we use the unsafe
|
||||
// _NZCVSelect rather than the safe CF-aware version.
|
||||
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert), _Constant(1), _Constant(0));
|
||||
}
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
@@ -1956,57 +2046,55 @@ private:
|
||||
return _AddShift(OpSize::i64Bit, X, LoadDF(), ShiftType::LSL, Shift);
|
||||
}
|
||||
|
||||
// Safe version of NZCVSelect that handles inverted carries automatically.
|
||||
Ref NZCVSelect(OpSize OpSize, CondClassType Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
|
||||
switch (Cond) {
|
||||
case IR::COND_UGE: /* cs */
|
||||
case IR::COND_ULT: /* cc */
|
||||
// Invert the condition to match our expectations.
|
||||
if (CarryIsInverted != CFInverted) {
|
||||
Cond = {Cond == COND_UGE ? COND_ULT : COND_UGE};
|
||||
}
|
||||
break;
|
||||
|
||||
case IR::COND_UGT: /* hi */
|
||||
case IR::COND_ULE: /* ls */
|
||||
// No clever optimization we can do here, rectify carry itself.
|
||||
RectifyCarryInvert(CarryIsInverted);
|
||||
break;
|
||||
|
||||
default:
|
||||
// No other condition codes read carry so no need to rectify.
|
||||
break;
|
||||
}
|
||||
|
||||
return _NZCVSelect(OpSize, Cond, TrueV, FalseV);
|
||||
}
|
||||
|
||||
// Compares two floats and sets flags for a COMISS instruction
|
||||
void Comiss(size_t ElementSize, Ref Src1, Ref Src2, bool InvalidateAF = false) {
|
||||
// First, set flags according to Arm FCMP.
|
||||
HandleNZCVWrite();
|
||||
_FCmp(ElementSize, Src1, Src2);
|
||||
CFInverted = false;
|
||||
ComissFlags(InvalidateAF);
|
||||
}
|
||||
|
||||
// Sets flags for a COMISS instruction
|
||||
void ComissFlags(bool InvalidateAF = false) {
|
||||
// Now set COMISS flags by converts NZCV from the Arm representation to an
|
||||
// eXternal representation that's totally not a euphemism for x86, nuh-uh.
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
// We need to set PF according to the unordered flag. We'd rather do this
|
||||
// after axflag, since some impls fuse fcmp+axflag, so we want to do this
|
||||
// after. We can recover "unordered" after axflag as (Z && !C), but
|
||||
// there's no condition code for this so it would take 2 instructions
|
||||
// instead of one, which seems worse than doing 1 op before and breaking
|
||||
// the fusion.
|
||||
//
|
||||
// We set PF to unordered (V), but our PF representation is inverted so we
|
||||
// actually set to !V. This is one instruction with the VC cond code.
|
||||
Ref PFInvert = _NZCVSelect(OpSize::i32Bit, CondClassType {COND_FNU}, _Constant(1), _Constant(0));
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PFInvert);
|
||||
|
||||
// For the rest, this one weird a64 instruction maps exactly to what x86
|
||||
// needs. What a coincidence!
|
||||
_AXFlag();
|
||||
PossiblySetNZCVBits = ~0;
|
||||
|
||||
// It does assume we invert CF internally, which is still TODO for us. For
|
||||
// now, add a cfinv to deal. Hopefully we delete this later.
|
||||
CarryInvert();
|
||||
} else {
|
||||
Ref Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
Ref C_inv = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
Ref V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
// We want to zero SF/OF, and then set CF/ZF. Zeroing up front lets us do
|
||||
// this all with shifted-or's on non-flagm platforms.
|
||||
ZeroNZCV();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Or(OpSize::i32Bit, C_inv, V));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Or(OpSize::i32Bit, Z, V));
|
||||
|
||||
// Note that we store PF inverted.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Xor(OpSize::i32Bit, V, _Constant(1)));
|
||||
}
|
||||
// We need to set PF according to the unordered flag. We'd rather do this
|
||||
// after axflag, since some impls fuse fcmp+axflag, so we want to do this
|
||||
// after. We can recover "unordered" after axflag as (Z && !C), but
|
||||
// there's no condition code for this so it would take 2 instructions
|
||||
// instead of one, which seems worse than doing 1 op before and breaking
|
||||
// the fusion.
|
||||
//
|
||||
// We set PF to unordered (V), but our PF representation is inverted so we
|
||||
// actually set to !V. This is one instruction with the VC cond code.
|
||||
Ref V_inv = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(V_inv);
|
||||
|
||||
if (!InvalidateAF) {
|
||||
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
|
||||
@@ -2014,6 +2102,28 @@ private:
|
||||
// byte to zero will indeed zero AF as intended.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Convert NZCV from the Arm representation to an eXternal representation
|
||||
// that's totally not a euphemism for x86, nuh-uh. But maps to exactly we
|
||||
// need, what a coincidence!
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
_AXFlag();
|
||||
} else {
|
||||
// AXFLAG is defined in the Arm spec as
|
||||
//
|
||||
// gt: nzCv -> nzCv
|
||||
// lt: Nzcv -> nzcv <==> 1 + 0
|
||||
// eq: nZCv -> nZCv <==> 1 + (~0)
|
||||
// un: nzCV -> nZcv <==> 0 + 0
|
||||
//
|
||||
// For the latter 3 cases, we therefore get the right NZCV by adding V_inv
|
||||
// to (eq ? ~0 : 0). The remaining case is forced with ccmn.
|
||||
Ref Eq = NZCVSelect(OpSize::i64Bit, {COND_EQ}, _Constant(~0ull), _Constant(0));
|
||||
_CondAddNZCV(OpSize::i64Bit, V_inv, Eq, {COND_FLEU}, 0x2 /* nzCv */);
|
||||
}
|
||||
|
||||
PossiblySetNZCVBits = ~0;
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
// Set x87 comparison flags based on the result set by Arm FCMP. Clobbers
|
||||
@@ -2027,9 +2137,10 @@ private:
|
||||
// Convert to x86 flags, saves us from or'ing after.
|
||||
_AXFlag();
|
||||
PossiblySetNZCVBits = ~0;
|
||||
CFInverted = true;
|
||||
|
||||
// Copy the values. CF is inverted from the axflag result, ZF is as-is.
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true));
|
||||
// Copy the values.
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
} else {
|
||||
Ref Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
@@ -2050,18 +2161,10 @@ private:
|
||||
auto OldPF = GetRFLAG(X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
HandleNZCV_RMW();
|
||||
CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF));
|
||||
CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF, CFInverted));
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
std::pair<Ref, Ref> ExtractPair(OpSize Size, Ref Pair) {
|
||||
// Extract high first. This is a hack to improve coalescing.
|
||||
Ref Hi = _ExtractElementPair(Size, Pair, 1);
|
||||
Ref Lo = _ExtractElementPair(Size, Pair, 0);
|
||||
|
||||
return std::make_pair(Lo, Hi);
|
||||
}
|
||||
|
||||
// Helper to derive Dest by a given builder-using Expression with the opcode
|
||||
// replaced with NewOp. Useful for generic building code. Not safe in general.
|
||||
// but does the right handling of ImplicitFlagClobber at least and must be
|
||||
@@ -2132,7 +2235,7 @@ private:
|
||||
CachedIndexedNamedVectorConstants.clear();
|
||||
}
|
||||
|
||||
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP) const;
|
||||
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP);
|
||||
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
|
||||
@@ -2226,6 +2329,8 @@ private:
|
||||
void CalculatePF(Ref Res);
|
||||
void CalculateAF(Ref Src1, Ref Src2);
|
||||
|
||||
Ref IncrementByCarry(OpSize OpSize, Ref Src);
|
||||
|
||||
void CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub);
|
||||
Ref CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2);
|
||||
Ref CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2);
|
||||
@@ -2241,14 +2346,7 @@ private:
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_BEXTR(Ref Src);
|
||||
void CalculateFlags_BLSI(uint8_t SrcSize, Ref Src);
|
||||
void CalculateFlags_BLSMSK(uint8_t SrcSize, Ref Res, Ref Src);
|
||||
void CalculateFlags_BLSR(uint8_t SrcSize, Ref Res, Ref Src);
|
||||
void CalculateFlags_POPCOUNT(Ref Src);
|
||||
void CalculateFlags_BZHI(uint8_t SrcSize, Ref Result, Ref Src);
|
||||
void CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result);
|
||||
void CalculateFlags_RDRAND(Ref Src);
|
||||
/** @} */
|
||||
|
||||
Ref AndConst(FEXCore::IR::OpSize Size, Ref Node, uint64_t Const) {
|
||||
@@ -2285,8 +2383,16 @@ private:
|
||||
uint64_t Entry;
|
||||
IROp_IRHeader* CurrentHeader {};
|
||||
|
||||
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) {
|
||||
if (Class == FPRClass) {
|
||||
return CTX->IsVectorAtomicTSOEnabled();
|
||||
} else {
|
||||
return CTX->IsAtomicTSOEnabled();
|
||||
}
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) {
|
||||
if (CTX->IsAtomicTSOEnabled()) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
@@ -2294,7 +2400,7 @@ private:
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) {
|
||||
if (CTX->IsAtomicTSOEnabled()) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
@@ -2302,7 +2408,7 @@ private:
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, uint8_t Align = 1) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !A.NonTSO;
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
@@ -2312,8 +2418,46 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
AddressMode SelectPairAddressMode(AddressMode A, uint8_t Size) {
|
||||
AddressMode Out {};
|
||||
|
||||
signed OffsetEl = A.Offset / Size;
|
||||
if ((A.Offset % Size) == 0 && OffsetEl >= -64 && OffsetEl < 64) {
|
||||
Out.Offset = A.Offset;
|
||||
A.Offset = 0;
|
||||
}
|
||||
|
||||
Out.Base = LoadEffectiveAddress(A, true, false);
|
||||
return Out;
|
||||
}
|
||||
|
||||
|
||||
RefPair LoadMemPair(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Base, unsigned Offset) {
|
||||
RefPair Values = AllocatePair(Class, Size);
|
||||
_LoadMemPair(Class, Size, Base, Offset, Values.Low, Values.High);
|
||||
return Values;
|
||||
}
|
||||
|
||||
RefPair _LoadMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, uint8_t Align = 1) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use ldp if possible, otherwise fallback on two loads.
|
||||
if (!AtomicTSO && !A.Segment && Size >= 4 & Size <= 16) {
|
||||
A = SelectPairAddressMode(A, Size);
|
||||
return LoadMemPair(Class, Size, A.Base, A.Offset);
|
||||
} else {
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
return {
|
||||
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
|
||||
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, Ref Value, uint8_t Align = 1) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !A.NonTSO;
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
@@ -2323,10 +2467,47 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
void _StoreMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, Ref Value1, Ref Value2, uint8_t Align = 1) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use stp if possible, otherwise fallback on two stores.
|
||||
if (!AtomicTSO && !A.Segment && Size >= 4 & Size <= 16) {
|
||||
A = SelectPairAddressMode(A, Size);
|
||||
_StoreMemPair(Class, Size, Value1, Value2, A.Base, A.Offset);
|
||||
} else {
|
||||
_StoreMemAutoTSO(Class, Size, A, Value1, 1);
|
||||
A.Offset += Size;
|
||||
_StoreMemAutoTSO(Class, Size, A, Value2, 1);
|
||||
}
|
||||
}
|
||||
|
||||
Ref Prefetch(bool ForStore, bool Stream, uint8_t CacheLevel, Ref ssa0) {
|
||||
return _Prefetch(ForStore, Stream, CacheLevel, ssa0, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
Ref Pop(uint8_t Size, Ref SP_RMW) {
|
||||
Ref Value = _AllocateGPR(false);
|
||||
_Pop(Size, SP_RMW, Value);
|
||||
return Value;
|
||||
}
|
||||
|
||||
Ref Pop(uint8_t Size) {
|
||||
Ref SP = _RMWHandle(LoadGPRRegister(X86State::REG_RSP));
|
||||
Ref Value = _AllocateGPR(false);
|
||||
|
||||
_Pop(Size, SP, Value);
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, SP);
|
||||
return Value;
|
||||
}
|
||||
|
||||
void Push(uint8_t Size, Ref Value) {
|
||||
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
|
||||
auto NewSP = _Push(CTX->GetGPRSize(), Size, Value, OldSP);
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
}
|
||||
|
||||
void InstallHostSpecificOpcodeHandlers();
|
||||
|
||||
///< Segment telemetry tracking
|
||||
|
||||
@@ -508,10 +508,11 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
LOGMAN_THROW_AA_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
|
||||
}
|
||||
|
||||
return {
|
||||
.Low = _LoadMemAutoTSO(FPRClass, 16, A, 1),
|
||||
.High = NeedsHigh ? _LoadMemAutoTSO(FPRClass, 16, HighA, 1) : nullptr,
|
||||
};
|
||||
if (NeedsHigh) {
|
||||
return _LoadMemPairAutoTSO(FPRClass, 16, A, 1);
|
||||
} else {
|
||||
return {.Low = _LoadMemAutoTSO(FPRClass, 16, A, 1)};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -557,13 +558,10 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
|
||||
} else {
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
|
||||
|
||||
_StoreMemAutoTSO(FPRClass, 16, A, Src.Low, 1);
|
||||
|
||||
if (Src.High) {
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
_StoreMemAutoTSO(FPRClass, 16, HighA, Src.High, 1);
|
||||
_StoreMemPairAutoTSO(FPRClass, 16, A, Src.Low, Src.High, 1);
|
||||
} else {
|
||||
_StoreMemAutoTSO(FPRClass, 16, A, Src.Low, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1158,7 +1156,7 @@ void OpDispatchBuilder::AVX128_VFCMP(OpcodeArgs) {
|
||||
};
|
||||
|
||||
AVX128_VectorBinaryImpl(Op, GetSrcSize(Op), ElementSize, [this, &Capture](size_t _ElementSize, Ref Src1, Ref Src2) {
|
||||
return VFCMPOpImpl(Capture.Op, _ElementSize, Src1, Src2, Capture.CompType);
|
||||
return VFCMPOpImpl(OpSize::i128Bit, _ElementSize, Src1, Src2, Capture.CompType);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1678,27 +1676,27 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
}
|
||||
}();
|
||||
|
||||
auto Convert = [this](size_t Size, Ref Src, IROps Op) -> Ref {
|
||||
auto Convert = [this](Ref Src, IROps Op) -> Ref {
|
||||
size_t ElementSize = SrcElementSize;
|
||||
if (Widen) {
|
||||
DeriveOp(Extended, Op, _VSXTL(Size, ElementSize, Src));
|
||||
DeriveOp(Extended, Op, _VSXTL(OpSize::i128Bit, ElementSize, Src));
|
||||
Src = Extended;
|
||||
ElementSize <<= 1;
|
||||
}
|
||||
|
||||
return _Vector_SToF(Size, ElementSize, Src);
|
||||
return _Vector_SToF(OpSize::i128Bit, ElementSize, Src);
|
||||
};
|
||||
|
||||
RefPair Result {};
|
||||
Result.Low = Convert(Size, Src.Low, IROps::OP_VSXTL);
|
||||
Result.Low = Convert(Src.Low, IROps::OP_VSXTL);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
} else {
|
||||
if (Widen) {
|
||||
Result.High = Convert(Size, Src.Low, IROps::OP_VSXTL2);
|
||||
Result.High = Convert(Src.Low, IROps::OP_VSXTL2);
|
||||
} else {
|
||||
Result.High = Convert(Size, Src.High, IROps::OP_VSXTL);
|
||||
Result.High = Convert(Src.High, IROps::OP_VSXTL);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2173,18 +2171,20 @@ void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_SaveAVXState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref Upper = AVX128_LoadXMMRegister(i, true);
|
||||
_StoreMem(FPRClass, 16, Upper, MemBase, _Constant(i * 16 + 576), 16, MEM_OFFSET_SXTX, 1);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
RefPair Pair = LoadContextPair(16, AVXHigh0Index + i);
|
||||
_StoreMemPair(FPRClass, 16, Pair.Low, Pair.High, MemBase, i * 16 + 576);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_RestoreAVXState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref YMMHReg = _LoadMem(FPRClass, 16, MemBase, _Constant(i * 16 + 576), 16, MEM_OFFSET_SXTX, 1);
|
||||
AVX128_StoreXMMRegister(i, YMMHReg, true);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
auto YMMHRegs = LoadMemPair(FPRClass, 16, MemBase, i * 16 + 576);
|
||||
|
||||
AVX128_StoreXMMRegister(i, YMMHRegs.Low, true);
|
||||
AVX128_StoreXMMRegister(i + 1, YMMHRegs.High, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2234,7 +2234,7 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs) {
|
||||
|
||||
// For 256-bit, we need to split up the operation. This is nontrivial.
|
||||
// Let's go the simple route here.
|
||||
Ref ZF, CF;
|
||||
Ref ZF, CFInv;
|
||||
Ref ZeroConst = _Constant(0);
|
||||
Ref OneConst = _Constant(1);
|
||||
|
||||
@@ -2278,13 +2278,12 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs) {
|
||||
|
||||
// ExtGPR will either be [0, 8] or [0, 16] If 0 then set Flag.
|
||||
auto ExtGPR = _VExtractToGPR(OpSize::i128Bit, ElementSize, AddWide, 0);
|
||||
CF = _Select(IR::COND_EQ, ExtGPR, ZeroConst, OneConst, ZeroConst);
|
||||
CFInv = _Select(IR::COND_NEQ, ExtGPR, ZeroConst, OneConst, ZeroConst);
|
||||
}
|
||||
|
||||
// As in PTest, this sets Z appropriately while zeroing the rest of NZCV.
|
||||
SetNZ_ZeroCV(32, ZF);
|
||||
SetRFLAG(CF, FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
SetCFInverted(CFInv);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -2324,15 +2323,14 @@ void OpDispatchBuilder::AVX128_PTest(OpcodeArgs) {
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
Test2 = _Select(FEXCore::IR::COND_EQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
Test2 = _Select(FEXCore::IR::COND_NEQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
|
||||
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
|
||||
// Set ZF according to Test1. SF will be zeroed since we do a 32-bit test on
|
||||
// the results of a 16-bit value from the UMaxV, so the 32-bit sign bit is
|
||||
// cleared even if the 16-bit scalars were negative.
|
||||
SetNZ_ZeroCV(32, Test1);
|
||||
SetRFLAG(Test2, FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
SetCFInverted(Test2);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -2488,14 +2486,14 @@ void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Sr
|
||||
|
||||
const OpSize ElementSize = Op->Flags & X86Tables::DecodeFlags::FLAG_OPTION_AVX_W ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
auto Dest = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit);
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
|
||||
auto Dest = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit).Low;
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit).Low;
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit).Low;
|
||||
|
||||
RefPair Sources[3] = {Dest, Src1, Src2};
|
||||
Ref Sources[3] = {Dest, Src1, Src2};
|
||||
|
||||
DeriveOp(Result_Low, IROp,
|
||||
_VFMLAScalarInsert(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].Low, Sources[Src2Idx - 1].Low, Sources[AddendIdx - 1].Low));
|
||||
_VFMLAScalarInsert(OpSize::i128Bit, ElementSize, Dest, Sources[Src1Idx - 1], Sources[Src2Idx - 1], Sources[AddendIdx - 1]));
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result_Low));
|
||||
}
|
||||
|
||||
|
||||
@@ -46,6 +46,9 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, Ref Src) {
|
||||
InvalidateDeferredFlags();
|
||||
}
|
||||
|
||||
// PF and CF are both stored inverted, so hoist the invert.
|
||||
auto SrcInverted = _Not(OpSize::i32Bit, Src);
|
||||
|
||||
for (size_t i = 0; i < NumFlags; ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
|
||||
@@ -59,15 +62,15 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, Ref Src) {
|
||||
// So we write out the whole flags byte to AF without an extract.
|
||||
static_assert(FEXCore::X86State::RFLAG_AF_RAW_LOC == 4);
|
||||
SetRFLAG(Src, FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
} else if (FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
// PF is stored parity flipped
|
||||
Ref Tmp = _Bfe(OpSize::i32Bit, 1, FlagOffset, Src);
|
||||
Tmp = _Xor(OpSize::i32Bit, Tmp, _Constant(1));
|
||||
SetRFLAG(Tmp, FlagOffset);
|
||||
} else if (FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC || FlagOffset == FEXCore::X86State::RFLAG_CF_RAW_LOC) {
|
||||
// PF and CF are both stored parity flipped.
|
||||
SetRFLAG(SrcInverted, FlagOffset, FlagOffset, true);
|
||||
} else {
|
||||
SetRFLAG(Src, FlagOffset, FlagOffset, true);
|
||||
}
|
||||
}
|
||||
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
@@ -267,6 +270,12 @@ void OpDispatchBuilder::CalculateDeferredFlags() {
|
||||
NZCVDirty = false;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
|
||||
// If CF not inverted, we use .cc since the increment happens when the
|
||||
// condition is false. If CF inverted, invert to use .cs. A bit mindbendy.
|
||||
return _NZCVSelectIncrement(OpSize, {CFInverted ? COND_UGE : COND_ULT}, Src, Src);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
@@ -276,25 +285,27 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
if (SrcSize >= 4) {
|
||||
RectifyCarryInvert(false);
|
||||
HandleNZCV_RMW();
|
||||
Res = _AdcWithFlags(OpSize, Src1, Src2);
|
||||
CFInverted = false;
|
||||
} else {
|
||||
// Need to zero-extend for correct comparisons below
|
||||
Src2 = _Bfe(OpSize, SrcSize * 8, 0, Src2);
|
||||
|
||||
// Note that we do not extend Src2PlusCF, since we depend on proper
|
||||
// 32-bit arithmetic to correctly handle the Src2 = 0xffff case.
|
||||
Ref Src2PlusCF = _Adc(OpSize, _Constant(0), Src2);
|
||||
Ref Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
// Need to zero-extend for the comparison.
|
||||
Res = _Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_ULT, Res, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Res, Src2PlusCF, One, Zero);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
|
||||
SetCFInverted(SelectCFInv);
|
||||
CalculateOF(SrcSize, Res, Src1, Src2, false);
|
||||
}
|
||||
|
||||
@@ -311,28 +322,26 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
// Rectify input carry
|
||||
CarryInvert();
|
||||
|
||||
// Arm's subtraction has inverted CF from x86, so rectify the input and
|
||||
// invert the output.
|
||||
RectifyCarryInvert(true);
|
||||
HandleNZCV_RMW();
|
||||
Res = _SbbWithFlags(OpSize, Src1, Src2);
|
||||
|
||||
// Rectify output carry
|
||||
CarryInvert();
|
||||
CFInverted = true;
|
||||
} else {
|
||||
// Zero extend for correct comparison behaviour with Src1 = 0xffff.
|
||||
Src1 = _Bfe(OpSize, SrcSize * 8, 0, Src1);
|
||||
Src2 = _Bfe(OpSize, SrcSize * 8, 0, Src2);
|
||||
|
||||
auto Src2PlusCF = _Adc(OpSize, _Constant(0), Src2);
|
||||
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
Res = _Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_ULT, Src1, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Src1, Src2PlusCF, One, Zero);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
|
||||
SetCFInverted(SelectCFInv);
|
||||
CalculateOF(SrcSize, Res, Src1, Src2, true);
|
||||
}
|
||||
|
||||
@@ -342,7 +351,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
auto OldCFInv = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
|
||||
HandleNZCVWrite();
|
||||
|
||||
@@ -358,12 +367,13 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// If we're updating CF, we need to invert it for correctness. If we're not
|
||||
// updating CF, we need to restore the CF since we stomped over it.
|
||||
// If we're updating CF, we need it to be inverted because SubNZCV is inverted
|
||||
// from x86. If we're not updating CF, we need to restore the CF since we
|
||||
// stomped over it.
|
||||
if (UpdateCF) {
|
||||
CarryInvert();
|
||||
CFInverted = true;
|
||||
} else {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
|
||||
SetCFInverted(OldCFInv);
|
||||
}
|
||||
|
||||
return Res;
|
||||
@@ -371,7 +381,7 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
auto OldCFInv = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
|
||||
HandleNZCVWrite();
|
||||
|
||||
@@ -388,8 +398,11 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
CalculatePF(Res);
|
||||
|
||||
// We stomped over CF while calculation flags, restore it.
|
||||
if (!UpdateCF) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
|
||||
if (UpdateCF) {
|
||||
// Adds match between x86 and arm64.
|
||||
CFInverted = false;
|
||||
} else {
|
||||
SetCFInverted(OldCFInv);
|
||||
}
|
||||
|
||||
return Res;
|
||||
@@ -404,10 +417,11 @@ void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, Ref Res, Ref High) {
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
|
||||
_SubNZCV(OpSize::i64Bit, High, SignBit);
|
||||
|
||||
// If High = SignBit, then sets to nZcv. Else sets to nzCV. Since SF/ZF
|
||||
// undefined, this does what we need.
|
||||
// If High = SignBit, then sets to nZCv. Else sets to nzcV. Since SF/ZF
|
||||
// undefined, this does what we need after inverting carry.
|
||||
auto Zero = _Constant(0);
|
||||
_CondAddNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
@@ -421,9 +435,10 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
// The result register will be all zero if it can't fit due to how multiplication behaves
|
||||
_SubNZCV(Size, High, Zero);
|
||||
|
||||
// If High = 0, then sets to nZcv. Else sets to nzCV. Since SF/ZF undefined,
|
||||
// If High = 0, then sets to nZCv. Else sets to nzcV. Since SF/ZF undefined,
|
||||
// this does what we need.
|
||||
_CondAddNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
_CondSubNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
@@ -452,7 +467,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref U
|
||||
// nothing to do in that case since we already cleared CF above.
|
||||
auto SrcSizeBits = SrcSize * 8;
|
||||
if (Shift < SrcSizeBits) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, SrcSizeBits - Shift, true);
|
||||
SetCFDirect(Src1, SrcSizeBits - Shift, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -477,11 +492,8 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift - 1, true);
|
||||
}
|
||||
// Extract the last bit shifted in to CF
|
||||
SetCFDirect(Src1, Shift - 1, true);
|
||||
|
||||
CalculatePF(Res);
|
||||
InvalidateAF();
|
||||
@@ -497,11 +509,8 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
// set below.
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src1, Shift - 1, true);
|
||||
}
|
||||
// Extract the last bit shifted in to CF
|
||||
SetCFDirect(Src1, Shift - 1, true);
|
||||
|
||||
CalculatePF(Res);
|
||||
InvalidateAF();
|
||||
@@ -546,64 +555,6 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BEXTR(Ref Src) {
|
||||
// ZF is set properly. CF and OF are defined as being set to zero. SF, PF, and
|
||||
// AF are undefined.
|
||||
SetNZ_ZeroCV(GetOpSize(Src), Src);
|
||||
InvalidatePF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, Ref Result) {
|
||||
// CF is cleared if Src is zero, otherwise it's set. However, Src is zero iff
|
||||
// Result is zero, so we can test the result instead. So, CF is just the
|
||||
// inverted ZF.
|
||||
//
|
||||
// ZF/SF/OF set as usual.
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
InvalidatePF_AF();
|
||||
|
||||
auto CFOp = GetRFLAG(X86State::RFLAG_ZF_RAW_LOC, true /* Invert */);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
InvalidatePF_AF();
|
||||
|
||||
// CF set according to the Src
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto CFOp = _Select(IR::COND_EQ, Src, Zero, One, Zero);
|
||||
|
||||
// The output of BLSMSK is always nonzero, so TST will clear Z (along with C
|
||||
// and O) while setting S.
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto CFOp = _Select(IR::COND_EQ, Src, Zero, One, Zero);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
InvalidatePF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_POPCOUNT(Ref Result) {
|
||||
// We need to set ZF while clearing the rest of NZCV. The result of a popcount
|
||||
// is in the range [0, 63]. In particular, it is always positive. So a
|
||||
// combined NZ test will correctly zero SF/CF/OF while setting ZF.
|
||||
SetNZ_ZeroCV(OpSize::i32Bit, Result);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
InvalidatePF_AF();
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
// Test ZF of result, SF is undefined so this is ok.
|
||||
@@ -613,16 +564,7 @@ void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result) {
|
||||
// Result is <= SrcSize * 8, we equivalently check if the log2(SrcSize * 8)
|
||||
// bit is set. No masking is needed because no higher bits could be set.
|
||||
unsigned CarryBit = FEXCore::ilog2(SrcSize * 8u);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Result, CarryBit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RDRAND(Ref Src) {
|
||||
// OF, SF, ZF, AF, PF all zero
|
||||
ZeroNZCV();
|
||||
ZeroPF_AF();
|
||||
|
||||
// CF is set to the incoming source
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
SetCFDirect(Result, CarryBit);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -702,8 +702,7 @@ void OpDispatchBuilder::MOVQMMXOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Src, 1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::MOVMSKOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, size_t ElementSize) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
uint8_t NumElements = Size / ElementSize;
|
||||
|
||||
@@ -752,9 +751,6 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::MOVMSKOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::MOVMSKOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
@@ -780,8 +776,7 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PUNPCKLOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUNPCKLOp(OpcodeArgs, size_t ElementSize) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
@@ -791,13 +786,7 @@ void OpDispatchBuilder::PUNPCKLOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PUNPCKLOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKLOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKLOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKLOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPUNPCKLOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPUNPCKLOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -817,13 +806,7 @@ void OpDispatchBuilder::VPUNPCKLOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPUNPCKLOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKLOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKLOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKLOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PUNPCKHOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUNPCKHOp(OpcodeArgs, size_t ElementSize) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
@@ -832,13 +815,7 @@ void OpDispatchBuilder::PUNPCKHOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PUNPCKHOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKHOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKHOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PUNPCKHOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPUNPCKHOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPUNPCKHOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -858,11 +835,6 @@ void OpDispatchBuilder::VPUNPCKHOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPUNPCKHOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKHOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKHOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPUNPCKHOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::GeneratePSHUFBMask(uint8_t SrcSize) {
|
||||
// PSHUFB doesn't 100% match VTBL behaviour
|
||||
// VTBL will set the element zero if the index is greater than
|
||||
@@ -949,8 +921,7 @@ void OpDispatchBuilder::PSHUFW8ByteOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
template<bool Low>
|
||||
void OpDispatchBuilder::PSHUFWOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSHUFWOp(OpcodeArgs, bool Low) {
|
||||
constexpr auto IdentityCopy = 0b11'10'01'00;
|
||||
|
||||
uint16_t Shuffle = Op->Src[1].Data.Literal.Value;
|
||||
@@ -1000,9 +971,6 @@ void OpDispatchBuilder::PSHUFWOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSHUFWOp<false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSHUFWOp<true>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle) {
|
||||
constexpr auto IdentityCopy = 0b11'10'01'00;
|
||||
|
||||
@@ -1222,8 +1190,7 @@ void OpDispatchBuilder::PSHUFDOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Single128Bit4ByteVectorShuffle(Src, Shuffle), -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize, bool Low>
|
||||
void OpDispatchBuilder::VPSHUFWOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSHUFWOp(OpcodeArgs, size_t ElementSize, bool Low) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is256Bit = SrcSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
auto Shuffle = Op->Src[1].Literal();
|
||||
@@ -1271,9 +1238,6 @@ void OpDispatchBuilder::VPSHUFWOp(OpcodeArgs) {
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::VPSHUFWOp<2, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSHUFWOp<2, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSHUFWOp<4, true>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::SHUFOpImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, Ref Src1, Ref Src2, uint8_t Shuffle) {
|
||||
// Since 256-bit variants and up don't lane cross, we can construct
|
||||
@@ -1466,8 +1430,7 @@ Ref OpDispatchBuilder::SHUFOpImpl(OpcodeArgs, size_t DstSize, size_t ElementSize
|
||||
return Dest;
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::SHUFOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::SHUFOp(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Src1Node = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src2Node = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
uint8_t Shuffle = Op->Src[1].Literal();
|
||||
@@ -1475,11 +1438,8 @@ void OpDispatchBuilder::SHUFOp(OpcodeArgs) {
|
||||
Ref Result = SHUFOpImpl(Op, GetDstSize(Op), ElementSize, Src1Node, Src2Node, Shuffle);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::SHUFOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::SHUFOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VSHUFOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VSHUFOp(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Src1Node = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2Node = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
uint8_t Shuffle = Op->Src[2].Literal();
|
||||
@@ -1487,8 +1447,6 @@ void OpDispatchBuilder::VSHUFOp(OpcodeArgs) {
|
||||
Ref Result = SHUFOpImpl(Op, GetDstSize(Op), ElementSize, Src1Node, Src2Node, Shuffle);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::VSHUFOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VSHUFOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VANDNOp(OpcodeArgs) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
@@ -1524,8 +1482,7 @@ template void OpDispatchBuilder::VHADDPOp<IR::OP_VADDP, 4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VHADDPOp<IR::OP_VFADDP, 4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VHADDPOp<IR::OP_VFADDP, 8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VBROADCASTOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VBROADCASTOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
Ref Result {};
|
||||
|
||||
@@ -1544,12 +1501,6 @@ void OpDispatchBuilder::VBROADCASTOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VBROADCASTOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VBROADCASTOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VBROADCASTOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VBROADCASTOp<8>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VBROADCASTOp<16>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PINSROpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, const X86Tables::DecodedOperand& Imm) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
@@ -1660,8 +1611,7 @@ void OpDispatchBuilder::VINSERTPSOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PExtrOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PExtrOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
@@ -1672,7 +1622,7 @@ void OpDispatchBuilder::PExtrOp(OpcodeArgs) {
|
||||
// is the same except that REX.W or VEX.W is set to 1. Incredibly frustrating.
|
||||
// Use the destination size as the element size in this case.
|
||||
size_t OverridenElementSize = ElementSize;
|
||||
if constexpr (ElementSize == 4) {
|
||||
if (ElementSize == 4) {
|
||||
OverridenElementSize = DstSize;
|
||||
}
|
||||
|
||||
@@ -1693,11 +1643,6 @@ void OpDispatchBuilder::PExtrOp(OpcodeArgs) {
|
||||
_VStoreVectorElement(16, OverridenElementSize, Src, Index, Dest);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PExtrOp<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PExtrOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PExtrOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PExtrOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VEXTRACT128Op(OpcodeArgs) {
|
||||
const auto DstIsXMM = Op->Dest.IsGPR();
|
||||
const auto StoreSize = DstIsXMM ? 32 : 16;
|
||||
@@ -1761,8 +1706,7 @@ Ref OpDispatchBuilder::PSRLDOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref
|
||||
return _VUShrSWide(Size, ElementSize, Src, ShiftVec);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSRLDOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSRLDOp(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = PSRLDOpImpl(Op, ElementSize, Dest, Src);
|
||||
@@ -1770,12 +1714,7 @@ void OpDispatchBuilder::PSRLDOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSRLDOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRLDOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRLDOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSRLDOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSRLDOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -1789,12 +1728,7 @@ void OpDispatchBuilder::VPSRLDOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSRLDOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRLDOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRLDOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSRLI(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSRLI(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t ShiftConstant = Op->Src[1].Literal();
|
||||
if (ShiftConstant == 0) [[unlikely]] {
|
||||
// Nothing to do, value is already in Dest.
|
||||
@@ -1808,12 +1742,7 @@ void OpDispatchBuilder::PSRLI(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Shift, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSRLI<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRLI<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRLI<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSRLIOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSRLIOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const uint64_t ShiftConstant = Op->Src[1].Literal();
|
||||
@@ -1832,10 +1761,6 @@ void OpDispatchBuilder::VPSRLIOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSRLIOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRLIOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRLIOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PSLLIImpl(OpcodeArgs, size_t ElementSize, Ref Src, uint64_t Shift) {
|
||||
if (Shift == 0) [[unlikely]] {
|
||||
// If zero-shift then just return the source.
|
||||
@@ -1845,8 +1770,7 @@ Ref OpDispatchBuilder::PSLLIImpl(OpcodeArgs, size_t ElementSize, Ref Src, uint64
|
||||
return _VShlI(Size, ElementSize, Src, Shift);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSLLI(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSLLI(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t ShiftConstant = Op->Src[1].Literal();
|
||||
if (ShiftConstant == 0) [[unlikely]] {
|
||||
// Nothing to do, value is already in Dest.
|
||||
@@ -1859,12 +1783,7 @@ void OpDispatchBuilder::PSLLI(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSLLI<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSLLI<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSLLI<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSLLIOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSLLIOp(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t ShiftConstant = Op->Src[1].Literal();
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
@@ -1878,10 +1797,6 @@ void OpDispatchBuilder::VPSLLIOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSLLIOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSLLIOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSLLIOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PSLLImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
|
||||
@@ -1889,8 +1804,7 @@ Ref OpDispatchBuilder::PSLLImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref Shi
|
||||
return _VUShlSWide(Size, ElementSize, Src, ShiftVec);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSLL(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSLL(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = PSLLImpl(Op, ElementSize, Dest, Src);
|
||||
@@ -1898,12 +1812,7 @@ void OpDispatchBuilder::PSLL(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSLL<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSLL<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSLL<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSLLOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSLLOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -1917,10 +1826,6 @@ void OpDispatchBuilder::VPSLLOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSLLOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSLLOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSLLOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PSRAOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
|
||||
@@ -1928,8 +1833,7 @@ Ref OpDispatchBuilder::PSRAOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref S
|
||||
return _VSShrSWide(Size, ElementSize, Src, ShiftVec);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSRAOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSRAOp(OpcodeArgs, size_t ElementSize) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = PSRAOpImpl(Op, ElementSize, Dest, Src);
|
||||
@@ -1937,11 +1841,7 @@ void OpDispatchBuilder::PSRAOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSRAOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRAOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSRAOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSRAOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -1955,9 +1855,6 @@ void OpDispatchBuilder::VPSRAOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSRAOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRAOp<4>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PSRLDQ(OpcodeArgs) {
|
||||
const uint64_t Shift = Op->Src[1].Literal();
|
||||
if (Shift == 0) [[unlikely]] {
|
||||
@@ -2059,8 +1956,7 @@ void OpDispatchBuilder::VPSLLDQOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PSRAIOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PSRAIOp(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t Shift = Op->Src[1].Literal();
|
||||
if (Shift == 0) [[unlikely]] {
|
||||
// Nothing to do, value is already in Dest.
|
||||
@@ -2074,11 +1970,7 @@ void OpDispatchBuilder::PSRAIOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::PSRAIOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PSRAIOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPSRAIOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPSRAIOp(OpcodeArgs, size_t ElementSize) {
|
||||
const uint64_t Shift = Op->Src[1].Literal();
|
||||
const auto Size = GetDstSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
@@ -2097,9 +1989,6 @@ void OpDispatchBuilder::VPSRAIOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPSRAIOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPSRAIOp<4>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::AVXVariableShiftImpl(OpcodeArgs, IROps IROp) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
@@ -2529,9 +2418,7 @@ void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType) {
|
||||
}
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::VFCMPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType) {
|
||||
Ref Result {};
|
||||
switch (CompType & 0x7) {
|
||||
case 0x0: // EQ
|
||||
@@ -2567,7 +2454,7 @@ void OpDispatchBuilder::VFCMPOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags);
|
||||
const uint8_t CompType = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
Ref Result = VFCMPOpImpl(Op, ElementSize, Dest, Src, CompType);
|
||||
Ref Result = VFCMPOpImpl(OpSizeFromSrc(Op), ElementSize, Dest, Src, CompType);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -2585,7 +2472,7 @@ void OpDispatchBuilder::AVXVFCMPOp(OpcodeArgs) {
|
||||
|
||||
Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags);
|
||||
Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Ref Result = VFCMPOpImpl(Op, ElementSize, Src1, Src2, CompType);
|
||||
Ref Result = VFCMPOpImpl(OpSizeFromSrc(Op), ElementSize, Src1, Src2, CompType);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -2736,37 +2623,33 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
|
||||
// MXCSR_MASK: Mask for writes to the MXCSR register
|
||||
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
|
||||
// This is implementation dependent
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
Ref MMReg = LoadContext(MM0Index + i);
|
||||
|
||||
_StoreMem(FPRClass, 16, MMReg, MemBase, _Constant(i * 16 + 32), 16, MEM_OFFSET_SXTX, 1);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
|
||||
RefPair MMRegs = LoadContextPair(16, MM0Index + i);
|
||||
_StoreMemPair(FPRClass, 16, MMRegs.Low, MMRegs.High, MemBase, i * 16 + 32);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveSSEState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref XMMReg = LoadXMMRegister(i);
|
||||
|
||||
_StoreMem(FPRClass, 16, XMMReg, MemBase, _Constant(i * 16 + 160), 16, MEM_OFFSET_SXTX, 1);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
_StoreMemPair(FPRClass, 16, LoadXMMRegister(i), LoadXMMRegister(i + 1), MemBase, i * 16 + 160);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveMXCSRState(Ref MemBase) {
|
||||
_StoreMem(GPRClass, 4, GetMXCSR(), MemBase, _Constant(24), 4, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
// Store the mask for all bits.
|
||||
_StoreMem(GPRClass, 4, _Constant(0xFFFF), MemBase, _Constant(28), 4, MEM_OFFSET_SXTX, 1);
|
||||
// Store MXCSR and the mask for all bits.
|
||||
_StoreMemPair(GPRClass, 4, GetMXCSR(), _Constant(0xFFFF), MemBase, 24);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveAVXState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref Upper = _VDupElement(32, 16, LoadXMMRegister(i), 1);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
Ref Upper0 = _VDupElement(32, 16, LoadXMMRegister(i + 0), 1);
|
||||
Ref Upper1 = _VDupElement(32, 16, LoadXMMRegister(i + 1), 1);
|
||||
|
||||
_StoreMem(FPRClass, 16, Upper, MemBase, _Constant(i * 16 + 576), 16, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemPair(FPRClass, 16, Upper0, Upper1, MemBase, i * 16 + 576);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2868,18 +2751,22 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
|
||||
StoreContext(AbridgedFTWIndex, _LoadMem(GPRClass, 1, MemBase, _Constant(4), 2, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
auto MMReg = _LoadMem(FPRClass, 16, MemBase, _Constant(i * 16 + 32), 16, MEM_OFFSET_SXTX, 1);
|
||||
StoreContext(MM0Index + i, MMReg);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
|
||||
auto MMRegs = LoadMemPair(FPRClass, 16, MemBase, i * 16 + 32);
|
||||
|
||||
StoreContext(MM0Index + i, MMRegs.Low);
|
||||
StoreContext(MM0Index + i + 1, MMRegs.High);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreSSEState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref XMMReg = _LoadMem(FPRClass, 16, MemBase, _Constant(i * 16 + 160), 16, MEM_OFFSET_SXTX, 1);
|
||||
StoreXMMRegister(i, XMMReg);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
auto XMMRegs = LoadMemPair(FPRClass, 16, MemBase, i * 16 + 160);
|
||||
|
||||
StoreXMMRegister(i, XMMRegs.Low);
|
||||
StoreXMMRegister(i + 1, XMMRegs.High);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2896,11 +2783,12 @@ void OpDispatchBuilder::RestoreMXCSRState(Ref MXCSR) {
|
||||
void OpDispatchBuilder::RestoreAVXState(Ref MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
Ref XMMReg = LoadXMMRegister(i);
|
||||
Ref YMMHReg = _LoadMem(FPRClass, 16, MemBase, _Constant(i * 16 + 576), 16, MEM_OFFSET_SXTX, 1);
|
||||
Ref YMM = _VInsElement(32, 16, 1, 0, XMMReg, YMMHReg);
|
||||
StoreXMMRegister(i, YMM);
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
Ref XMMReg0 = LoadXMMRegister(i + 0);
|
||||
Ref XMMReg1 = LoadXMMRegister(i + 1);
|
||||
auto YMMHRegs = LoadMemPair(FPRClass, 16, MemBase, i * 16 + 576);
|
||||
StoreXMMRegister(i + 0, _VInsElement(32, 16, 1, 0, XMMReg0, YMMHRegs.Low));
|
||||
StoreXMMRegister(i + 1, _VInsElement(32, 16, 1, 0, XMMReg1, YMMHRegs.High));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3015,8 +2903,7 @@ void OpDispatchBuilder::PACKUSOp(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::PACKUSOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PACKUSOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPACKUSOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPACKUSOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
@@ -3032,9 +2919,6 @@ void OpDispatchBuilder::VPACKUSOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPACKUSOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPACKUSOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::PACKSSOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
@@ -3047,8 +2931,7 @@ void OpDispatchBuilder::PACKSSOp(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::PACKSSOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PACKSSOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPACKSSOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPACKSSOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
@@ -3064,9 +2947,6 @@ void OpDispatchBuilder::VPACKSSOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPACKSSOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPACKSSOp<4>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PMULLOpImpl(OpSize Size, size_t ElementSize, bool Signed, Ref Src1, Ref Src2) {
|
||||
if (Size == OpSize::i64Bit) {
|
||||
if (Signed) {
|
||||
@@ -3505,8 +3385,7 @@ void OpDispatchBuilder::HSUBP(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::HSUBP<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::HSUBP<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VHSUBPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VHSUBPOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
@@ -3523,9 +3402,6 @@ void OpDispatchBuilder::VHSUBPOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VHSUBPOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VHSUBPOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PHSUBOpImpl(OpSize Size, Ref Src1, Ref Src2, size_t ElementSize) {
|
||||
auto Even = _VUnZip(Size, ElementSize, Src1, Src2);
|
||||
auto Odd = _VUnZip2(Size, ElementSize, Src1, Src2);
|
||||
@@ -3543,8 +3419,7 @@ void OpDispatchBuilder::PHSUB(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::PHSUB<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::PHSUB<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPHSUBOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPHSUBOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
@@ -3558,9 +3433,6 @@ void OpDispatchBuilder::VPHSUBOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPHSUBOp<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPHSUBOp<4>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::PHADDSOpImpl(OpSize Size, Ref Src1, Ref Src2) {
|
||||
const uint8_t ElementSize = 2;
|
||||
|
||||
@@ -4027,8 +3899,7 @@ template void OpDispatchBuilder::VectorBlend<2>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorBlend<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorBlend<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs, size_t ElementSize) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
@@ -4047,14 +3918,10 @@ void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) {
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::VectorVariableBlend<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorVariableBlend<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorVariableBlend<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs, size_t ElementSize) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
constexpr auto ElementSizeBits = ElementSize * 8;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
@@ -4067,9 +3934,6 @@ void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs) {
|
||||
Ref Result = _VBSL(SrcSize, Shifted, Src2, Src1);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
template void OpDispatchBuilder::AVXVectorVariableBlend<1>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVectorVariableBlend<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXVectorVariableBlend<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) {
|
||||
// Invalidate deferred flags early
|
||||
@@ -4088,15 +3952,14 @@ void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) {
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
Test2 = _Select(FEXCore::IR::COND_EQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
Test2 = _Select(FEXCore::IR::COND_NEQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
|
||||
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
|
||||
// Set ZF according to Test1. SF will be zeroed since we do a 32-bit test on
|
||||
// the results of a 16-bit value from the UMaxV, so the 32-bit sign bit is
|
||||
// cleared even if the 16-bit scalars were negative.
|
||||
SetNZ_ZeroCV(32, Test1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Test2);
|
||||
|
||||
SetCFInverted(Test2);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -4130,12 +3993,11 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, size_t ElementSize, Ref Src1
|
||||
Ref ZeroConst = _Constant(0);
|
||||
Ref OneConst = _Constant(1);
|
||||
|
||||
Ref CFResult = _Select(IR::COND_EQ, AndNotGPR, ZeroConst, OneConst, ZeroConst);
|
||||
Ref CFInv = _Select(IR::COND_NEQ, AndNotGPR, ZeroConst, OneConst, ZeroConst);
|
||||
|
||||
// As in PTest, this sets Z appropriately while zeroing the rest of NZCV.
|
||||
SetNZ_ZeroCV(32, AndGPR);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFResult);
|
||||
|
||||
SetCFInverted(CFInv);
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -4890,8 +4752,7 @@ void OpDispatchBuilder::VZEROOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs, size_t ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Selector = Op->Src[1].Literal() & 0xFF;
|
||||
@@ -4899,7 +4760,7 @@ void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = LoadZeroVector(DstSize);
|
||||
|
||||
if constexpr (ElementSize == 8) {
|
||||
if (ElementSize == 8) {
|
||||
Result = _VInsElement(DstSize, ElementSize, 0, Selector & 0b0001, Result, Src);
|
||||
Result = _VInsElement(DstSize, ElementSize, 1, (Selector & 0b0010) >> 1, Result, Src);
|
||||
|
||||
@@ -4924,9 +4785,6 @@ void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VPERMILImmOp<4>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VPERMILImmOp<8>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::VPERMILRegOpImpl(OpSize DstSize, size_t ElementSize, Ref Src, Ref Indices) {
|
||||
// NOTE: See implementation of VPERMD for the gist of what we do to make this work.
|
||||
//
|
||||
@@ -5068,6 +4926,7 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
|
||||
// Set all of the necessary flags. NZCV stored in bits 28...31 like the hw op.
|
||||
SetNZCV(IntermediateResult);
|
||||
CFInverted = false;
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
@@ -622,7 +622,7 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, size_t Width, bool Integer, OpDispatch
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
|
||||
SetCFDirect(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
|
||||
@@ -31,32 +31,9 @@ X86GeneratedCode::X86GeneratedCode() {
|
||||
0x0F, 0x37, // CALLBACKRET FEX Instruction
|
||||
};
|
||||
|
||||
// Signal return handlers need to be bit-exact to what the Linux kernel provides in VDSO.
|
||||
// GDB and unwinding libraries key off of these instructions to understand if the stack frame is a signal frame or not.
|
||||
// This two code sections match exactly what libSegFault expects.
|
||||
//
|
||||
// Typically this handlers are provided by the 32-bit VDSO thunk library, but that isn't available in all cases.
|
||||
// Falling back to this generated code segment still allows a backtrace to work, just might not show
|
||||
// the symbol as VDSO since there is no ELF to parse.
|
||||
constexpr std::array<uint8_t, 9> sigreturn_32_code = {
|
||||
0x58, // pop eax
|
||||
0xb8, 0x77, 0x00, 0x00, 0x00, // mov eax, 0x77
|
||||
0xcd, 0x80, // int 0x80
|
||||
0x90, // nop
|
||||
};
|
||||
|
||||
constexpr std::array<uint8_t, 7> rt_sigreturn_32_code = {
|
||||
0xb8, 0xad, 0x00, 0x00, 0x00, // mov eax, 0xad
|
||||
0xcd, 0x80, // int 0x80
|
||||
};
|
||||
|
||||
CallbackReturn = reinterpret_cast<uint64_t>(CodePtr);
|
||||
sigreturn_32 = CallbackReturn + SignalReturnCode.size();
|
||||
rt_sigreturn_32 = sigreturn_32 + sigreturn_32_code.size();
|
||||
|
||||
memcpy(reinterpret_cast<void*>(CallbackReturn), &SignalReturnCode.at(0), SignalReturnCode.size());
|
||||
memcpy(reinterpret_cast<void*>(sigreturn_32), &sigreturn_32_code.at(0), sigreturn_32_code.size());
|
||||
memcpy(reinterpret_cast<void*>(rt_sigreturn_32), &rt_sigreturn_32_code.at(0), rt_sigreturn_32_code.size());
|
||||
|
||||
mprotect(CodePtr, CODE_SIZE, PROT_READ);
|
||||
#endif
|
||||
|
||||
@@ -17,8 +17,6 @@ public:
|
||||
~X86GeneratedCode();
|
||||
|
||||
uint64_t CallbackReturn {};
|
||||
uint64_t sigreturn_32 {};
|
||||
uint64_t rt_sigreturn_32 {};
|
||||
|
||||
private:
|
||||
void* CodePtr {};
|
||||
|
||||
@@ -97,7 +97,6 @@ struct TrampolineInstanceInfo {
|
||||
// Opaque type pointing to an instance of HostToGuestTrampolineTemplate and its
|
||||
// embedded TrampolineInstanceInfo
|
||||
struct HostToGuestTrampolinePtr;
|
||||
const auto HostToGuestTrampolineSize = __stop_HostToGuestTrampolineTemplate - __start_HostToGuestTrampolineTemplate;
|
||||
|
||||
static TrampolineInstanceInfo& GetInstanceInfo(HostToGuestTrampolinePtr* Trampoline) {
|
||||
const auto Length = __stop_HostToGuestTrampolineTemplate - __start_HostToGuestTrampolineTemplate;
|
||||
@@ -438,6 +437,8 @@ MakeHostTrampolineForGuestFunction(void* HostPacker, uintptr_t GuestTarget, uint
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding host trampoline for guest function {:#x} via unpacker {:#x}", GuestTarget, GuestUnpacker);
|
||||
|
||||
const auto HostToGuestTrampolineSize = __stop_HostToGuestTrampolineTemplate - __start_HostToGuestTrampolineTemplate;
|
||||
|
||||
if (ThunkHandler->HostTrampolineInstanceDataAvailable < HostToGuestTrampolineSize) {
|
||||
const auto allocation_step = 16 * 1024;
|
||||
ThunkHandler->HostTrampolineInstanceDataAvailable = allocation_step;
|
||||
|
||||
@@ -44,7 +44,7 @@
|
||||
" * Textual class to group IR ops by type",
|
||||
"* DestClass",
|
||||
" * SSA class of the return when the return type is `SSA`",
|
||||
" * Not used if the destination type is one of {GPR, GPRPair, FPR}",
|
||||
" * Not used if the destination type is one of {GPR, FPR}",
|
||||
"* DestSize",
|
||||
" * The size of the destination type",
|
||||
"* EmitValidation",
|
||||
@@ -67,6 +67,8 @@
|
||||
"constexpr uint8_t COND_SLT = 11",
|
||||
"constexpr uint8_t COND_SGT = 12",
|
||||
"constexpr uint8_t COND_SLE = 13",
|
||||
"constexpr uint8_t COND_TSTZ = 14 /* bit test zero */",
|
||||
"constexpr uint8_t COND_TSTNZ = 15 /* bit test nonzero */",
|
||||
|
||||
"constexpr uint8_t COND_FLU = 16 /* float less or unordred */",
|
||||
"constexpr uint8_t COND_FGE = 17 /* float greater or equal */",
|
||||
@@ -81,7 +83,6 @@
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {2}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRFixedClass {3}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRPairClass {4}",
|
||||
"constexpr FEXCore::IR::RegisterClassType ComplexClass {5}",
|
||||
"constexpr FEXCore::IR::RegisterClassType InvalidClass {7}",
|
||||
"",
|
||||
@@ -146,7 +147,6 @@
|
||||
"OpSize": "FEXCore::IR::OpSize",
|
||||
"SSA": "OrderedNode*",
|
||||
"GPR": "OrderedNode*",
|
||||
"GPRPair": "OrderedNode*",
|
||||
"FPR": "OrderedNode*",
|
||||
"FenceType": "FenceType",
|
||||
"RegisterClass": "RegisterClassType",
|
||||
@@ -245,24 +245,38 @@
|
||||
"Print SSA:$Value": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Debug operation that prints an SSA value to the console",
|
||||
"May only print 64bits of the value",
|
||||
"Depending on backend, may only support GPR printing"
|
||||
],
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) != GPRPairClass"
|
||||
]
|
||||
"May only print 64bits of the value"]
|
||||
},
|
||||
"GPRPair = RDRAND i1:$GetReseeded": {
|
||||
"GPR = AllocateGPR i1:$ForPair": {
|
||||
"Desc": ["Silly pseudo-instruction to allocate a register for a future destination",
|
||||
"Note: if an instruction uses allocated destinations-as-sources,",
|
||||
"it cannot use a regular destination too. This ensures RA correctness.",
|
||||
"This is a kludge to deal with the IR's lack of multiple destinations",
|
||||
"If ForPair is set, RA will try to allocate the base of a register pair"],
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = AllocateFPR u8:#RegisterSize, u8:#ElementSize": {
|
||||
"Desc": ["Like AllocateGPR, but for FPR"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"GPR = AllocateGPRAfter GPR:$After": {
|
||||
"Desc": ["Silly pseudo-instruction to allocate a register for a future destination",
|
||||
"This is a kludge to deal with the IR's lack of multiple destinations",
|
||||
"RA will attempt to allocate to the register after $After.",
|
||||
"It may not succeed."],
|
||||
"DestSize": "8"
|
||||
},
|
||||
"GPR = RDRAND i1:$GetReseeded": {
|
||||
"Desc": ["Uses the hardware random number generator to generate a 64bit number",
|
||||
"The boolean argument asks if we should be reading the reseeded number or not",
|
||||
"Reseeded RNG calculation is more expensive and will be heavier to use",
|
||||
"The first GPR pair element is the 64-bit number",
|
||||
"The second GPR pair element is a bool if the number is valid",
|
||||
"Returns the 64-bit number",
|
||||
"Sets the Z flag if the number is valid.",
|
||||
"RNG hardware is allowed to fail early and return. Software must always check this"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
"Yield": {
|
||||
"HasSideEffects": true,
|
||||
@@ -277,10 +291,7 @@
|
||||
},
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, u8:$CompareSize{0}, i1:$FromNZCV{false}": {
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Cmp1) == WalkFindRegClass($Cmp2)"
|
||||
]
|
||||
"RAOverride": "2"
|
||||
},
|
||||
"ExitFunction GPR:$NewRIP": {
|
||||
"Desc": ["Exits the current JIT function with a target RIP"
|
||||
@@ -317,42 +328,18 @@
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
"GPRPair = CPUID GPR:$Function, GPR:$Leaf": {
|
||||
"Desc": ["Calls in to the CPUID handler function to return emulated CPUID",
|
||||
"Returns a 128bit GPR pair that fits emulated EAX, EBX, EDX, ECX respectively"
|
||||
],
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
"GPR:$EAX, GPR:$EBX, GPR:$ECX, GPR:$EDX = CPUID GPR:$Function, GPR:$Leaf": {
|
||||
"Desc": ["Calls in to the CPUID handler function to return emulated CPUID"],
|
||||
"DestSize": "4",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPRPair = XGetBV GPR:$Function": {
|
||||
"Desc": ["Calls in to the XCR handler function to return emulated XCR",
|
||||
"Returns a 64bit GPR pair that fits emulated EAX, EDX respectively"
|
||||
],
|
||||
"DestSize": "8",
|
||||
"NumElements": "2"
|
||||
"GPR:$EAX, GPR:$EDX = XGetBV GPR:$Function": {
|
||||
"Desc": ["Calls in to the XCR handler function to return emulated XCR"],
|
||||
"DestSize": "4",
|
||||
"HasSideEffects": true
|
||||
}
|
||||
},
|
||||
"Moves": {
|
||||
"GPR = ExtractElementPair OpSize:#Size, GPRPair:$Pair, u8:$Element": {
|
||||
"Desc": ["Extracts a register for the register pair"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"GPRPair = CreateElementPair OpSize:#Size, GPR:$Lower, GPR:$Upper": {
|
||||
"Desc": ["Inserts a register for the register pair",
|
||||
"ssa0 is the lower incoming register",
|
||||
"ssa1 is the upper incoming register"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"NumElements": "2",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = Copy GPR:$Source": {
|
||||
"Desc": ["GPR copy, generated by RA to split live ranges"],
|
||||
"DestSize": "8"
|
||||
@@ -403,6 +390,20 @@
|
||||
]
|
||||
},
|
||||
|
||||
"SSA:$Value1, SSA:$Value2 = LoadContextPair u8:#ByteSize, RegisterClass:$Class, u32:$Offset": {
|
||||
"Desc": ["Loads a pair of values from the context with offset",
|
||||
"Value0 = Ctx[Offset], Value1 = Ctx[Offset + ByteSize]"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
"StoreContext u8:#ByteSize, RegisterClass:$Class, SSA:$Value, u32:$Offset": {
|
||||
"Desc": ["Stores a value to the context with offset",
|
||||
"Ctx[Offset] = Value",
|
||||
@@ -420,6 +421,24 @@
|
||||
]
|
||||
},
|
||||
|
||||
"StoreContextPair u8:#ByteSize, RegisterClass:$Class, SSA:$Value1, SSA:$Value2, u32:$Offset": {
|
||||
"Desc": ["Stores a pair of values to the context with offset",
|
||||
"Ctx[Offset] = Value1, Ctx[Offset + ByteSize] = Value2",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value1) == $Class",
|
||||
"WalkFindRegClass($Value2) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
|
||||
]
|
||||
},
|
||||
|
||||
"SSA = LoadContextIndexed GPR:$Index, u8:#ByteSize, u32:$BaseOffset, u32:$Stride, RegisterClass:$Class": {
|
||||
"Desc": ["Loads a value from the context with offset and indexed by SSA value",
|
||||
"Dest = Ctx[BaseOffset + Index * Stride]"
|
||||
@@ -493,6 +512,12 @@
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"SSA:$Value1, SSA:$Value2 = LoadMemPair RegisterClass:$Class, u8:#Size, GPR:$Addr, u32:$Offset": {
|
||||
"Desc": ["Load a pair of values from memory."],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
"StoreMem RegisterClass:$Class, u8:#Size, SSA:$Value, GPR:$Addr, GPR:$Offset, u8:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": [ "Stores a value to memory.",
|
||||
"Zero Extends if value's type is too small",
|
||||
@@ -505,6 +530,19 @@
|
||||
]
|
||||
},
|
||||
|
||||
"StoreMemPair RegisterClass:$Class, u8:#Size, SSA:$Value1, SSA:$Value2, GPR:$Addr, u32:$Offset": {
|
||||
"Desc": [ "Stores a pair of values to memory.",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value1) == $Class",
|
||||
"WalkFindRegClass($Value2) == $Class"
|
||||
]
|
||||
},
|
||||
|
||||
"SSA = LoadMemTSO RegisterClass:$Class, u8:#Size, GPR:$Addr, GPR:$Offset, u8:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a x86 TSO compatible load from memory. Offset must be Invalid()."
|
||||
],
|
||||
@@ -598,6 +636,22 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = RMWHandle GPR:$Value": {
|
||||
"Desc": [
|
||||
"This is a special move that indicates the result will be poisoned by a non-SSA instruction writing to its result.",
|
||||
"In effect, it serves to prevent invalid optimizations with non-SSA instructions."
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"TiedSource": 0
|
||||
},
|
||||
"GPR:$Addr, GPR:$Value = Pop u8:$Size, GPR:$Addr": {
|
||||
"Desc": [
|
||||
"Pops a value from the address, updating the new pointer after incrementing.",
|
||||
"The address is incremented by the size via an RMW source/destintaion."
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = MemSet i1:$IsAtomic, u8:$Size, GPR:$Prefix, GPR:$Addr, GPR:$Value, GPR:$Length, GPR:$Direction": {
|
||||
"Desc": ["Duplicates behaviour of x86 STOS repeat",
|
||||
"Returns the final address that gets generated without the prefix appended."
|
||||
@@ -605,13 +659,12 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
"GPRPair = MemCpy i1:$IsAtomic, u8:$Size, GPR:$Dest, GPR:$Src, GPR:$Length, GPR:$Direction": {
|
||||
"GPR:$DstAddress, GPR:$SrcAddress = MemCpy i1:$IsAtomic, u8:$Size, GPR:$Dest, GPR:$Src, GPR:$Length, GPR:$Direction": {
|
||||
"Desc": ["Duplicates behaviour of x86 MOVS repeat",
|
||||
"Returns the final addresses of destination and src addresses after they have been incremented or decremented"
|
||||
"Returns the final addresses after they have been incremented or decremented"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
"DestSize": "8"
|
||||
},
|
||||
"CacheLineClear GPR:$Addr, i1:$Serialize": {
|
||||
"Desc": ["Does a 64 byte cacheline clear at the address specified",
|
||||
@@ -708,19 +761,17 @@
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPRPair = CASPair OpSize:#Size, GPRPair:$Expected, GPRPair:$Desired, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Does a compare and exchange with two GPRPair values",
|
||||
"GPR:$Lo, GPR:$Hi = CASPair OpSize:#Size, GPR:$ExpectedLo, GPR:$ExpectedHi, GPR:$DesiredLo, GPR:$DesiredHi, GPR:$Addr": {
|
||||
"Desc": ["Does a compare and exchange with two pairs of values",
|
||||
"ssa0 is the comparison value",
|
||||
"ssa1 is the new value",
|
||||
"ssa2 is the memory location",
|
||||
"Returns a pair containing the value in memory"
|
||||
"Returns the lower & upper halves of the value in memory."
|
||||
],
|
||||
"HasDest": true,
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"NumElements": "2",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"AtomicAdd OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
@@ -933,14 +984,6 @@
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
"GPRPair = TruncElementPair GPRPair:$Pair, u8:#ByteSize": {
|
||||
"Desc": [
|
||||
"Truncates each element of a pair to the destination size",
|
||||
"TODO: This IR op should get removed"
|
||||
],
|
||||
"DestSize": "ByteSize * 2",
|
||||
"NumElements": "2"
|
||||
},
|
||||
"GPR = CycleCounter": {
|
||||
"Desc": ["Returns the host 64bit cycle counter",
|
||||
"Useful when emulating rdtsc",
|
||||
@@ -1128,6 +1171,21 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AdcZero OpSize:#Size, GPR:$Src1": {
|
||||
"Desc": ["Adds GPR with inverted carry-in"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AdcZeroWithFlags OpSize:#Size, GPR:$Src1": {
|
||||
"Desc": ["Adds and set NZCV for the sum of GPR and inverted carry-in given as NZCV"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = SbbWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Subtracts and set NZCV for the difference of two GPRs and carry-in given as NZCV"],
|
||||
"HasSideEffects": true,
|
||||
@@ -1179,10 +1237,11 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"CmpPairZ OpSize:#Size, GPRPair:$Src1, GPRPair:$Src2": {
|
||||
"CmpPairZ OpSize:#Size, GPR:$Src1Lo, GPR:$Src1Hi, GPR:$Src2Lo, GPR:$Src2Hi": {
|
||||
"Desc": ["Compares register pairs and sets Z accordingly, preserving N/Z/V.",
|
||||
"This accelerates cmpxchg."],
|
||||
"HasSideEffects": true
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"SubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the difference of two GPRs. ",
|
||||
@@ -1296,12 +1355,16 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = ShiftFlags OpSize:$Size, GPR:$Result, GPR:$Src1, ShiftType:$Shift, GPR:$Src2, GPR:$PFInput": {
|
||||
"GPR = ShiftFlags OpSize:$Size, GPR:$Result, GPR:$Src1, ShiftType:$Shift, GPR:$Src2, GPR:$PFInput, i1:$InvertCF": {
|
||||
"Desc": ["Set NZCV flags for specified variable integer shift with given result.",
|
||||
"Returns updated raw PF."],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
"RotateFlags OpSize:$Size, GPR:$Result, GPR:$Shift, i1:$Left": {
|
||||
"Desc": ["Set NZCV flags for specified variable integer rotate with given result."],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Ror OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer rotate right"
|
||||
],
|
||||
@@ -1445,6 +1508,16 @@
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = NZCVSelectIncrement OpSize:#ResultSize, CondClass:$Cond, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
"Desc": ["Select and increment based on value in NZCV flags",
|
||||
"op:",
|
||||
"Dest = Cond ? TrueVal : (FalseVal + 1)"
|
||||
],
|
||||
"DestSize": "ResultSize",
|
||||
"EmitValidation": [
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Select OpSize:#ResultSize, OpSize:$CompareSize, CondClass:$Cond, SSA:$Cmp1, SSA:$Cmp2, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
"Desc": ["Ternary selection of GPRs",
|
||||
"op:",
|
||||
@@ -1717,41 +1790,45 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMLAScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"FPR = VFMLAScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (Vector1 * Vector2) + Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending.",
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFMLSScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"FPR = VFMLSScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (Vector1 * Vector2) - Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending.",
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFNMLAScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"FPR = VFNMLAScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (-Vector1 * Vector2) + Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending.",
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFNMLSScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"FPR = VFNMLSScalarInsert u8:#RegisterSize, u8:#ElementSize, FPR:$Upper, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (-Vector1 * Vector2) - Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending.",
|
||||
"Upper elements copied from Upper"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
"TiedSource": 0
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
|
||||
@@ -52,7 +52,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {"EQ", "NEQ", "UGE", "ULT", "MI", "PL", "VS", "VC",
|
||||
"UGT", "ULE", "SGE", "SLT", "SGT", "SLE", "ANDZ", "ANDNZ",
|
||||
"UGT", "ULE", "SGE", "SLT", "SGT", "SLE", "TSTZ", "TSTNZ",
|
||||
"FLU", "FGE", "FLEU", "FGT", "FU", "FNU"};
|
||||
|
||||
*out << CondNames[Arg];
|
||||
@@ -77,8 +77,6 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
*out << "FPR";
|
||||
} else if (Arg == FPRFixedClass.Val) {
|
||||
*out << "FPRFixed";
|
||||
} else if (Arg == GPRPairClass.Val) {
|
||||
*out << "GPRPair";
|
||||
} else {
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
@@ -100,7 +98,6 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
@@ -316,7 +313,6 @@ void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocation
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::GPRPairClass.Val: *out << "(GPRPair"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
|
||||
@@ -38,7 +38,6 @@ FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
auto Class = GetOpRegClass(Node);
|
||||
switch (Class) {
|
||||
case GPRClass:
|
||||
case GPRPairClass:
|
||||
case FPRClass:
|
||||
case GPRFixedClass:
|
||||
case FPRFixedClass:
|
||||
|
||||
@@ -16,7 +16,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <bit>
|
||||
@@ -79,16 +78,6 @@ private:
|
||||
|
||||
fextl::unordered_map<uint64_t, Ref> ConstPool;
|
||||
|
||||
// Pool inline constant generation. These are typically very small and pool efficiently.
|
||||
fextl::robin_map<uint64_t, Ref> InlineConstantGen;
|
||||
Ref CreateInlineConstant(IREmitter* IREmit, uint64_t Constant) {
|
||||
const auto it = InlineConstantGen.find(Constant);
|
||||
if (it != InlineConstantGen.end()) {
|
||||
return it->second;
|
||||
}
|
||||
auto Result = InlineConstantGen.insert_or_assign(Constant, IREmit->_InlineConstant(Constant));
|
||||
return Result.first->second;
|
||||
}
|
||||
bool SupportsTSOImm9 {};
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
|
||||
@@ -112,7 +101,7 @@ private:
|
||||
|
||||
if (IsSIMM9 || IsExtended) {
|
||||
IREmit->SetWriteCursor(IR.GetNode(Offset));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Offset_Index, CreateInlineConstant(IREmit, Imm));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Offset_Index, IREmit->_InlineConstant(Imm));
|
||||
OffsetScale = 1;
|
||||
}
|
||||
}
|
||||
@@ -151,6 +140,14 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
bool IsConstant1 = IREmit->IsValueConstant(IROp->Args[0], &Constant1);
|
||||
bool IsConstant2 = IREmit->IsValueConstant(IROp->Args[1], &Constant2);
|
||||
|
||||
/* IsImmAddSub assumes the constants are sign-extended, take care of that
|
||||
* here so we get the optimization for 32-bit adds too.
|
||||
*/
|
||||
if (Op->Header.Size == 4) {
|
||||
Constant1 = (int64_t)(int32_t)Constant1;
|
||||
Constant2 = (int64_t)(int32_t)Constant2;
|
||||
}
|
||||
|
||||
if (IsConstant1 && IsConstant2 && IROp->Op == OP_ADD) {
|
||||
uint64_t NewConstant = (Constant1 + Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
@@ -197,7 +194,8 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// Order matter for short circuit evaluation, subsequent ifs read constant2.
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
uint64_t NewConstant = (Constant1 & Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (Constant2 == 1) {
|
||||
@@ -209,7 +207,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
Constant2 == 1 && Constant3 == 0) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
}
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID() || (Constant2 & getMask(IROp)) == getMask(IROp)) {
|
||||
// AND with same value results in original value
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
}
|
||||
@@ -483,14 +481,14 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
// If the CPUID needs a constant leaf to be optimized then this can't work if we didn't const-prop the leaf register.
|
||||
if (!(SupportsConstant.NeedsLeaf == CPUIDEmu::NeedsLeafConstant::NEEDSLEAFCONSTANT && !IsConstantLeaf)) {
|
||||
// Calculate the constant data and replace all uses.
|
||||
// DCE will remove the CPUID IR operation.
|
||||
const auto ConstantCPUIDResult = CPUID->RunFunction(ConstantFunction, ConstantLeaf);
|
||||
uint64_t ResultsLower = (static_cast<uint64_t>(ConstantCPUIDResult.ebx) << 32) | ConstantCPUIDResult.eax;
|
||||
uint64_t ResultsUpper = (static_cast<uint64_t>(ConstantCPUIDResult.edx) << 32) | ConstantCPUIDResult.ecx;
|
||||
const auto Result = CPUID->RunFunction(ConstantFunction, ConstantLeaf);
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto ElementPair = IREmit->_CreateElementPair(IR::OpSize::i128Bit, IREmit->_Constant(ResultsLower), IREmit->_Constant(ResultsUpper));
|
||||
// Replace all CPUID uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ElementPair);
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEAX), IREmit->_Constant(Result.eax));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEBX), IREmit->_Constant(Result.ebx));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutECX), IREmit->_Constant(Result.ecx));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEDX), IREmit->_Constant(Result.edx));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -502,12 +500,11 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
uint64_t ConstantFunction {};
|
||||
if (IREmit->IsValueConstant(Op->Function, &ConstantFunction) && CPUID->DoesXCRFunctionReportConstantData(ConstantFunction)) {
|
||||
const auto ConstantXCRResult = CPUID->RunXCRFunction(ConstantFunction);
|
||||
const auto Result = CPUID->RunXCRFunction(ConstantFunction);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto ElementPair =
|
||||
IREmit->_CreateElementPair(IR::OpSize::i64Bit, IREmit->_Constant(ConstantXCRResult.eax), IREmit->_Constant(ConstantXCRResult.edx));
|
||||
// Replace all xgetbv uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ElementPair);
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEAX), IREmit->_Constant(Result.eax));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEDX), IREmit->_Constant(Result.edx));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -569,8 +566,6 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
|
||||
void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR) {
|
||||
InlineConstantGen.clear();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
switch (IROp->Op) {
|
||||
case OP_LSHR:
|
||||
@@ -588,7 +583,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
Constant2 &= 63;
|
||||
}
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -604,7 +599,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
// constant would be in bounds after the JIT's 24/16 shift.
|
||||
if (ARMEmitter::IsImmAddSub(Constant2) && IROp->Size >= 4) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
} else if (IROp->Op == OP_SUBNZCV || IROp->Op == OP_SUBWITHFLAGS || IROp->Op == OP_SUB) {
|
||||
// TODO: Generalize this
|
||||
@@ -612,7 +607,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(0));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -626,7 +621,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(0));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -637,7 +632,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(0));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -649,7 +644,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
if (ARMEmitter::IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -657,7 +652,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
if (Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(0));
|
||||
}
|
||||
}
|
||||
break;
|
||||
@@ -667,7 +662,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant1)) {
|
||||
if (IsImmLogical(Constant1, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant1));
|
||||
}
|
||||
}
|
||||
break;
|
||||
@@ -677,7 +672,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant1)) {
|
||||
if (ARMEmitter::IsImmAddSub(Constant1)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant1));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -689,8 +684,8 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
(Constant2 == 1 || Constant2 == AllOnes) && Constant3 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[2]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, CreateInlineConstant(IREmit, Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 3, CreateInlineConstant(IREmit, Constant3));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, IREmit->_InlineConstant(Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 3, IREmit->_InlineConstant(Constant3));
|
||||
}
|
||||
|
||||
break;
|
||||
@@ -704,11 +699,11 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
uint64_t Constant1 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant1) && Constant1 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant1));
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant0) && (Constant0 == 1 || Constant0 == AllOnes)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[0]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, Constant0));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(Constant0));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -719,7 +714,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
if (ARMEmitter::IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
}
|
||||
break;
|
||||
@@ -730,7 +725,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
uint64_t Constant {};
|
||||
if (IREmit->IsValueConstant(Op->NewRIP, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, Constant));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(Constant));
|
||||
} else {
|
||||
auto NewRIP = IREmit->GetOpHeader(Op->NewRIP);
|
||||
if (NewRIP->Op == OP_ENTRYPOINTOFFSET) {
|
||||
@@ -751,7 +746,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
if (IsImmLogical(Constant2, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[1]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
}
|
||||
}
|
||||
break;
|
||||
@@ -787,7 +782,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
uint64_t Constant {};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, IREmit->_InlineConstant(Constant));
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -797,7 +792,7 @@ void ConstProp::ConstantInlining(IREmitter* IREmit, const IRListView& CurrentIR)
|
||||
uint64_t Constant {};
|
||||
if (IREmit->IsValueConstant(Op->Direction, &Constant)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Direction));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, CreateInlineConstant(IREmit, Constant));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Direction_Index, IREmit->_InlineConstant(Constant));
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -27,7 +27,7 @@ constexpr int PropagationRounds = 5;
|
||||
static inline uint64_t RegBit(RegisterClassType Class, uint32_t Reg) {
|
||||
uint32_t AdjustedReg = (Class == FPRClass) ? (32 + Reg) : Reg;
|
||||
|
||||
return 1UL << AdjustedReg;
|
||||
return 1ULL << AdjustedReg;
|
||||
}
|
||||
|
||||
class DeadStoreElimination final : public FEXCore::IR::Pass {
|
||||
@@ -76,6 +76,16 @@ void DeadStoreElimination::Run(IREmitter* IREmit) {
|
||||
} else if (IROp->Op == OP_LOADREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
BlockInfo.reg.reads |= RegBit(Op->Class, Op->Reg);
|
||||
} else if (IROp->Op == OP_INVALIDATEFLAGS) {
|
||||
auto Op = IROp->C<IR::IROp_InvalidateFlags>();
|
||||
|
||||
if (Op->Flags & (1u << X86State::RFLAG_PF_RAW_LOC)) {
|
||||
BlockInfo.reg.writes |= RegBit(GPRClass, Core::CPUState::PF_AS_GREG);
|
||||
}
|
||||
|
||||
if (Op->Flags & (1u << X86State::RFLAG_AF_RAW_LOC)) {
|
||||
BlockInfo.reg.writes |= RegBit(GPRClass, Core::CPUState::AF_AS_GREG);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,13 +22,11 @@ namespace FEXCore::IR::Validation {
|
||||
struct RegState {
|
||||
static constexpr IR::NodeID UninitializedValue {0};
|
||||
static constexpr IR::NodeID InvalidReg {0xffff'ffff};
|
||||
static constexpr IR::NodeID CorruptedPair {0xffff'fffe};
|
||||
|
||||
// This class makes some assumptions about how the host registers are arranged and mapped to virtual registers:
|
||||
// 1. There will be less than 32 GPRs and 32 FPRs
|
||||
// 2. If the GPRFixed class is used, there will be 16 GPRs and 16 FixedGPRs max
|
||||
// 3. Same with FPRFixed
|
||||
// 4. If the GPRPairClass is used, it is assumed each GPRPair N will map onto GPRs N and N + 1
|
||||
|
||||
// These assumptions were all true for the state of the arm64 and x86 jits at the time this was written
|
||||
|
||||
@@ -49,11 +47,6 @@ struct RegState {
|
||||
// On arm64, there are 16 Fixed and 12 normal
|
||||
FPRsFixed[Reg.Reg] = ssa;
|
||||
return true;
|
||||
case GPRPairClass:
|
||||
// Alias paired registers onto both
|
||||
GPRs[Reg.Reg] = ssa;
|
||||
GPRs[Reg.Reg + 1] = ssa;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -66,12 +59,6 @@ struct RegState {
|
||||
case GPRFixedClass: return GPRsFixed[Reg.Reg];
|
||||
case FPRClass: return FPRs[Reg.Reg];
|
||||
case FPRFixedClass: return FPRsFixed[Reg.Reg];
|
||||
case GPRPairClass:
|
||||
// Make sure both halves of the Pair contain the same SSA
|
||||
if (GPRs[Reg.Reg] == GPRs[Reg.Reg + 1]) {
|
||||
return GPRs[Reg.Reg];
|
||||
}
|
||||
return CorruptedPair;
|
||||
}
|
||||
return InvalidReg;
|
||||
}
|
||||
@@ -139,14 +126,6 @@ void RAValidation::Run(IREmitter* IREmit) {
|
||||
if (CurrentSSAAtReg == RegState::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
} else if (CurrentSSAAtReg == RegState::CorruptedPair) {
|
||||
HadError |= true;
|
||||
|
||||
auto Lower = BlockRegState.Get(PhysicalRegister(GPRClass, uint8_t(PhyReg.Reg * 2) + 1));
|
||||
auto Upper = BlockRegState.Get(PhysicalRegister(GPRClass, PhyReg.Reg * 2 + 1));
|
||||
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects paired reg{} to contain %{}, but it actually contains {{%{}, %{}}}\n", ID, i,
|
||||
PhyReg.Reg, ArgID, Lower, Upper);
|
||||
} else if (CurrentSSAAtReg == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
|
||||
|
||||
@@ -7,6 +7,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
@@ -49,6 +50,11 @@ struct FlagInfo {
|
||||
// all be eliminated.
|
||||
bool CanReplace;
|
||||
IROps Replacement;
|
||||
|
||||
// If true, the opcode can be replaced with ReplacementNoWrite if its register
|
||||
// write is unused but its flags are still needed.
|
||||
bool CanReplaceWrite;
|
||||
IROps ReplacementNoWrite;
|
||||
};
|
||||
|
||||
class DeadFlagCalculationEliminination final : public FEXCore::IR::Pass {
|
||||
@@ -115,6 +121,8 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_ADD,
|
||||
.CanReplaceWrite = true,
|
||||
.ReplacementNoWrite = OP_ADDNZCV,
|
||||
};
|
||||
|
||||
case OP_SUBWITHFLAGS:
|
||||
@@ -122,6 +130,8 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_SUB,
|
||||
.CanReplaceWrite = true,
|
||||
.ReplacementNoWrite = OP_SUBNZCV,
|
||||
};
|
||||
|
||||
case OP_ADCWITHFLAGS:
|
||||
@@ -130,6 +140,16 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_ADC,
|
||||
.CanReplaceWrite = true,
|
||||
.ReplacementNoWrite = OP_ADCNZCV,
|
||||
};
|
||||
|
||||
case OP_ADCZEROWITHFLAGS:
|
||||
return {
|
||||
.Read = FLAG_C,
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_ADCZERO,
|
||||
};
|
||||
|
||||
case OP_SBBWITHFLAGS:
|
||||
@@ -138,6 +158,8 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
.Write = FLAG_NZCV,
|
||||
.CanReplace = true,
|
||||
.Replacement = OP_SBB,
|
||||
.CanReplaceWrite = true,
|
||||
.ReplacementNoWrite = OP_SBBNZCV,
|
||||
};
|
||||
|
||||
case OP_SHIFTFLAGS:
|
||||
@@ -151,6 +173,16 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
.CanEliminate = true,
|
||||
};
|
||||
|
||||
case OP_ROTATEFLAGS:
|
||||
// _RotateFlags conditionally sets CV, again modeled as RMW.
|
||||
return {
|
||||
.Read = FLAG_C | FLAG_V,
|
||||
.Write = FLAG_C | FLAG_V,
|
||||
.CanEliminate = true,
|
||||
};
|
||||
|
||||
case OP_RDRAND: return {.Write = FLAG_NZCV};
|
||||
|
||||
case OP_ADDNZCV:
|
||||
case OP_SUBNZCV:
|
||||
case OP_TESTNZ:
|
||||
@@ -201,7 +233,8 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
.CanEliminate = true,
|
||||
};
|
||||
|
||||
case OP_NZCVSELECT: {
|
||||
case OP_NZCVSELECT:
|
||||
case OP_NZCVSELECTINCREMENT: {
|
||||
auto Op = IROp->CW<IR::IROp_NZCVSelect>();
|
||||
return {.Read = FlagsForCondClassType(Op->Cond)};
|
||||
}
|
||||
@@ -407,6 +440,10 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
}
|
||||
} else {
|
||||
FlagsRead &= ~Info.Write;
|
||||
|
||||
if (Info.CanReplaceWrite && CodeNode->GetUses() == 0) {
|
||||
IROp->Op = Info.ReplacementNoWrite;
|
||||
}
|
||||
}
|
||||
|
||||
// If we eliminated the instruction, we eliminate its read too. This
|
||||
|
||||
@@ -185,11 +185,11 @@ private:
|
||||
};
|
||||
|
||||
RegisterClass* GetClass(PhysicalRegister Reg) {
|
||||
return &Classes[(Reg.Class == GPRPairClass) ? GPRClass : Reg.Class];
|
||||
return &Classes[Reg.Class];
|
||||
};
|
||||
|
||||
uint32_t GetRegBits(PhysicalRegister Reg) {
|
||||
return ((Reg.Class == GPRPairClass) ? 0b11 : 0b1) << Reg.Reg;
|
||||
return 1 << Reg.Reg;
|
||||
};
|
||||
|
||||
bool IsInRegisterFile(Ref Old) {
|
||||
@@ -273,7 +273,7 @@ private:
|
||||
// the next set bit and then clearing on each iteration.
|
||||
#define foreach_bit(b, x) for (uint32_t __x = (x), b; ((b) = __builtin_ffs(__x) - 1, __x); __x &= ~(1 << (b)))
|
||||
|
||||
void SpillReg(RegisterClass* Class, IROp_Header* Exclude, bool Pair) {
|
||||
void SpillReg(RegisterClass* Class, IROp_Header* Exclude) {
|
||||
// Find the best node to spill according to the "furthest-first" heuristic.
|
||||
// Since we defined IPs relative to the end of the block, the furthest
|
||||
// next-use has the /smallest/ unsigned IP.
|
||||
@@ -282,12 +282,6 @@ private:
|
||||
uint8_t BestReg = ~0;
|
||||
|
||||
foreach_bit(i, Class->Allocated) {
|
||||
// We have to prioritize the pair region if we're allocating for a Pair.
|
||||
// See the comment at the call site in AssignReg.
|
||||
if (Pair && Candidate != nullptr && i >= PairRegs) {
|
||||
break;
|
||||
}
|
||||
|
||||
Ref Old = Class->RegToSSA[i];
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Old != nullptr, "Invariant3");
|
||||
@@ -366,22 +360,6 @@ private:
|
||||
SSAToReg[Index] = Reg;
|
||||
};
|
||||
|
||||
// Get the mask of available registers for a given register class
|
||||
uint32_t AvailableMask(RegisterClass* Class, bool Pair) {
|
||||
uint32_t Available = Class->Available;
|
||||
|
||||
if (Pair) {
|
||||
// Only choose base register R if R and R + 1 are both free
|
||||
Available &= (Available >> 1);
|
||||
|
||||
// Only consider aligned registers in the pair region
|
||||
constexpr uint32_t EVEN_BITS = 0x55555555;
|
||||
Available &= (EVEN_BITS & ((1u << PairRegs) - 1));
|
||||
}
|
||||
|
||||
return Available;
|
||||
};
|
||||
|
||||
// Assign a register for a given Node, spilling if necessary.
|
||||
void AssignReg(IROp_Header* IROp, Ref CodeNode, IROp_Header* Pivot) {
|
||||
const uint32_t Node = IR->GetID(CodeNode).Value;
|
||||
@@ -411,97 +389,47 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
RegisterClassType OrigClassType = GetRegClassFromNode(IR, IROp);
|
||||
bool Pair = OrigClassType == GPRPairClass;
|
||||
RegisterClassType ClassType = Pair ? GPRClass : OrigClassType;
|
||||
// Try to coalesce reserved pairs. Just a heuristic to remove some moves.
|
||||
if (IROp->Op == OP_ALLOCATEGPR) {
|
||||
if (IROp->C<IROp_AllocateGPR>()->ForPair) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
|
||||
// Only choose base register R if R and R + 1 are both free
|
||||
Available &= (Available >> 1);
|
||||
|
||||
// Only consider aligned registers in the pair region
|
||||
constexpr uint32_t EVEN_BITS = 0x55555555;
|
||||
Available &= (EVEN_BITS & ((1u << PairRegs) - 1));
|
||||
|
||||
if (Available) {
|
||||
unsigned Reg = std::countr_zero(Available);
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, Reg));
|
||||
return;
|
||||
}
|
||||
}
|
||||
} else if (IROp->Op == OP_ALLOCATEGPRAFTER) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
auto After = SSAToReg[IR->GetID(IR->GetNode(IROp->Args[0])).Value];
|
||||
if ((After.Reg & 1) == 0 && Available & (1ull << (After.Reg + 1))) {
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, After.Reg + 1));
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
RegisterClassType ClassType = GetRegClassFromNode(IR, IROp);
|
||||
RegisterClass* Class = &Classes[ClassType];
|
||||
|
||||
// Spill to make room in the register file. Free registers need not be
|
||||
// contiguous, we'll shuffle later.
|
||||
//
|
||||
// There is one subtlety: when allocating a pair, we need at least 1 free
|
||||
// register in the pair region. Else, we could end up trying to allocate a
|
||||
// pair when the only free 2 regs are outside the pair region, and the pair
|
||||
// region is made of all pairs (so nothing to shuffle). With 1 free
|
||||
// register in the pair region, we'll be able to shuffle.
|
||||
//
|
||||
// When spilling for pairs, SpillReg prioritizes spilling the pair region
|
||||
// which ensures this loop is well-behaved.
|
||||
while (std::popcount(Class->Available) < (Pair ? 2 : 1) || (Pair && !(Class->Available & ((1u << PairRegs) - 1)))) {
|
||||
if (!Class->Available) {
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
SpillReg(Class, Pivot, Pair);
|
||||
SpillReg(Class, Pivot);
|
||||
}
|
||||
|
||||
// There are now enough free registers, but they may be fragmented.
|
||||
// Pick a scalar blocking a pair and shuffle to make room.
|
||||
uint32_t Available = AvailableMask(Class, Pair);
|
||||
if (!Available) {
|
||||
LOGMAN_THROW_A_FMT(OrigClassType == GPRPairClass, "Already spilled");
|
||||
|
||||
// Find the first free scalar. There are at least 2.
|
||||
unsigned Hole = std::countr_zero(Class->Available);
|
||||
LOGMAN_THROW_AA_FMT(Class->Available & (1u << Hole), "Definition");
|
||||
|
||||
// Its neighbour is blocking the pair.
|
||||
unsigned Blocked = Hole ^ 1;
|
||||
LOGMAN_THROW_AA_FMT(!(Class->Available & (1u << Blocked)), "Invariant7");
|
||||
LOGMAN_THROW_AA_FMT(Hole < PairRegs, "Pairable register");
|
||||
|
||||
// Find another free scalar to evict the neighbour
|
||||
unsigned NewReg = std::countr_zero(Class->Available & ~(1u << Hole));
|
||||
LOGMAN_THROW_AA_FMT(Class->Available & (1u << NewReg), "Ensured space");
|
||||
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
Ref Old = Class->RegToSSA[Blocked];
|
||||
LOGMAN_THROW_A_FMT(GetRegClassFromNode(IR, IR->GetOp<IROp_Header>(Old)) == GPRClass, "Only scalars have free neighbours");
|
||||
FreeReg(PhysicalRegister(GPRClass, Blocked));
|
||||
|
||||
Ref Clobber = nullptr;
|
||||
|
||||
// If that scalar is free because it is killed by this instruction, it
|
||||
// needs to be shuffled too, since the copy would clobber it.
|
||||
for (auto s = 0; s < IR::GetRAArgs(Pivot->Op); ++s) {
|
||||
// It is possible that the argument is to be remapped, but the actual
|
||||
// remapping in the IR only happens later in the pass so we need to
|
||||
// Map() explicitly. This can be hit with SRA shuffles.
|
||||
Ref New = Map(IR->GetNode(Pivot->Args[s]));
|
||||
const PhysicalRegister ClobberReg = SSAToReg[IR->GetID(New).Value];
|
||||
|
||||
if (ClobberReg.Class == GPRClass && ClobberReg.Reg == NewReg) {
|
||||
Clobber = IR->GetNode(Pivot->Args[s]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Clobber) {
|
||||
// Swap the registers.
|
||||
LOGMAN_THROW_A_FMT(IsOld(Clobber), "Not yet mapped");
|
||||
|
||||
auto ClobberNew = IREmit->_Swap1(Map(Clobber), Map(Old));
|
||||
Remap(Clobber, ClobberNew);
|
||||
|
||||
auto New = IREmit->_Swap2();
|
||||
Remap(Old, New);
|
||||
|
||||
SetReg(New, PhysicalRegister(GPRClass, NewReg));
|
||||
SetReg(ClobberNew, PhysicalRegister(GPRClass, Blocked));
|
||||
FreeReg(PhysicalRegister(GPRClass, Blocked));
|
||||
} else {
|
||||
// Otherwise, simply copy.
|
||||
auto Copy = IREmit->_Copy(Map(Old));
|
||||
|
||||
Remap(Old, Copy);
|
||||
SetReg(Copy, PhysicalRegister(GPRClass, NewReg));
|
||||
}
|
||||
|
||||
Available = AvailableMask(Class, Pair);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Available != 0, "Post-condition of spill and shuffle");
|
||||
|
||||
// Assign a free register in the appropriate class.
|
||||
unsigned Reg = std::countr_zero(Available);
|
||||
SetReg(CodeNode, PhysicalRegister(OrigClassType, Reg));
|
||||
LOGMAN_THROW_AA_FMT(Class->Available != 0, "Post-condition of spilling");
|
||||
unsigned Reg = std::countr_zero(Class->Available);
|
||||
SetReg(CodeNode, PhysicalRegister(ClassType, Reg));
|
||||
};
|
||||
|
||||
bool IsRAOp(IROps Op) {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
@@ -10,10 +11,114 @@
|
||||
#include <FEXCore/Utils/ArchHelpers/Arm64.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <csignal>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t CASPAL_MASK = 0xBF'E0'FC'00;
|
||||
constexpr uint32_t CASPAL_INST = 0x08'60'FC'00;
|
||||
|
||||
constexpr uint32_t CASAL_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t CASAL_INST = 0x08'E0'FC'00;
|
||||
|
||||
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
|
||||
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
|
||||
|
||||
constexpr uint32_t RCPC2_MASK = 0x3F'E0'0C'00;
|
||||
constexpr uint32_t LDAPUR_INST = 0x19'40'00'00;
|
||||
constexpr uint32_t STLUR_INST = 0x19'00'00'00;
|
||||
|
||||
constexpr uint32_t LDAXP_MASK = 0xBF'FF'80'00;
|
||||
constexpr uint32_t LDAXP_INST = 0x88'7F'80'00;
|
||||
|
||||
constexpr uint32_t STLXP_MASK = 0xBF'E0'80'00;
|
||||
constexpr uint32_t STLXP_INST = 0x88'20'80'00;
|
||||
|
||||
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
|
||||
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
|
||||
constexpr uint32_t LDAR_INST = 0x08'DF'FC'00;
|
||||
constexpr uint32_t LDAPR_INST = 0x38'BF'C0'00;
|
||||
constexpr uint32_t STLR_INST = 0x08'9F'FC'00;
|
||||
|
||||
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
|
||||
|
||||
constexpr uint32_t LDSTREGISTER_MASK = 0b0011'1011'0010'0000'0000'1100'0000'0000;
|
||||
constexpr uint32_t LDR_INST = 0b0011'1000'0111'1111'0110'1000'0000'0000;
|
||||
constexpr uint32_t STR_INST = 0b0011'1000'0011'1111'0110'1000'0000'0000;
|
||||
|
||||
constexpr uint32_t LDSTUNSCALED_MASK = 0b0011'1011'0010'0000'0000'1100'0000'0000;
|
||||
constexpr uint32_t LDUR_INST = 0b0011'1000'0100'0000'0000'0000'0000'0000;
|
||||
constexpr uint32_t STUR_INST = 0b0011'1000'0000'0000'0000'0000'0000'0000;
|
||||
|
||||
constexpr uint32_t LDSTP_MASK = 0b0011'1011'1000'0000'0000'0000'0000'0000;
|
||||
constexpr uint32_t STP_INST = 0b0010'1001'0000'0000'0000'0000'0000'0000;
|
||||
|
||||
constexpr uint32_t CBNZ_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t CBNZ_INST = 0x35'00'00'00;
|
||||
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'20'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t ADD_SHIFT_INST = 0x0B'20'00'00;
|
||||
constexpr uint32_t SUB_SHIFT_INST = 0x4B'20'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t CMP_SHIFT_INST = 0x6B'20'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t BIC_INST = 0x0A'20'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t ORN_INST = 0x2A'20'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
constexpr uint32_t EON_INST = 0x4A'20'00'00;
|
||||
|
||||
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
|
||||
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
|
||||
|
||||
constexpr uint32_t CLREX_MASK = 0xFF'FF'F0'FF;
|
||||
constexpr uint32_t CLREX_INST = 0xD5'03'30'5F;
|
||||
|
||||
enum ExclusiveAtomicPairType {
|
||||
TYPE_SWAP,
|
||||
TYPE_ADD,
|
||||
TYPE_SUB,
|
||||
TYPE_AND,
|
||||
TYPE_BIC,
|
||||
TYPE_OR,
|
||||
TYPE_ORN,
|
||||
TYPE_EOR,
|
||||
TYPE_EON,
|
||||
TYPE_NEG, // This is just a sub with zero. Need to know the differences
|
||||
};
|
||||
|
||||
// Load ops are 4 bits
|
||||
// Acquire and release bits are independent on the instruction
|
||||
constexpr uint32_t ATOMIC_ADD_OP = 0b0000;
|
||||
constexpr uint32_t ATOMIC_CLR_OP = 0b0001;
|
||||
constexpr uint32_t ATOMIC_EOR_OP = 0b0010;
|
||||
constexpr uint32_t ATOMIC_SET_OP = 0b0011;
|
||||
constexpr uint32_t ATOMIC_SWAP_OP = 0b1000;
|
||||
|
||||
constexpr uint32_t REGISTER_MASK = 0b11111;
|
||||
constexpr uint32_t RD_OFFSET = 0;
|
||||
constexpr uint32_t RN_OFFSET = 5;
|
||||
constexpr uint32_t RM_OFFSET = 16;
|
||||
|
||||
constexpr uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 | 0b1011'0000'0000; // Inner shareable all
|
||||
|
||||
constexpr uint32_t DMB_LD = 0b1101'0101'0000'0011'0011'0000'1011'1111 | 0b1101'0000'0000; // Inner shareable load
|
||||
|
||||
inline uint32_t GetRdReg(uint32_t Instr) {
|
||||
return (Instr >> RD_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
inline uint32_t GetRnReg(uint32_t Instr) {
|
||||
return (Instr >> RN_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
inline uint32_t GetRmReg(uint32_t Instr) {
|
||||
return (Instr >> RM_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock, TYPE_HAS_SPLIT_LOCKS);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock16B, TYPE_16BYTE_SPLIT);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas16Tear, TYPE_CAS_16BIT_TEAR);
|
||||
@@ -239,7 +344,9 @@ std::pair<uint64_t, uint64_t> DoLoad128(uint64_t Addr) {
|
||||
}
|
||||
|
||||
static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint32_t DesiredReg2, uint32_t ExpectedReg1,
|
||||
uint32_t ExpectedReg2, uint32_t AddressReg) {
|
||||
uint32_t ExpectedReg2, uint32_t AddressReg, uint32_t* StrictSplitLockMutex) {
|
||||
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
if (Size == 0) {
|
||||
// 32bit
|
||||
uint64_t Addr = GPRs[AddressReg];
|
||||
@@ -260,11 +367,17 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
// Check for Split lock across a cacheline
|
||||
if ((Addr & 63) > 56) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
|
||||
uint64_t Alignment = Addr & 0b111;
|
||||
Addr &= ~0b111ULL;
|
||||
@@ -415,7 +528,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
return false;
|
||||
}
|
||||
|
||||
bool HandleCASPAL(uint32_t Instr, uint64_t* GPRs) {
|
||||
bool HandleCASPAL(uint32_t Instr, uint64_t* GPRs, uint32_t* StrictSplitLockMutex) {
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
|
||||
uint32_t DesiredReg1 = Instr & 0b11111;
|
||||
@@ -424,10 +537,10 @@ bool HandleCASPAL(uint32_t Instr, uint64_t* GPRs) {
|
||||
uint32_t ExpectedReg2 = ExpectedReg1 + 1;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
return RunCASPAL(GPRs, Size, DesiredReg1, DesiredReg2, ExpectedReg1, ExpectedReg2, AddressReg);
|
||||
return RunCASPAL(GPRs, Size, DesiredReg1, DesiredReg2, ExpectedReg1, ExpectedReg2, AddressReg, StrictSplitLockMutex);
|
||||
}
|
||||
|
||||
uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uint64_t* GPRs) {
|
||||
uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uint64_t* GPRs, uint32_t* StrictSplitLockMutex) {
|
||||
// caspair
|
||||
// [1] ldaxp(TMP2.W(), TMP3.W(), MemOperand(MemSrc)); <-- DataReg & AddrReg
|
||||
// [2] cmp(TMP2.W(), Expected.first.W()); <-- ExpectedReg1
|
||||
@@ -505,7 +618,7 @@ uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uint64_t*
|
||||
GPRs[DataReg] = GPRs[ExpectedReg1];
|
||||
GPRs[DataReg2] = GPRs[ExpectedReg2];
|
||||
|
||||
if (RunCASPAL(GPRs, Size, DesiredReg1, DesiredReg2, DataReg, DataReg2, AddrReg)) {
|
||||
if (RunCASPAL(GPRs, Size, DesiredReg1, DesiredReg2, DataReg, DataReg2, AddrReg, StrictSplitLockMutex)) {
|
||||
return 9 * sizeof(uint32_t); // skip to mov + clrex
|
||||
} else {
|
||||
return 0;
|
||||
@@ -534,7 +647,7 @@ static bool HandleAtomicVectorStore(uint32_t Instr, uintptr_t ProgramCounter) {
|
||||
PC[1] = STP;
|
||||
PC[2] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ClearICache(&PC[0], 16);
|
||||
ClearICache(&PC[0], 12);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -549,16 +662,23 @@ using CASDesiredFn = T (*)(T Src, T Desired);
|
||||
|
||||
template<bool Retry>
|
||||
static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr, CASExpectedFn<uint16_t> ExpectedFunction,
|
||||
CASDesiredFn<uint16_t> DesiredFunction) {
|
||||
CASDesiredFn<uint16_t> DesiredFunction, uint32_t* StrictSplitLockMutex) {
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
|
||||
if ((Addr & 63) == 63) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
}
|
||||
|
||||
// 16 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) == 15) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
|
||||
// Address crosses over 16byte or 64byte threshold
|
||||
// Need a dual 8bit CAS loop
|
||||
@@ -818,16 +938,23 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
|
||||
template<bool Retry>
|
||||
static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr, CASExpectedFn<uint32_t> ExpectedFunction,
|
||||
CASDesiredFn<uint32_t> DesiredFunction) {
|
||||
CASDesiredFn<uint32_t> DesiredFunction, uint32_t* StrictSplitLockMutex) {
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
|
||||
if ((Addr & 63) > 60) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
}
|
||||
|
||||
// 32 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 12) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
|
||||
// Address crosses over 16byte threshold
|
||||
// Needs dual 4 byte CAS loop
|
||||
@@ -1043,16 +1170,23 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
|
||||
template<bool Retry>
|
||||
static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr, CASExpectedFn<uint64_t> ExpectedFunction,
|
||||
CASDesiredFn<uint64_t> DesiredFunction) {
|
||||
CASDesiredFn<uint64_t> DesiredFunction, uint32_t* StrictSplitLockMutex) {
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
|
||||
if ((Addr & 63) > 56) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
}
|
||||
|
||||
// 64bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
|
||||
uint64_t Alignment = Addr & 0b111;
|
||||
Addr &= ~0b111ULL;
|
||||
@@ -1201,7 +1335,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
}
|
||||
}
|
||||
|
||||
static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg) {
|
||||
static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg, uint32_t* StrictSplitLockMutex) {
|
||||
uint64_t Addr = GPRs[AddressReg];
|
||||
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
@@ -1224,7 +1358,8 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
[](uint16_t, uint16_t Desired) -> uint16_t {
|
||||
// Desired is just Desired
|
||||
return Desired;
|
||||
});
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
@@ -1242,7 +1377,8 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
[](uint32_t, uint32_t Desired) -> uint32_t {
|
||||
// Desired is just Desired
|
||||
return Desired;
|
||||
});
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
@@ -1260,7 +1396,8 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
[](uint64_t, uint64_t Desired) -> uint64_t {
|
||||
// Desired is just Desired
|
||||
return Desired;
|
||||
});
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
@@ -1273,16 +1410,16 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool HandleCASAL(uint64_t* GPRs, uint32_t Instr) {
|
||||
static bool HandleCASAL(uint64_t* GPRs, uint32_t Instr, uint32_t* StrictSplitLockMutex) {
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t DesiredReg = Instr & 0b11111;
|
||||
uint32_t ExpectedReg = (Instr >> 16) & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
return RunCASAL(GPRs, Size, DesiredReg, ExpectedReg, AddressReg);
|
||||
return RunCASAL(GPRs, Size, DesiredReg, ExpectedReg, AddressReg, StrictSplitLockMutex);
|
||||
}
|
||||
|
||||
static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs) {
|
||||
static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs, uint32_t* StrictSplitLockMutex) {
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
uint32_t ResultReg = Instr & 0b11111;
|
||||
uint32_t SourceReg = (Instr >> 16) & 0b11111;
|
||||
@@ -1330,7 +1467,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs) {
|
||||
|
||||
auto Res = DoCAS16<true>(GPRs[SourceReg],
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction);
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (ResultReg != 31) {
|
||||
@@ -1375,7 +1512,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs) {
|
||||
|
||||
auto Res = DoCAS32<true>(GPRs[SourceReg],
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction);
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (ResultReg != 31) {
|
||||
@@ -1420,7 +1557,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs) {
|
||||
|
||||
auto Res = DoCAS64<true>(GPRs[SourceReg],
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction);
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (ResultReg != 31) {
|
||||
@@ -1466,7 +1603,7 @@ static bool HandleAtomicLoad(uint32_t Instr, uint64_t* GPRs, int64_t Offset) {
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset) {
|
||||
static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset, uint32_t* StrictSplitLockMutex) {
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
@@ -1487,7 +1624,8 @@ static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset) {
|
||||
[](uint16_t, uint16_t Desired) -> uint16_t {
|
||||
// Desired is just Desired
|
||||
return Desired;
|
||||
});
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
return true;
|
||||
} else if (Size == 4) {
|
||||
DoCAS32<DoRetry>(
|
||||
@@ -1501,7 +1639,8 @@ static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset) {
|
||||
[](uint32_t, uint32_t Desired) -> uint32_t {
|
||||
// Desired is just Desired
|
||||
return Desired;
|
||||
});
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
return true;
|
||||
} else if (Size == 8) {
|
||||
DoCAS64<DoRetry>(
|
||||
@@ -1515,14 +1654,15 @@ static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset) {
|
||||
[](uint64_t, uint64_t Desired) -> uint64_t {
|
||||
// Desired is just Desired
|
||||
return Desired;
|
||||
});
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
static uint64_t HandleCAS_NoAtomics(uintptr_t ProgramCounter, uint64_t* GPRs) {
|
||||
static uint64_t HandleCAS_NoAtomics(uintptr_t ProgramCounter, uint64_t* GPRs, uint32_t* StrictSplitLockMutex) {
|
||||
// ARMv8.0 CAS
|
||||
// [1] ldaxrb(TMP2.W(), MemOperand(MemSrc))
|
||||
// [2] cmp (TMP2.W(), Expected.W())
|
||||
@@ -1558,14 +1698,14 @@ static uint64_t HandleCAS_NoAtomics(uintptr_t ProgramCounter, uint64_t* GPRs) {
|
||||
// set up CASAL by doing mov(TMP2, Expected)
|
||||
GPRs[ResultReg] = GPRs[ExpectedReg];
|
||||
|
||||
if (RunCASAL(GPRs, Size, DesiredReg, ResultReg, AddressReg)) {
|
||||
if (RunCASAL(GPRs, Size, DesiredReg, ResultReg, AddressReg, StrictSplitLockMutex)) {
|
||||
return 7 * sizeof(uint32_t); // jump to mov to allocated register
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_t* GPRs) {
|
||||
static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_t* GPRs, uint32_t* StrictSplitLockMutex) {
|
||||
uint32_t* PC = (uint32_t*)ProgramCounter;
|
||||
uint32_t Instr = PC[0];
|
||||
|
||||
@@ -1642,7 +1782,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
} else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_INST ||
|
||||
(NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_SHIFT_INST) {
|
||||
return HandleCAS_NoAtomics(ProgramCounter, GPRs); // ARMv8.0 CAS
|
||||
return HandleCAS_NoAtomics(ProgramCounter, GPRs, StrictSplitLockMutex); // ARMv8.0 CAS
|
||||
} else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::AND_INST) {
|
||||
AtomicOp = ExclusiveAtomicPairType::TYPE_AND;
|
||||
DataSourceReg = GetRmReg(NextInstr);
|
||||
@@ -1754,7 +1894,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
|
||||
auto Res = DoCAS16<DoRetry>(GPRs[DataSourceReg],
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction);
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
|
||||
if (AtomicFetch && ResultReg != 31) {
|
||||
// On atomic fetch then we store the resulting value back in to the loadacquire destination register
|
||||
@@ -1781,7 +1921,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
|
||||
auto Res = DoCAS32<DoRetry>(GPRs[DataSourceReg],
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction);
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
|
||||
if (AtomicFetch && ResultReg != 31) {
|
||||
// On atomic fetch then we store the resulting value back in to the loadacquire destination register
|
||||
@@ -1808,7 +1948,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
|
||||
auto Res = DoCAS64<DoRetry>(GPRs[DataSourceReg],
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction);
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
if (AtomicFetch && ResultReg != 31) {
|
||||
// On atomic fetch then we store the resulting value back in to the loadacquire destination register
|
||||
// We want the memory value BEFORE the ALU op
|
||||
@@ -1830,178 +1970,235 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
#endif
|
||||
|
||||
constexpr auto NotHandled = std::make_pair(false, 0);
|
||||
if constexpr (!is_arm64) {
|
||||
return NotHandled;
|
||||
}
|
||||
|
||||
if constexpr (is_arm64) {
|
||||
uint32_t* PC = (uint32_t*)ProgramCounter;
|
||||
uint32_t Instr = PC[0];
|
||||
uint32_t* PC = (uint32_t*)ProgramCounter;
|
||||
uint32_t Instr = PC[0];
|
||||
|
||||
// 1 = 16bit
|
||||
// 2 = 32bit
|
||||
// 3 = 64bit
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
// 1 = 16bit
|
||||
// 2 = 32bit
|
||||
// 3 = 64bit
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
|
||||
// ParanoidTSO path doesn't modify any code.
|
||||
if (HandleType == UnalignedHandlerType::Paranoid) [[unlikely]] {
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & RCPC2_MASK) == LDAPUR_INST) { // LDAPUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, Offset)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAPUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, Offset)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader*>(BlockBegin);
|
||||
auto InlineTail = reinterpret_cast<CPU::CPUBackend::JITCodeTail*>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
|
||||
// Lock code mutex during any SIGBUS handling that potentially changes code.
|
||||
// Need to be careful to not read any code part-way through modification.
|
||||
FEXCore::Utils::SpinWaitLock::UniqueSpinMutex lk(&InlineTail->SpinLockFutex);
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
uint32_t* StrictSplitLockMutex {CTX->Config.StrictInProcessSplitLocks ? &CTX->StrictSplitLockMutex : nullptr};
|
||||
|
||||
// ParanoidTSO path doesn't modify any code.
|
||||
if (HandleType == UnalignedHandlerType::Paranoid) [[unlikely]] {
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
uint32_t LDR = 0b0011'1000'0111'1111'0110'1000'0000'0000;
|
||||
LDR |= Size << 30;
|
||||
LDR |= AddrReg << 5;
|
||||
LDR |= DataReg;
|
||||
PC[0] = LDR;
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
PC[1] = DMB_LD; // Back-patch the half-barrier.
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
ClearICache(&PC[0], 16);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
} else if ((Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
uint32_t STR = 0b0011'1000'0011'1111'0110'1000'0000'0000;
|
||||
STR |= Size << 30;
|
||||
STR |= AddrReg << 5;
|
||||
STR |= DataReg;
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
PC[-1] = DMB; // Back-patch the half-barrier.
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, 0, StrictSplitLockMutex)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
PC[0] = STR;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
} else if ((Instr & RCPC2_MASK) == LDAPUR_INST) { // LDAPUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
uint32_t LDUR = 0b0011'1000'0100'0000'0000'0000'0000'0000;
|
||||
LDUR |= Size << 30;
|
||||
LDUR |= AddrReg << 5;
|
||||
LDUR |= DataReg;
|
||||
LDUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[0] = LDUR;
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
PC[1] = DMB_LD; // Back-patch the half-barrier.
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, Offset)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAPUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
ClearICache(&PC[0], 16);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
} else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
uint32_t STUR = 0b0011'1000'0000'0000'0000'0000'0000'0000;
|
||||
STUR |= Size << 30;
|
||||
STUR |= AddrReg << 5;
|
||||
STUR |= DataReg;
|
||||
STUR |= Instr & (0b1'1111'1111 << 9);
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
PC[-1] = DMB; // Back-patch the half-barrier.
|
||||
}
|
||||
PC[0] = STUR;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
} else if ((Instr & ArchHelpers::Arm64::LDAXP_MASK) == ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
// Should be compare and swap pair only. LDAXP not used elsewhere
|
||||
uint64_t BytesToSkip = ArchHelpers::Arm64::HandleCASPAL_ARMv8(Instr, ProgramCounter, GPRs);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, BytesToSkip);
|
||||
} else {
|
||||
if (ArchHelpers::Arm64::HandleAtomicVectorStore(Instr, ProgramCounter)) {
|
||||
return std::make_pair(true, 0);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::STLXP_MASK) == ArchHelpers::Arm64::STLXP_INST) { // STLXP
|
||||
// Should not trigger - middle of an LDAXP/STAXP pair.
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
} else if ((Instr & ArchHelpers::Arm64::CASPAL_MASK) == ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (ArchHelpers::Arm64::HandleCASPAL(Instr, GPRs)) {
|
||||
// Extract the 9-bit offset from the instruction
|
||||
int32_t Offset = static_cast<int32_t>(Instr) << 11 >> 23;
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, Offset, StrictSplitLockMutex)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::CASAL_MASK) == ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (ArchHelpers::Arm64::HandleCASAL(GPRs, Instr)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::ATOMIC_MEM_MASK) == ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (ArchHelpers::Arm64::HandleAtomicMemOp(Instr, GPRs)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}: PC: 0x{:x} Instruction: 0x{:08x}\n", Op, ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::LDAXR_MASK) == ArchHelpers::Arm64::LDAXR_INST) { // LDAXR*
|
||||
uint64_t BytesToSkip = ArchHelpers::Arm64::HandleAtomicLoadstoreExclusive(ProgramCounter, GPRs);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, BytesToSkip);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXR: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
}
|
||||
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader*>(BlockBegin);
|
||||
auto InlineTail = reinterpret_cast<CPU::CPUBackend::JITCodeTail*>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
|
||||
// Check some instructions first that don't do any backpatching.
|
||||
if ((Instr & ArchHelpers::Arm64::CASPAL_MASK) == ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (ArchHelpers::Arm64::HandleCASPAL(Instr, GPRs, StrictSplitLockMutex)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::CASAL_MASK) == ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (ArchHelpers::Arm64::HandleCASAL(GPRs, Instr, StrictSplitLockMutex)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST || // LDAPR*
|
||||
(Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
// This must fall through to the spin-lock implementation below.
|
||||
// This mask has a partial overlap with ATOMIC_MEM_INST so we need to check this here.
|
||||
} else if ((Instr & ArchHelpers::Arm64::ATOMIC_MEM_MASK) == ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op
|
||||
if (ArchHelpers::Arm64::HandleAtomicMemOp(Instr, GPRs, StrictSplitLockMutex)) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, 4);
|
||||
} else {
|
||||
uint8_t Op = (PC[0] >> 12) & 0xF;
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}: PC: 0x{:x} Instruction: 0x{:08x}\n", Op, ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::LDAXR_MASK) == ArchHelpers::Arm64::LDAXR_INST) { // LDAXR*
|
||||
uint64_t BytesToSkip = ArchHelpers::Arm64::HandleAtomicLoadstoreExclusive(ProgramCounter, GPRs, StrictSplitLockMutex);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, BytesToSkip);
|
||||
}
|
||||
// Explicit fallthrough to the backpatch handler below!
|
||||
} else if ((Instr & ArchHelpers::Arm64::LDAXP_MASK) == ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
// Should be compare and swap pair only. LDAXP not used elsewhere
|
||||
uint64_t BytesToSkip = ArchHelpers::Arm64::HandleCASPAL_ARMv8(Instr, ProgramCounter, GPRs, StrictSplitLockMutex);
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
return std::make_pair(true, BytesToSkip);
|
||||
}
|
||||
}
|
||||
|
||||
// Lock code mutex during any SIGBUS handling that potentially changes code.
|
||||
// Due to code buffer sharing between threads, code must be carefully backpatched from last to first.
|
||||
// Multiple threads can be attempting to handle the SIGBUS or even be executing the code being backpatched.
|
||||
FEXCore::Utils::SpinWaitLock::UniqueSpinMutex lk(&InlineTail->SpinLockFutex);
|
||||
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
uint32_t LDR = LDR_INST;
|
||||
LDR |= Size << 30;
|
||||
LDR |= AddrReg << 5;
|
||||
LDR |= DataReg;
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
// Ordering matters with cross-thread visibility!
|
||||
std::atomic_ref<uint32_t>(PC[1]).store(DMB_LD, std::memory_order_release); // Back-patch the half-barrier.
|
||||
}
|
||||
std::atomic_ref<uint32_t>(PC[0]).store(LDR, std::memory_order_release);
|
||||
ClearICache(&PC[0], 8);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
} else if ((Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
uint32_t STR = STR_INST;
|
||||
STR |= Size << 30;
|
||||
STR |= AddrReg << 5;
|
||||
STR |= DataReg;
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
std::atomic_ref<uint32_t>(PC[-1]).store(DMB, std::memory_order_release); // Back-patch the half-barrier.
|
||||
}
|
||||
std::atomic_ref<uint32_t>(PC[0]).store(STR, std::memory_order_release);
|
||||
ClearICache(&PC[-1], 8);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
} else if ((Instr & RCPC2_MASK) == LDAPUR_INST) { // LDAPUR*
|
||||
// Extract the 9-bit offset from the instruction
|
||||
uint32_t LDUR = LDUR_INST;
|
||||
LDUR |= Size << 30;
|
||||
LDUR |= AddrReg << 5;
|
||||
LDUR |= DataReg;
|
||||
LDUR |= Instr & (0b1'1111'1111 << 9);
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
// Ordering matters with cross-thread visibility!
|
||||
std::atomic_ref<uint32_t>(PC[1]).store(DMB_LD, std::memory_order_release); // Back-patch the half-barrier.
|
||||
}
|
||||
std::atomic_ref<uint32_t>(PC[0]).store(LDUR, std::memory_order_release);
|
||||
ClearICache(&PC[0], 8);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
} else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
uint32_t STUR = STUR_INST;
|
||||
STUR |= Size << 30;
|
||||
STUR |= AddrReg << 5;
|
||||
STUR |= DataReg;
|
||||
STUR |= Instr & (0b1'1111'1111 << 9);
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
std::atomic_ref<uint32_t>(PC[-1]).store(DMB, std::memory_order_release); // Back-patch the half-barrier.
|
||||
}
|
||||
std::atomic_ref<uint32_t>(PC[0]).store(STUR, std::memory_order_release);
|
||||
|
||||
ClearICache(&PC[-1], 8);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
} else if ((Instr & ArchHelpers::Arm64::LDAXP_MASK) == ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
/// This is handling the case of paranoid ARMv8.0-a atomic stores.
|
||||
/// This backpatches the ldaxp+stlxp+cbnz if the previous `HandleCASPAL_ARMv8` didn't handle the case.
|
||||
if (ArchHelpers::Arm64::HandleAtomicVectorStore(Instr, ProgramCounter)) {
|
||||
return std::make_pair(true, 0);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::STLXP_MASK) == ArchHelpers::Arm64::STLXP_INST) { // STLXP
|
||||
// Should not trigger - middle of an LDAXP/STAXP pair.
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
|
||||
// Check if another thread backpatched this instruction before this thread got here
|
||||
// Since we got here, this can happen in a couple situations:
|
||||
// - Unhandled instruction (Shouldn't occur, FEX programmer error added a new unhandled atomic)
|
||||
// - Another thread backpatched an atomic access to be a non-atomic access
|
||||
auto AtomicInst = std::atomic_ref<uint32_t>(PC[0]).load(std::memory_order_acquire);
|
||||
if ((AtomicInst & LDSTREGISTER_MASK) == LDR_INST || (AtomicInst & LDSTUNSCALED_MASK) == LDUR_INST) {
|
||||
// This atomic instruction was backpatched to a load.
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
// Check if the next instruction is a DMB.
|
||||
auto DMBInst = std::atomic_ref<uint32_t>(PC[1]).load(std::memory_order_acquire);
|
||||
if (DMBInst == DMB_LD) {
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
} else {
|
||||
// No DMB instruction with this HandleType.
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
} else if ((AtomicInst & LDSTREGISTER_MASK) == STR_INST || (AtomicInst & LDSTUNSCALED_MASK) == STUR_INST) {
|
||||
if (HandleType != UnalignedHandlerType::NonAtomic) {
|
||||
// Check if the previous instruction is a DMB.
|
||||
auto DMBInst = std::atomic_ref<uint32_t>(PC[-1]).load(std::memory_order_acquire);
|
||||
if (DMBInst == DMB) {
|
||||
// Return handled, make sure to adjust PC so we run the DMB.
|
||||
return std::make_pair(true, -4);
|
||||
}
|
||||
} else {
|
||||
// No DMB instruction with this HandleType.
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
} else if (AtomicInst == DMB) {
|
||||
// ARMv8.0-a LDAXP backpatch handling. Will have turned in to the following:
|
||||
// - PC[0] = DMB
|
||||
// - PC[1] = STP
|
||||
// - PC[2] = DMB
|
||||
auto STPInst = std::atomic_ref<uint32_t>(PC[1]).load(std::memory_order_acquire);
|
||||
auto DMBInst = std::atomic_ref<uint32_t>(PC[2]).load(std::memory_order_acquire);
|
||||
if ((STPInst & LDSTP_MASK) == STP_INST && DMBInst == DMB) {
|
||||
// Code that was backpatched is what was expected for ARMv8.0-a LDAXP.
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return NotHandled;
|
||||
}
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
namespace FEXCore::Utils::SpinWaitLock {
|
||||
|
||||
@@ -9,9 +9,12 @@ namespace FEXCore::Utils::SpinWaitLock {
|
||||
*
|
||||
* Spin-loops on mobile devices with a battery can be a bad idea as they burn a bunch of power. This attempts to mitigate some of the impact
|
||||
* by putting the CPU in to a lower-power state using WFE.
|
||||
* On platforms tested, WFE will put the CPU in to a lower power state for upwards of 52ns per WFE. Which isn't a significant amount of time
|
||||
* but should still have power savings. Ideally WFE would be able to keep the CPU in a lower power state for longer. This also has the added
|
||||
* benefit that atomics aren't abusing the caches when spinning on a cacheline, which has knock-on powersaving benefits.
|
||||
* On platforms tested, WFE will put the CPU in to a lower power state for upwards of 0.11ms(!) per WFE. Which isn't a significant amount of
|
||||
* time but should still have power savings. Ideally WFE would be able to keep the CPU in a lower power state for longer. This also has the
|
||||
* added benefit that atomics aren't abusing the caches when spinning on a cacheline, which has knock-on powersaving benefits.
|
||||
*
|
||||
* This short timeout is because the Linux kernel has a 100 microsecond architecture timer which wakes up WFE and WFI. Nothing can be
|
||||
* improved beyond that period.
|
||||
*
|
||||
* FEAT_WFxT adds a new instruction with a timeout, but since the spurious wake-up is so aggressive it isn't worth using.
|
||||
*
|
||||
@@ -269,6 +272,12 @@ static inline void unlock(T* Futex) {
|
||||
template<typename T>
|
||||
class UniqueSpinMutex final {
|
||||
public:
|
||||
// Move-only type
|
||||
UniqueSpinMutex(const UniqueSpinMutex&) = delete;
|
||||
UniqueSpinMutex& operator=(const UniqueSpinMutex&) = delete;
|
||||
UniqueSpinMutex(UniqueSpinMutex&& rhs) = default;
|
||||
UniqueSpinMutex& operator=(UniqueSpinMutex&&) = default;
|
||||
|
||||
UniqueSpinMutex(T* Futex)
|
||||
: Futex {Futex} {
|
||||
FEXCore::Utils::SpinWaitLock::lock(Futex);
|
||||
|
||||
@@ -259,8 +259,6 @@ public:
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void AppendThunkDefinitions(std::span<const FEXCore::IR::ThunkDefinition> Definitions) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void GetVDSOSigReturn(VDSOSigReturn* VDSOPointers) = 0;
|
||||
|
||||
/**
|
||||
* @brief Informs the context if hardware TSO is supported.
|
||||
* Once hardware TSO is enabled, then TSO emulation through atomics is disabled and relies on the hardware.
|
||||
|
||||
@@ -162,6 +162,10 @@ struct CPUState {
|
||||
// zero DF.
|
||||
flags[X86State::RFLAG_DF_RAW_LOC] = 0x1;
|
||||
|
||||
// Likewise, SF/ZF/CF/OF must be cleared. This would be simply zeroing
|
||||
// NZCV... but we invert CF inside the JIT. So set just bit 29 (carry).
|
||||
flags[X86State::RFLAG_NZCV_3_LOC] = (1 << (29 - 24));
|
||||
|
||||
// Default mxcsr value
|
||||
// All exception masks enabled.
|
||||
mxcsr = 0x1F80;
|
||||
@@ -341,6 +345,11 @@ struct CpuStateFrame {
|
||||
|
||||
InternalThreadState* Thread;
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Set by the kernel on ARM64EC whenever the JIT should cooperatively suspend running guest code.
|
||||
uint32_t SuspendDoorbell {};
|
||||
#endif
|
||||
|
||||
// Pointers that the JIT needs to load to remove relocations
|
||||
JITPointers Pointers;
|
||||
};
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -12,6 +14,7 @@ struct HostFeatures {
|
||||
*/
|
||||
uint32_t DCacheLineSize {};
|
||||
uint32_t ICacheLineSize {};
|
||||
bool SupportsCacheMaintenanceOps {};
|
||||
bool SupportsAES {};
|
||||
bool SupportsCRC {};
|
||||
bool SupportsCLZERO {};
|
||||
@@ -36,5 +39,9 @@ struct HostFeatures {
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
bool SupportsFloatExceptions {};
|
||||
|
||||
// MIDR information
|
||||
// Also used for determining number of CPU cores for CPUID
|
||||
fextl::vector<uint32_t> CPUMIDRs;
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -11,103 +11,6 @@ struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t CASPAL_MASK = 0xBF'E0'FC'00;
|
||||
constexpr uint32_t CASPAL_INST = 0x08'60'FC'00;
|
||||
|
||||
constexpr uint32_t CASAL_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t CASAL_INST = 0x08'E0'FC'00;
|
||||
|
||||
constexpr uint32_t ATOMIC_MEM_MASK = 0x3B200C00;
|
||||
constexpr uint32_t ATOMIC_MEM_INST = 0x38200000;
|
||||
|
||||
constexpr uint32_t RCPC2_MASK = 0x3F'E0'0C'00;
|
||||
constexpr uint32_t LDAPUR_INST = 0x19'40'00'00;
|
||||
constexpr uint32_t STLUR_INST = 0x19'00'00'00;
|
||||
|
||||
constexpr uint32_t LDAXP_MASK = 0xBF'FF'80'00;
|
||||
constexpr uint32_t LDAXP_INST = 0x88'7F'80'00;
|
||||
|
||||
constexpr uint32_t STLXP_MASK = 0xBF'E0'80'00;
|
||||
constexpr uint32_t STLXP_INST = 0x88'20'80'00;
|
||||
|
||||
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
|
||||
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
|
||||
constexpr uint32_t LDAR_INST = 0x08'DF'FC'00;
|
||||
constexpr uint32_t LDAPR_INST = 0x38'BF'C0'00;
|
||||
constexpr uint32_t STLR_INST = 0x08'9F'FC'00;
|
||||
|
||||
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
|
||||
|
||||
constexpr uint32_t CBNZ_MASK = 0x7F'00'00'00;
|
||||
constexpr uint32_t CBNZ_INST = 0x35'00'00'00;
|
||||
|
||||
constexpr uint32_t ALU_OP_MASK = 0x7F'20'00'00;
|
||||
constexpr uint32_t ADD_INST = 0x0B'00'00'00;
|
||||
constexpr uint32_t SUB_INST = 0x4B'00'00'00;
|
||||
constexpr uint32_t ADD_SHIFT_INST = 0x0B'20'00'00;
|
||||
constexpr uint32_t SUB_SHIFT_INST = 0x4B'20'00'00;
|
||||
constexpr uint32_t CMP_INST = 0x6B'00'00'00;
|
||||
constexpr uint32_t CMP_SHIFT_INST = 0x6B'20'00'00;
|
||||
constexpr uint32_t AND_INST = 0x0A'00'00'00;
|
||||
constexpr uint32_t BIC_INST = 0x0A'20'00'00;
|
||||
constexpr uint32_t OR_INST = 0x2A'00'00'00;
|
||||
constexpr uint32_t ORN_INST = 0x2A'20'00'00;
|
||||
constexpr uint32_t EOR_INST = 0x4A'00'00'00;
|
||||
constexpr uint32_t EON_INST = 0x4A'20'00'00;
|
||||
|
||||
constexpr uint32_t CCMP_MASK = 0x7F'E0'0C'10;
|
||||
constexpr uint32_t CCMP_INST = 0x7A'40'00'00;
|
||||
|
||||
constexpr uint32_t CLREX_MASK = 0xFF'FF'F0'FF;
|
||||
constexpr uint32_t CLREX_INST = 0xD5'03'30'5F;
|
||||
|
||||
enum ExclusiveAtomicPairType {
|
||||
TYPE_SWAP,
|
||||
TYPE_ADD,
|
||||
TYPE_SUB,
|
||||
TYPE_AND,
|
||||
TYPE_BIC,
|
||||
TYPE_OR,
|
||||
TYPE_ORN,
|
||||
TYPE_EOR,
|
||||
TYPE_EON,
|
||||
TYPE_NEG, // This is just a sub with zero. Need to know the differences
|
||||
};
|
||||
|
||||
// Load ops are 4 bits
|
||||
// Acquire and release bits are independent on the instruction
|
||||
constexpr uint32_t ATOMIC_ADD_OP = 0b0000;
|
||||
constexpr uint32_t ATOMIC_CLR_OP = 0b0001;
|
||||
constexpr uint32_t ATOMIC_EOR_OP = 0b0010;
|
||||
constexpr uint32_t ATOMIC_SET_OP = 0b0011;
|
||||
constexpr uint32_t ATOMIC_SMAX_OP = 0b0100;
|
||||
constexpr uint32_t ATOMIC_SMIN_OP = 0b0101;
|
||||
constexpr uint32_t ATOMIC_UMAX_OP = 0b0110;
|
||||
constexpr uint32_t ATOMIC_UMIN_OP = 0b0111;
|
||||
constexpr uint32_t ATOMIC_SWAP_OP = 0b1000;
|
||||
|
||||
constexpr uint32_t REGISTER_MASK = 0b11111;
|
||||
constexpr uint32_t RD_OFFSET = 0;
|
||||
constexpr uint32_t RN_OFFSET = 5;
|
||||
constexpr uint32_t RM_OFFSET = 16;
|
||||
|
||||
constexpr uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 | 0b1011'0000'0000; // Inner shareable all
|
||||
|
||||
constexpr uint32_t DMB_LD = 0b1101'0101'0000'0011'0011'0000'1011'1111 | 0b1101'0000'0000; // Inner shareable load
|
||||
|
||||
inline uint32_t GetRdReg(uint32_t Instr) {
|
||||
return (Instr >> RD_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
inline uint32_t GetRnReg(uint32_t Instr) {
|
||||
return (Instr >> RN_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
inline uint32_t GetRmReg(uint32_t Instr) {
|
||||
return (Instr >> RM_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
enum class UnalignedHandlerType {
|
||||
///< Don't backpatch code, instead handle inside SIGBUS handler.
|
||||
Paranoid,
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
#include <FEXCore/Utils/File.h>
|
||||
|
||||
#include <fmt/format.h>
|
||||
#include <fmt/ranges.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace fextl::fmt {
|
||||
|
||||
@@ -5,6 +5,11 @@ import sys
|
||||
logger = logging.getLogger()
|
||||
logger.setLevel(logging.ERROR)
|
||||
|
||||
def insert_before(d, key, item):
|
||||
items = list(d.items())
|
||||
items.insert(list(d.keys()).index(key), item)
|
||||
return dict(items)
|
||||
|
||||
def update_performance_numbers(performance_json_path, performance_json, new_json_numbers):
|
||||
for key, items in new_json_numbers.items():
|
||||
if len(key) == 0:
|
||||
@@ -18,6 +23,12 @@ def update_performance_numbers(performance_json_path, performance_json, new_json
|
||||
performance_json["Instructions"][key]["ExpectedInstructionCount"] = items["ExpectedInstructionCount"]
|
||||
if "ExpectedArm64ASM" in items:
|
||||
performance_json["Instructions"][key]["ExpectedArm64ASM"] = items["ExpectedArm64ASM"]
|
||||
if "x86Insts" in performance_json["Instructions"][key]:
|
||||
d = performance_json["Instructions"][key]
|
||||
d.pop('x86InstructionCount', None)
|
||||
d = insert_before(d, "ExpectedInstructionCount",
|
||||
("x86InstructionCount", len(d["x86Insts"])))
|
||||
performance_json["Instructions"][key] = d
|
||||
|
||||
# Output to the original file.
|
||||
with open(performance_json_path, "w") as json_file:
|
||||
|
||||
@@ -16,7 +16,7 @@ if (NOT MINGW_BUILD)
|
||||
endif()
|
||||
|
||||
add_library(${NAME} STATIC ${SRCS})
|
||||
target_link_libraries(${NAME} FEXCore_Base cpp-optparse tiny-json json-maker FEXHeaderUtils)
|
||||
target_link_libraries(${NAME} FEXCore_Base cpp-optparse tiny-json FEXHeaderUtils)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/External/cpp-optparse/)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_BINARY_DIR}/generated)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_SOURCE_DIR}/External/xbyak/)
|
||||
+79
-69
@@ -17,7 +17,6 @@
|
||||
#include <pwd.h>
|
||||
#endif
|
||||
#include <utility>
|
||||
#include <json-maker.h>
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEX::Config {
|
||||
@@ -338,61 +337,69 @@ fextl::string RecoverGuestProgramFilename(fextl::string Program, bool ExecFDInte
|
||||
return Program;
|
||||
}
|
||||
|
||||
ApplicationNames LoadConfig(fextl::unique_ptr<FEX::ArgLoader::ArgLoader> ArgsLoader, bool LoadProgramConfig, char** const envp,
|
||||
bool ExecFDInterp, int ProgramFDFromEnv) {
|
||||
FEX::Config::InitializeConfigs();
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(CreateMainLayer());
|
||||
|
||||
auto Args = ArgsLoader->Get();
|
||||
ApplicationNames GetApplicationNames(fextl::vector<fextl::string> Args, bool ExecFDInterp, int ProgramFDFromEnv) {
|
||||
if (Args.empty()) {
|
||||
// Early exit if we weren't passed an argument
|
||||
return {};
|
||||
}
|
||||
|
||||
fextl::string Program {};
|
||||
fextl::string ProgramName {};
|
||||
if (LoadProgramConfig) {
|
||||
if (Args.empty()) {
|
||||
// Early exit if we weren't passed an argument
|
||||
return {};
|
||||
}
|
||||
|
||||
Args[0] = RecoverGuestProgramFilename(std::move(Args[0]), ExecFDInterp, ProgramFDFromEnv);
|
||||
Program = Args[0];
|
||||
Args[0] = RecoverGuestProgramFilename(std::move(Args[0]), ExecFDInterp, ProgramFDFromEnv);
|
||||
Program = Args[0];
|
||||
|
||||
bool Wine = false;
|
||||
for (size_t CurrentProgramNameIndex = 0; CurrentProgramNameIndex < Args.size(); ++CurrentProgramNameIndex) {
|
||||
auto CurrentProgramName = FHU::Filesystem::GetFilename(Args[CurrentProgramNameIndex]);
|
||||
bool Wine = false;
|
||||
for (size_t CurrentProgramNameIndex = 0; CurrentProgramNameIndex < Args.size(); ++CurrentProgramNameIndex) {
|
||||
auto CurrentProgramName = FHU::Filesystem::GetFilename(Args[CurrentProgramNameIndex]);
|
||||
|
||||
if (CurrentProgramName == "wine-preloader" || CurrentProgramName == "wine64-preloader") {
|
||||
// Wine preloader is required to be in the format of `wine-preloader <wine executable>`
|
||||
// The preloader doesn't execve the executable, instead maps it directly itself
|
||||
// Skip the next argument since we know it is wine (potentially with custom wine executable name)
|
||||
++CurrentProgramNameIndex;
|
||||
Wine = true;
|
||||
} else if (CurrentProgramName == "wine" || CurrentProgramName == "wine64") {
|
||||
// Next argument, this isn't the program we want
|
||||
//
|
||||
// If we are running wine or wine64 then we should check the next argument for the application name instead.
|
||||
// wine will change the active program name with `setprogname` or `prctl(PR_SET_NAME`.
|
||||
// Since FEX needs this data far earlier than libraries we need a different check.
|
||||
Wine = true;
|
||||
} else {
|
||||
if (Wine == true) {
|
||||
// If this was path separated with '\' then we need to check that.
|
||||
auto WinSeparator = CurrentProgramName.find_last_of('\\');
|
||||
if (WinSeparator != CurrentProgramName.npos) {
|
||||
// Used windows separators
|
||||
CurrentProgramName = CurrentProgramName.substr(WinSeparator + 1);
|
||||
}
|
||||
if (CurrentProgramName == "wine-preloader" || CurrentProgramName == "wine64-preloader") {
|
||||
// Wine preloader is required to be in the format of `wine-preloader <wine executable>`
|
||||
// The preloader doesn't execve the executable, instead maps it directly itself
|
||||
// Skip the next argument since we know it is wine (potentially with custom wine executable name)
|
||||
++CurrentProgramNameIndex;
|
||||
Wine = true;
|
||||
} else if (CurrentProgramName == "wine" || CurrentProgramName == "wine64") {
|
||||
// Next argument, this isn't the program we want
|
||||
//
|
||||
// If we are running wine or wine64 then we should check the next argument for the application name instead.
|
||||
// wine will change the active program name with `setprogname` or `prctl(PR_SET_NAME`.
|
||||
// Since FEX needs this data far earlier than libraries we need a different check.
|
||||
Wine = true;
|
||||
} else {
|
||||
if (Wine == true) {
|
||||
// If this was path separated with '\' then we need to check that.
|
||||
auto WinSeparator = CurrentProgramName.find_last_of('\\');
|
||||
if (WinSeparator != CurrentProgramName.npos) {
|
||||
// Used windows separators
|
||||
CurrentProgramName = CurrentProgramName.substr(WinSeparator + 1);
|
||||
}
|
||||
|
||||
ProgramName = CurrentProgramName;
|
||||
|
||||
// Past any wine program names
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_GLOBAL_APP));
|
||||
ProgramName = CurrentProgramName;
|
||||
|
||||
// Past any wine program names
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return ApplicationNames {std::move(Program), std::move(ProgramName)};
|
||||
}
|
||||
|
||||
void LoadConfig(fextl::unique_ptr<FEX::ArgLoader::ArgLoader> ArgsLoader, fextl::string ProgramName, char** const envp,
|
||||
const PortableInformation& PortableInfo) {
|
||||
const bool IsPortable = PortableInfo.IsPortable;
|
||||
FEX::Config::InitializeConfigs(PortableInfo);
|
||||
FEXCore::Config::Initialize();
|
||||
if (!IsPortable) {
|
||||
FEXCore::Config::AddLayer(CreateGlobalMainLayer());
|
||||
}
|
||||
FEXCore::Config::AddLayer(CreateMainLayer());
|
||||
|
||||
if (!ProgramName.empty()) {
|
||||
if (!IsPortable) {
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_GLOBAL_APP));
|
||||
}
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(ProgramName, FEXCore::Config::LayerType::LAYER_LOCAL_APP));
|
||||
|
||||
auto SteamID = getenv("SteamAppId");
|
||||
@@ -400,12 +407,14 @@ ApplicationNames LoadConfig(fextl::unique_ptr<FEX::ArgLoader::ArgLoader> ArgsLoa
|
||||
// If a SteamID exists then let's search for Steam application configs as well.
|
||||
// We want to key off both the SteamAppId number /and/ the executable since we may not want to thunk all binaries.
|
||||
fextl::string SteamAppName = fextl::fmt::format("Steam_{}_{}", SteamID, ProgramName);
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP));
|
||||
if (!IsPortable) {
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP));
|
||||
}
|
||||
FEXCore::Config::AddLayer(CreateAppLayer(SteamAppName, FEXCore::Config::LayerType::LAYER_LOCAL_STEAM_APP));
|
||||
}
|
||||
}
|
||||
|
||||
if (ArgsLoader->GetLoadType() == FEX::ArgLoader::ArgLoader::LoadType::WITH_FEXLOADER_PARSER) {
|
||||
if (ArgsLoader && ArgsLoader->GetLoadType() == FEX::ArgLoader::ArgLoader::LoadType::WITH_FEXLOADER_PARSER) {
|
||||
FEXCore::Config::AddLayer(std::move(ArgsLoader));
|
||||
}
|
||||
|
||||
@@ -416,13 +425,6 @@ ApplicationNames LoadConfig(fextl::unique_ptr<FEX::ArgLoader::ArgLoader> ArgsLoa
|
||||
|
||||
FEXCore::Config::AddLayer(CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
|
||||
|
||||
if (LoadProgramConfig) {
|
||||
return ApplicationNames {std::move(Program), std::move(ProgramName)};
|
||||
} else {
|
||||
return {};
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
@@ -474,12 +476,16 @@ const char* GetHomeDirectory() {
|
||||
}
|
||||
#endif
|
||||
|
||||
fextl::string GetDataDirectory() {
|
||||
fextl::string DataDir {};
|
||||
fextl::string GetDataDirectory(const PortableInformation& PortableInfo) {
|
||||
const char* DataOverride = getenv("FEX_APP_DATA_LOCATION");
|
||||
|
||||
if (PortableInfo.IsPortable && !DataOverride) {
|
||||
return fextl::fmt::format("{}fex-emu/", PortableInfo.InterpreterPath);
|
||||
}
|
||||
|
||||
fextl::string DataDir {};
|
||||
const char* HomeDir = GetHomeDirectory();
|
||||
const char* DataXDG = getenv("XDG_DATA_HOME");
|
||||
const char* DataOverride = getenv("FEX_APP_DATA_LOCATION");
|
||||
if (DataOverride) {
|
||||
// Data override will override the complete directory
|
||||
DataDir = DataOverride;
|
||||
@@ -490,14 +496,18 @@ fextl::string GetDataDirectory() {
|
||||
return DataDir;
|
||||
}
|
||||
|
||||
fextl::string GetConfigDirectory(bool Global) {
|
||||
fextl::string GetConfigDirectory(bool Global, const PortableInformation& PortableInfo) {
|
||||
const char* ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (PortableInfo.IsPortable && (Global || !ConfigOverride)) {
|
||||
return fextl::fmt::format("{}fex-emu/", PortableInfo.InterpreterPath);
|
||||
}
|
||||
|
||||
fextl::string ConfigDir;
|
||||
if (Global) {
|
||||
ConfigDir = GLOBAL_DATA_DIRECTORY;
|
||||
} else {
|
||||
const char* HomeDir = GetHomeDirectory();
|
||||
const char* ConfigXDG = getenv("XDG_CONFIG_HOME");
|
||||
const char* ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (ConfigOverride) {
|
||||
// Config override completely overrides the config directory
|
||||
ConfigDir = ConfigOverride;
|
||||
@@ -516,15 +526,15 @@ fextl::string GetConfigDirectory(bool Global) {
|
||||
return ConfigDir;
|
||||
}
|
||||
|
||||
fextl::string GetConfigFileLocation(bool Global) {
|
||||
return GetConfigDirectory(Global) + "Config.json";
|
||||
fextl::string GetConfigFileLocation(bool Global, const PortableInformation& PortableInfo) {
|
||||
return GetConfigDirectory(Global, PortableInfo) + "Config.json";
|
||||
}
|
||||
|
||||
void InitializeConfigs() {
|
||||
FEXCore::Config::SetDataDirectory(GetDataDirectory());
|
||||
FEXCore::Config::SetConfigDirectory(GetConfigDirectory(false), false);
|
||||
FEXCore::Config::SetConfigDirectory(GetConfigDirectory(true), true);
|
||||
FEXCore::Config::SetConfigFileLocation(GetConfigFileLocation(false), false);
|
||||
FEXCore::Config::SetConfigFileLocation(GetConfigFileLocation(true), true);
|
||||
void InitializeConfigs(const PortableInformation& PortableInfo) {
|
||||
FEXCore::Config::SetDataDirectory(GetDataDirectory(PortableInfo));
|
||||
FEXCore::Config::SetConfigDirectory(GetConfigDirectory(false, PortableInfo), false);
|
||||
FEXCore::Config::SetConfigDirectory(GetConfigDirectory(true, PortableInfo), true);
|
||||
FEXCore::Config::SetConfigFileLocation(GetConfigFileLocation(false, PortableInfo), false);
|
||||
FEXCore::Config::SetConfigFileLocation(GetConfigFileLocation(true, PortableInfo), true);
|
||||
}
|
||||
} // namespace FEX::Config
|
||||
+21
-11
@@ -28,27 +28,37 @@ struct ApplicationNames {
|
||||
fextl::string ProgramName;
|
||||
};
|
||||
|
||||
struct PortableInformation {
|
||||
bool IsPortable;
|
||||
// Path of folder containing FEXInterpreter (including / at the end)
|
||||
fextl::string InterpreterPath;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Loads the FEX and application configurations for the application that is getting ready to run.
|
||||
*
|
||||
* @param ArgLoader Argument loader for argument based config options
|
||||
* @param LoadProgramConfig Do we want to load application specific configurations?
|
||||
* @param envp The `envp` passed to main(...)
|
||||
* @param ExecFDInterp If FEX was executed with binfmt_misc FD argument
|
||||
* @param ProgramFDFromEnv The execveat FD argument passed through FEX
|
||||
*
|
||||
* @return The application name and path structure
|
||||
*/
|
||||
ApplicationNames LoadConfig(fextl::unique_ptr<FEX::ArgLoader::ArgLoader> ArgLoader, bool LoadProgramConfig, char** const envp,
|
||||
bool ExecFDInterp, int ProgramFDFromEnv);
|
||||
ApplicationNames GetApplicationNames(fextl::vector<fextl::string> Args, bool ExecFDInterp, int ProgramFDFromEnv);
|
||||
|
||||
/**
|
||||
* @brief Loads the FEX and application configurations for the application that is getting ready to run.
|
||||
*
|
||||
* @param ArgLoader Optional argument loader for argument based config options
|
||||
* @param ProgramName Optional program name, if non-empty application specific configurations will be loaded
|
||||
* @param envp Optional `envp` passed to main(...)
|
||||
*/
|
||||
void LoadConfig(fextl::unique_ptr<FEX::ArgLoader::ArgLoader> ArgLoader = {}, fextl::string ProgramName = {}, char** const envp = nullptr,
|
||||
const PortableInformation& PortableInfo = {});
|
||||
|
||||
const char* GetHomeDirectory();
|
||||
|
||||
fextl::string GetDataDirectory();
|
||||
fextl::string GetConfigDirectory(bool Global);
|
||||
fextl::string GetConfigFileLocation(bool Global);
|
||||
fextl::string GetDataDirectory(const PortableInformation& PortableInfo);
|
||||
fextl::string GetConfigDirectory(bool Global, const PortableInformation& PortableInfo);
|
||||
fextl::string GetConfigFileLocation(bool Global, const PortableInformation& PortableInfo);
|
||||
|
||||
void InitializeConfigs();
|
||||
void InitializeConfigs(const PortableInformation& PortableInfo);
|
||||
|
||||
/**
|
||||
* @brief Loads the global FEX config
|
||||
|
||||
+391
-53
@@ -1,8 +1,11 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/HostFeatures.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#define XBYAK64
|
||||
@@ -17,6 +20,339 @@
|
||||
|
||||
namespace FEX {
|
||||
|
||||
void FillMIDRInformationViaLinux(FEXCore::HostFeatures* Features) {
|
||||
auto Cores = FEXCore::CPUInfo::CalculateNumberOfCPUs();
|
||||
Features->CPUMIDRs.resize(Cores);
|
||||
#ifdef _M_ARM_64
|
||||
for (size_t i = 0; i < Cores; ++i) {
|
||||
std::error_code ec {};
|
||||
fextl::string MIDRPath = fextl::fmt::format("/sys/devices/system/cpu/cpu{}/regs/identification/midr_el1", i);
|
||||
std::array<char, 18> Data;
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFileToBuffer(MIDRPath, Data) == sizeof(Data)) {
|
||||
uint64_t MIDR {};
|
||||
auto Results = std::from_chars(Data.data() + 2, Data.data() + sizeof(Data), MIDR, 16);
|
||||
if (Results.ec == std::errc()) {
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
Features->CPUMIDRs[i] = static_cast<uint32_t>(MIDR);
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#define GetSysReg(name, reg) \
|
||||
static uint64_t Get_##name() { \
|
||||
uint64_t Result {}; \
|
||||
__asm("mrs %[Res], " #reg : [Res] "=r"(Result)); \
|
||||
return Result; \
|
||||
}
|
||||
|
||||
GetSysReg(ISAR0_EL1, ID_AA64ISAR0_EL1);
|
||||
GetSysReg(PFR0_EL1, ID_AA64PFR0_EL1);
|
||||
GetSysReg(PFR1_EL1, ID_AA64PFR1_EL1);
|
||||
GetSysReg(MIDR_EL1, MIDR_EL1);
|
||||
GetSysReg(ISAR1_EL1, ID_AA64ISAR1_EL1);
|
||||
GetSysReg(MMFR0_EL1, ID_AA64MMFR0_EL1);
|
||||
GetSysReg(MMFR2_EL1, ID_AA64MMFR2_EL1);
|
||||
GetSysReg(ZFR0_EL1, s3_0_c0_c4_4); // Can't request by name
|
||||
GetSysReg(MMFR1_EL1, ID_AA64MMFR1_EL1);
|
||||
GetSysReg(ISAR2_EL1, ID_AA64ISAR2_EL1);
|
||||
|
||||
class CPUFeaturesFromID final : public FEX::CPUFeatures {
|
||||
public:
|
||||
CPUFeaturesFromID() {
|
||||
ISAR0.SetReg(Get_ISAR0_EL1());
|
||||
PFR0.SetReg(Get_PFR0_EL1());
|
||||
PFR1.SetReg(Get_PFR1_EL1());
|
||||
MIDR.SetReg(Get_MIDR_EL1());
|
||||
ISAR1.SetReg(Get_ISAR1_EL1());
|
||||
MMFR0.SetReg(Get_MMFR0_EL1());
|
||||
MMFR2.SetReg(Get_MMFR2_EL1());
|
||||
MMFR1.SetReg(Get_MMFR1_EL1());
|
||||
ISAR2.SetReg(Get_ISAR2_EL1());
|
||||
|
||||
if (PFR0.SupportsSVE()) {
|
||||
// Can only query if SVE is supported.
|
||||
ZFR0.SetReg(Get_ZFR0_EL1());
|
||||
}
|
||||
FillFeatureFlags();
|
||||
}
|
||||
};
|
||||
|
||||
FEX::CPUFeatures GetCPUFeaturesFromIDRegisters() {
|
||||
return CPUFeaturesFromID {};
|
||||
}
|
||||
#endif
|
||||
|
||||
class CPUFeaturesAll final : public FEX::CPUFeatures {
|
||||
public:
|
||||
CPUFeaturesAll() {
|
||||
// Special case, just set all feature flags
|
||||
for (uint32_t i = 0; i < FEXCore::ToUnderlying(FEX::CPUFeatures::Feature::MAX); ++i) {
|
||||
SetFeature(FEX::CPUFeatures::Feature {i});
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
void FEX::CPUFeatures::FillFeatureFlags() {
|
||||
// ISAR0
|
||||
if (ISAR0.SupportsAES()) {
|
||||
SetFeature(Feature::AES);
|
||||
}
|
||||
if (ISAR0.SupportsPMULL()) {
|
||||
SetFeature(Feature::PMULL);
|
||||
}
|
||||
if (ISAR0.SupportsSHA1()) {
|
||||
SetFeature(Feature::SHA1);
|
||||
}
|
||||
if (ISAR0.SupportsSHA2()) {
|
||||
SetFeature(Feature::SHA2);
|
||||
}
|
||||
if (ISAR0.SupportsSHA512()) {
|
||||
SetFeature(Feature::SHA512);
|
||||
}
|
||||
if (ISAR0.SupportsCRC32()) {
|
||||
SetFeature(Feature::CRC32);
|
||||
}
|
||||
if (ISAR0.SupportsLSE()) {
|
||||
SetFeature(Feature::LSE);
|
||||
}
|
||||
if (ISAR0.SupportsLSE128()) {
|
||||
SetFeature(Feature::LSE128);
|
||||
}
|
||||
if (ISAR0.SupportsTME()) {
|
||||
SetFeature(Feature::TME);
|
||||
}
|
||||
if (ISAR0.SupportsRDM()) {
|
||||
SetFeature(Feature::RDM);
|
||||
}
|
||||
if (ISAR0.SupportsSHA3()) {
|
||||
SetFeature(Feature::SHA3);
|
||||
}
|
||||
if (ISAR0.SupportsSM3()) {
|
||||
SetFeature(Feature::SM3);
|
||||
}
|
||||
if (ISAR0.SupportsSM4()) {
|
||||
SetFeature(Feature::SM4);
|
||||
}
|
||||
if (ISAR0.SupportsDotProd()) {
|
||||
SetFeature(Feature::DotProd);
|
||||
}
|
||||
if (ISAR0.SupportsFlagM()) {
|
||||
SetFeature(Feature::FlagM);
|
||||
}
|
||||
if (ISAR0.SupportsFlagM2()) {
|
||||
SetFeature(Feature::FlagM2);
|
||||
}
|
||||
if (ISAR0.SupportsRNDR()) {
|
||||
SetFeature(Feature::RNDR);
|
||||
}
|
||||
|
||||
// PFR0
|
||||
if (PFR0.SupportsFP()) {
|
||||
SetFeature(Feature::FP);
|
||||
}
|
||||
if (PFR0.SupportsHP()) {
|
||||
SetFeature(Feature::FP16);
|
||||
}
|
||||
if (PFR0.SupportsAdvSIMD()) {
|
||||
SetFeature(Feature::ASIMD);
|
||||
}
|
||||
if (PFR0.SupportsASIMDHP()) {
|
||||
SetFeature(Feature::ASIMD16);
|
||||
}
|
||||
if (PFR0.SupportsRAS()) {
|
||||
SetFeature(Feature::RAS);
|
||||
}
|
||||
if (PFR0.SupportsSVE()) {
|
||||
SetFeature(Feature::SVE);
|
||||
}
|
||||
if (PFR0.SupportsDIT()) {
|
||||
SetFeature(Feature::DIT);
|
||||
}
|
||||
if (PFR0.SupportsCSV2()) {
|
||||
SetFeature(Feature::CSV2);
|
||||
}
|
||||
if (PFR0.SupportsCSV3()) {
|
||||
SetFeature(Feature::CSV3);
|
||||
}
|
||||
|
||||
// PFR1
|
||||
if (PFR1.SupportsBTI()) {
|
||||
SetFeature(Feature::BTI);
|
||||
}
|
||||
if (PFR1.SupportsSSBS()) {
|
||||
SetFeature(Feature::SSBS);
|
||||
}
|
||||
if (PFR1.SupportsSSBS()) {
|
||||
SetFeature(Feature::SSBS2);
|
||||
}
|
||||
if (PFR1.SupportsMTE()) {
|
||||
SetFeature(Feature::MTE);
|
||||
}
|
||||
if (PFR1.SupportsMTE2()) {
|
||||
SetFeature(Feature::MTE2);
|
||||
}
|
||||
if (PFR1.SupportsMTE3()) {
|
||||
SetFeature(Feature::MTE3);
|
||||
}
|
||||
if (PFR1.SupportsSME()) {
|
||||
SetFeature(Feature::SME);
|
||||
}
|
||||
if (PFR1.SupportsSME2()) {
|
||||
SetFeature(Feature::SME2);
|
||||
}
|
||||
|
||||
// ISAR1
|
||||
if (ISAR1.SupportsDPB()) {
|
||||
SetFeature(Feature::DPB);
|
||||
}
|
||||
if (ISAR1.SupportsDPB2()) {
|
||||
SetFeature(Feature::DPB2);
|
||||
}
|
||||
if (ISAR1.SupportsJSCVT()) {
|
||||
SetFeature(Feature::JSCVT);
|
||||
}
|
||||
if (ISAR1.SupportsFCMA()) {
|
||||
SetFeature(Feature::FCMA);
|
||||
}
|
||||
if (ISAR1.SupportsLRCPC()) {
|
||||
SetFeature(Feature::LRCPC);
|
||||
}
|
||||
if (ISAR1.SupportsLRCPC2()) {
|
||||
SetFeature(Feature::LRCPC2);
|
||||
}
|
||||
if (ISAR1.SupportsLRCPC3()) {
|
||||
SetFeature(Feature::LRCPC3);
|
||||
}
|
||||
if (ISAR1.SupportsFRINTTS()) {
|
||||
SetFeature(Feature::FRINTTS);
|
||||
}
|
||||
if (ISAR1.SupportsSB()) {
|
||||
SetFeature(Feature::SB);
|
||||
}
|
||||
if (ISAR1.SupportsSPECRES()) {
|
||||
SetFeature(Feature::SPECRES);
|
||||
}
|
||||
if (ISAR1.SupportsSPECRES2()) {
|
||||
SetFeature(Feature::SPECRES2);
|
||||
}
|
||||
if (ISAR1.SupportsBF16()) {
|
||||
SetFeature(Feature::BF16);
|
||||
}
|
||||
if (ISAR1.SupportsSME_F64F64()) {
|
||||
SetFeature(Feature::SME_F64F64);
|
||||
}
|
||||
if (ISAR1.SupportsI8MM()) {
|
||||
SetFeature(Feature::I8MM);
|
||||
}
|
||||
if (ISAR1.SupportsXS()) {
|
||||
SetFeature(Feature::XS);
|
||||
}
|
||||
if (ISAR1.SupportsLS64()) {
|
||||
SetFeature(Feature::LS64);
|
||||
}
|
||||
if (ISAR1.SupportsLS64_V()) {
|
||||
SetFeature(Feature::LS64_V);
|
||||
}
|
||||
if (ISAR1.SupportsLS64_ACCDATA()) {
|
||||
SetFeature(Feature::LS64_ACCDATA);
|
||||
}
|
||||
|
||||
// MMFR0
|
||||
if (MMFR0.SupportsECV()) {
|
||||
SetFeature(Feature::ECV);
|
||||
}
|
||||
|
||||
// MMFR2
|
||||
if (MMFR2.SupportsLSE2()) {
|
||||
SetFeature(Feature::LSE2);
|
||||
}
|
||||
|
||||
// ZFR0
|
||||
if (Supports(Feature::SVE)) {
|
||||
if (ZFR0.SupportsSVE2()) {
|
||||
SetFeature(Feature::SVE2);
|
||||
}
|
||||
if (ZFR0.SupportsSVE2_1()) {
|
||||
SetFeature(Feature::SVE2_1);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_AES()) {
|
||||
SetFeature(Feature::SVE_AES);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_PMULL128()) {
|
||||
SetFeature(Feature::SVE_PMULL128);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_BitPerm()) {
|
||||
SetFeature(Feature::SVE_BitPerm);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_BF16()) {
|
||||
SetFeature(Feature::SVE_BF16);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_B16B16()) {
|
||||
SetFeature(Feature::SVE_B16B16);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_SHA3()) {
|
||||
SetFeature(Feature::SVE_SHA3);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_SM4()) {
|
||||
SetFeature(Feature::SVE_SM4);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_I8MM()) {
|
||||
SetFeature(Feature::SVE_I8MM);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_F32MM()) {
|
||||
SetFeature(Feature::SVE_F32MM);
|
||||
}
|
||||
if (ZFR0.SupportsSVE_F64MM()) {
|
||||
SetFeature(Feature::SVE_F64MM);
|
||||
}
|
||||
}
|
||||
|
||||
// MMFR1
|
||||
if (MMFR1.SupportsAFP()) {
|
||||
SetFeature(Feature::AFP);
|
||||
}
|
||||
|
||||
// ISAR2
|
||||
if (ISAR2.SupportsWFxt()) {
|
||||
SetFeature(Feature::WFxt);
|
||||
}
|
||||
if (ISAR2.SupportsRPRES()) {
|
||||
SetFeature(Feature::RPRES);
|
||||
}
|
||||
if (ISAR2.SupportsPACQARMA3()) {
|
||||
SetFeature(Feature::PACQARMA3);
|
||||
}
|
||||
if (ISAR2.SupportsMOPS()) {
|
||||
SetFeature(Feature::MOPS);
|
||||
}
|
||||
if (ISAR2.SupportsHBC()) {
|
||||
SetFeature(Feature::HBC);
|
||||
}
|
||||
if (ISAR2.SupportsCLRBHB()) {
|
||||
SetFeature(Feature::CLRBHB);
|
||||
}
|
||||
if (ISAR2.SupportsSYSREG128()) {
|
||||
SetFeature(Feature::SYSREG128);
|
||||
}
|
||||
if (ISAR2.SupportsSYSINSTR128()) {
|
||||
SetFeature(Feature::SYSINSTR128);
|
||||
}
|
||||
if (ISAR2.SupportsPRFMSLC()) {
|
||||
SetFeature(Feature::PRFMSLC);
|
||||
}
|
||||
if (ISAR2.SupportsRPRFM()) {
|
||||
SetFeature(Feature::RPRFM);
|
||||
}
|
||||
if (ISAR2.SupportsCSSC()) {
|
||||
SetFeature(Feature::CSSC);
|
||||
}
|
||||
}
|
||||
|
||||
// Data Zero Prohibited flag
|
||||
// 0b0 = ZVA/GVA/GZVA permitted
|
||||
// 0b1 = ZVA/GVA/GZVA prohibited
|
||||
@@ -42,12 +378,6 @@ static void SetFPCR(uint64_t Value) {
|
||||
__asm("msr FPCR, %[Value]" ::[Value] "r"(Value));
|
||||
}
|
||||
|
||||
static uint32_t GetMIDR() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], MIDR_EL1" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
__attribute__((naked)) static uint64_t ReadSVEVectorLengthInBits() {
|
||||
///< Can't use rdvl instruction directly because compilers will complain that sve/sme is required.
|
||||
__asm(R"(
|
||||
@@ -131,49 +461,39 @@ static void OverrideFeatures(FEXCore::HostFeatures* Features, uint64_t ForceSVEW
|
||||
Features->SupportsSVE256 = ForceSVEWidth && ForceSVEWidth >= 256;
|
||||
}
|
||||
|
||||
FEXCore::HostFeatures FetchHostFeatures() {
|
||||
FEXCore::HostFeatures FetchHostFeatures(FEX::CPUFeatures& Features, bool SupportsCacheMaintenanceOps, uint64_t CTR, uint64_t MIDR) {
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
// Vixl simulator doesn't support AFP.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kAFP);
|
||||
// Vixl simulator doesn't support RPRES.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kRPRES);
|
||||
#elif !defined(_WIN32)
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#else
|
||||
// Need to use ID registers in WINE.
|
||||
auto Features = vixl::CPUFeatures::InferFromIDRegisters();
|
||||
#endif
|
||||
|
||||
FEX_CONFIG_OPT(ForceSVEWidth, FORCESVEWIDTH);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
|
||||
HostFeatures.SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
HostFeatures.SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
HostFeatures.SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) && Features.Has(vixl::CPUFeatures::Feature::kSHA2);
|
||||
HostFeatures.SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
HostFeatures.SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
|
||||
HostFeatures.SupportsCacheMaintenanceOps = SupportsCacheMaintenanceOps;
|
||||
|
||||
HostFeatures.SupportsAES = Features.Supports(CPUFeatures::Feature::AES);
|
||||
HostFeatures.SupportsCRC = Features.Supports(CPUFeatures::Feature::CRC32);
|
||||
HostFeatures.SupportsSHA = Features.Supports(CPUFeatures::Feature::SHA1) && Features.Supports(CPUFeatures::Feature::SHA2);
|
||||
HostFeatures.SupportsAtomics = Features.Supports(CPUFeatures::Feature::LSE);
|
||||
HostFeatures.SupportsRAND = Features.Supports(CPUFeatures::Feature::RNDR);
|
||||
|
||||
// Only supported when FEAT_AFP is supported
|
||||
HostFeatures.SupportsAFP = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
HostFeatures.SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
HostFeatures.SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
HostFeatures.SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
HostFeatures.SupportsCSSC = Features.Has(vixl::CPUFeatures::Feature::kCSSC);
|
||||
HostFeatures.SupportsFCMA = Features.Has(vixl::CPUFeatures::Feature::kFcma);
|
||||
HostFeatures.SupportsFlagM = Features.Has(vixl::CPUFeatures::Feature::kFlagM);
|
||||
HostFeatures.SupportsFlagM2 = Features.Has(vixl::CPUFeatures::Feature::kAXFlag);
|
||||
HostFeatures.SupportsRPRES = Features.Has(vixl::CPUFeatures::Feature::kRPRES);
|
||||
HostFeatures.SupportsSVEBitPerm = Features.Has(vixl::CPUFeatures::Feature::kSVEBitPerm);
|
||||
HostFeatures.SupportsAFP = Features.Supports(CPUFeatures::Feature::AFP);
|
||||
HostFeatures.SupportsRCPC = Features.Supports(CPUFeatures::Feature::LRCPC);
|
||||
HostFeatures.SupportsTSOImm9 = Features.Supports(CPUFeatures::Feature::LRCPC2);
|
||||
HostFeatures.SupportsPMULL_128Bit = Features.Supports(CPUFeatures::Feature::PMULL);
|
||||
HostFeatures.SupportsCSSC = Features.Supports(CPUFeatures::Feature::CSSC);
|
||||
HostFeatures.SupportsFCMA = Features.Supports(CPUFeatures::Feature::FCMA);
|
||||
HostFeatures.SupportsFlagM = Features.Supports(CPUFeatures::Feature::FlagM);
|
||||
HostFeatures.SupportsFlagM2 = Features.Supports(CPUFeatures::Feature::FlagM2);
|
||||
HostFeatures.SupportsRPRES = Features.Supports(CPUFeatures::Feature::RPRES);
|
||||
HostFeatures.SupportsSVEBitPerm = Features.Supports(CPUFeatures::Feature::SVE_BitPerm);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode enable SVE with 256-bit wide registers.
|
||||
HostFeatures.SupportsSVE128 = ForceSVEWidth() ? ForceSVEWidth() >= 128 : true;
|
||||
HostFeatures.SupportsSVE256 = ForceSVEWidth() ? ForceSVEWidth() >= 256 : true;
|
||||
#else
|
||||
HostFeatures.SupportsSVE128 = Features.Has(vixl::CPUFeatures::Feature::kSVE2);
|
||||
HostFeatures.SupportsSVE256 = Features.Has(vixl::CPUFeatures::Feature::kSVE2) && ReadSVEVectorLengthInBits() >= 256;
|
||||
HostFeatures.SupportsSVE128 = Features.Supports(CPUFeatures::Feature::SVE2);
|
||||
HostFeatures.SupportsSVE256 = Features.Supports(CPUFeatures::Feature::SVE2) && ReadSVEVectorLengthInBits() >= 256;
|
||||
#endif
|
||||
HostFeatures.SupportsAVX = true;
|
||||
|
||||
@@ -184,14 +504,6 @@ FEXCore::HostFeatures FetchHostFeatures() {
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
__asm volatile("mrs %[ctr], ctr_el0" : [ctr] "=r"(CTR));
|
||||
|
||||
HostFeatures.DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
HostFeatures.ICacheLineSize = 4 << (CTR & 0xF);
|
||||
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps = (1U << 8) | // Invalid Operation float exception trap enable
|
||||
@@ -211,7 +523,6 @@ FEXCore::HostFeatures FetchHostFeatures() {
|
||||
SetFPCR(OriginalFPCR);
|
||||
|
||||
if (HostFeatures.SupportsRAND) {
|
||||
const auto MIDR = GetMIDR();
|
||||
constexpr uint32_t Implementer_QCOM = 0x51;
|
||||
constexpr uint32_t PartNum_Oryon1 = 0x001;
|
||||
const uint32_t MIDR_Implementer = (MIDR >> 24) & 0xFF;
|
||||
@@ -247,12 +558,14 @@ FEXCore::HostFeatures FetchHostFeatures() {
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(_M_X86_64)
|
||||
// Hardcoded cacheline size.
|
||||
HostFeatures.DCacheLineSize = 64U;
|
||||
HostFeatures.ICacheLineSize = 64U;
|
||||
if (CTR) {
|
||||
HostFeatures.DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
HostFeatures.ICacheLineSize = 4 << (CTR & 0xF);
|
||||
} else {
|
||||
HostFeatures.DCacheLineSize = HostFeatures.ICacheLineSize = 64;
|
||||
}
|
||||
|
||||
#if !defined(VIXL_SIMULATOR)
|
||||
#if defined(_M_X86_64) && !defined(VIXL_SIMULATOR)
|
||||
Xbyak::util::Cpu X86Features {};
|
||||
HostFeatures.SupportsAES = X86Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
HostFeatures.SupportsCRC = X86Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
@@ -277,7 +590,6 @@ FEXCore::HostFeatures FetchHostFeatures() {
|
||||
|
||||
HostFeatures.SupportsAFP = true;
|
||||
HostFeatures.SupportsFloatExceptions = true;
|
||||
#endif
|
||||
#endif
|
||||
HostFeatures.SupportsPreserveAllABI = FEX_HAS_PRESERVE_ALL_ATTR;
|
||||
|
||||
@@ -294,4 +606,30 @@ FEXCore::HostFeatures FetchHostFeatures() {
|
||||
OverrideFeatures(&HostFeatures, ForceSVEWidth());
|
||||
return HostFeatures;
|
||||
}
|
||||
|
||||
FEXCore::HostFeatures FetchHostFeatures() {
|
||||
#ifdef _M_X86_64
|
||||
CPUFeatures Features = CPUFeaturesAll {};
|
||||
|
||||
// Vixl simulator doesn't support AFP.
|
||||
Features.RemoveFeature(CPUFeatures::Feature::AFP);
|
||||
// Vixl simulator doesn't support RPRES.
|
||||
Features.RemoveFeature(CPUFeatures::Feature::RPRES);
|
||||
#else
|
||||
CPUFeatures Features = GetCPUFeaturesFromIDRegisters();
|
||||
#endif
|
||||
|
||||
uint64_t CTR = 0;
|
||||
uint64_t MIDR = 0;
|
||||
#ifdef _M_ARM_64
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
__asm volatile("mrs %[ctr], ctr_el0" : [ctr] "=r"(CTR));
|
||||
__asm volatile("mrs %[midr], midr_el1" : [midr] "=r"(MIDR));
|
||||
#endif
|
||||
|
||||
auto HostFeatures = FetchHostFeatures(Features, true, CTR, MIDR);
|
||||
FillMIDRInformationViaLinux(&HostFeatures);
|
||||
return HostFeatures;
|
||||
}
|
||||
} // namespace FEX
|
||||
@@ -1,7 +1,630 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
#include <cstddef>
|
||||
|
||||
namespace FEX {
|
||||
class CPUFeatures {
|
||||
public:
|
||||
enum class Feature : uint32_t {
|
||||
// ISAR0
|
||||
AES,
|
||||
PMULL,
|
||||
SHA1,
|
||||
SHA2,
|
||||
SHA512,
|
||||
CRC32,
|
||||
LSE,
|
||||
LSE128,
|
||||
TME,
|
||||
RDM,
|
||||
SHA3,
|
||||
SM3,
|
||||
SM4,
|
||||
DotProd,
|
||||
FlagM,
|
||||
FlagM2,
|
||||
RNDR,
|
||||
// PFR0
|
||||
FP,
|
||||
FP16,
|
||||
ASIMD,
|
||||
ASIMD16,
|
||||
RAS,
|
||||
SVE,
|
||||
DIT,
|
||||
CSV2,
|
||||
CSV3,
|
||||
// PFR1
|
||||
BTI,
|
||||
SSBS,
|
||||
SSBS2,
|
||||
MTE,
|
||||
MTE2,
|
||||
MTE3,
|
||||
SME,
|
||||
SME2,
|
||||
// ISAR1
|
||||
DPB,
|
||||
DPB2,
|
||||
JSCVT,
|
||||
FCMA,
|
||||
LRCPC,
|
||||
LRCPC2,
|
||||
LRCPC3,
|
||||
FRINTTS,
|
||||
SB,
|
||||
SPECRES,
|
||||
SPECRES2,
|
||||
BF16,
|
||||
SME_F64F64,
|
||||
I8MM,
|
||||
XS,
|
||||
LS64,
|
||||
LS64_V,
|
||||
LS64_ACCDATA,
|
||||
// MMFR0
|
||||
ECV,
|
||||
// MMFR2
|
||||
LSE2,
|
||||
// ZFR0
|
||||
SVE2,
|
||||
SVE2_1,
|
||||
SVE_AES,
|
||||
SVE_PMULL128,
|
||||
SVE_BitPerm,
|
||||
SVE_BF16,
|
||||
SVE_B16B16,
|
||||
SVE_SHA3,
|
||||
SVE_SM4,
|
||||
SVE_I8MM,
|
||||
SVE_F32MM,
|
||||
SVE_F64MM,
|
||||
// MMFR1
|
||||
AFP,
|
||||
// ISAR2
|
||||
WFxt,
|
||||
RPRES,
|
||||
PACQARMA3,
|
||||
MOPS,
|
||||
HBC,
|
||||
CLRBHB,
|
||||
SYSREG128,
|
||||
SYSINSTR128,
|
||||
PRFMSLC,
|
||||
RPRFM,
|
||||
CSSC,
|
||||
// Max indicator
|
||||
MAX,
|
||||
};
|
||||
|
||||
static_assert(FEXCore::ToUnderlying(Feature::MAX) < 128);
|
||||
static_assert((FEXCore::ToUnderlying(Feature::MAX) / (sizeof(uint64_t) * 8)) == 1);
|
||||
|
||||
bool Supports(Feature feat) const {
|
||||
const size_t DWordSelect = FEXCore::ToUnderlying(feat) / (sizeof(uint64_t) * 8);
|
||||
const size_t BitSelect = FEXCore::ToUnderlying(feat) - (DWordSelect * (sizeof(uint64_t) * 8));
|
||||
return (FeatureBits[DWordSelect] >> BitSelect) & 1;
|
||||
}
|
||||
|
||||
void RemoveFeature(Feature feat) {
|
||||
const size_t DWordSelect = FEXCore::ToUnderlying(feat) / (sizeof(uint64_t) * 8);
|
||||
const size_t BitSelect = FEXCore::ToUnderlying(feat) - (DWordSelect * (sizeof(uint64_t) * 8));
|
||||
FeatureBits[DWordSelect] &= ~(1ULL << BitSelect);
|
||||
}
|
||||
|
||||
protected:
|
||||
void FillFeatureFlags();
|
||||
|
||||
// This list is informed by Linux kernel's `Documentation/arch/arm64/cpu-feature-registers.rst`
|
||||
enum class FeatureRegType {
|
||||
ISAR0_EL1,
|
||||
PFR0_EL1,
|
||||
PFR1_EL1,
|
||||
MIDR_EL1,
|
||||
ISAR1_EL1,
|
||||
MMFR0_EL1,
|
||||
MMFR2_EL1,
|
||||
ZFR0_EL1,
|
||||
MMFR1_EL1,
|
||||
ISAR2_EL1,
|
||||
};
|
||||
|
||||
class FeatureReg {
|
||||
public:
|
||||
void SetReg(uint64_t _Reg) {
|
||||
Reg = _Reg;
|
||||
}
|
||||
|
||||
protected:
|
||||
// All feature flag fields are 4-bits.
|
||||
uint64_t GetField(uint64_t Offset) const {
|
||||
return (Reg >> Offset) & 0b1111;
|
||||
}
|
||||
uint64_t Reg {};
|
||||
};
|
||||
|
||||
#define FIELD_FETCHER(feature, field, minimum_field) \
|
||||
bool Supports##feature() const { \
|
||||
return GetField(field) >= minimum_field; \
|
||||
}
|
||||
class ISAR0Reg final : public FeatureReg {
|
||||
public:
|
||||
FIELD_FETCHER(AES, AES, 0b0001);
|
||||
FIELD_FETCHER(PMULL, AES, 0b0010);
|
||||
|
||||
FIELD_FETCHER(SHA1, SHA1, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SHA2, SHA2, 0b0001);
|
||||
FIELD_FETCHER(SHA512, SHA2, 0b0010);
|
||||
|
||||
FIELD_FETCHER(CRC32, CRC32, 0b0001);
|
||||
|
||||
FIELD_FETCHER(LSE, Atomic, 0b0010);
|
||||
FIELD_FETCHER(LSE128, Atomic, 0b0011);
|
||||
|
||||
FIELD_FETCHER(TME, TME, 0b0001);
|
||||
|
||||
FIELD_FETCHER(RDM, RDM, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SHA3, SHA3, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SM3, SM3, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SM4, SM4, 0b0001);
|
||||
|
||||
FIELD_FETCHER(DotProd, DP, 0b0001);
|
||||
|
||||
FIELD_FETCHER(FHM, FHM, 0b0001);
|
||||
|
||||
FIELD_FETCHER(FlagM, TS, 0b0001);
|
||||
FIELD_FETCHER(FlagM2, TS, 0b0010);
|
||||
|
||||
FIELD_FETCHER(TLBIOS, TLB, 0b0001);
|
||||
FIELD_FETCHER(TLBIRANGE, TLB, 0b0010);
|
||||
|
||||
FIELD_FETCHER(RNDR, RNDR, 0b0001);
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
RES0 = 0 * 4,
|
||||
AES = 1 * 4,
|
||||
SHA1 = 2 * 4,
|
||||
SHA2 = 3 * 4,
|
||||
CRC32 = 4 * 4,
|
||||
Atomic = 5 * 4,
|
||||
TME = 6 * 4,
|
||||
RDM = 7 * 4,
|
||||
SHA3 = 8 * 4,
|
||||
SM3 = 9 * 4,
|
||||
SM4 = 10 * 4,
|
||||
DP = 11 * 4,
|
||||
FHM = 12 * 4,
|
||||
TS = 13 * 4,
|
||||
TLB = 14 * 4,
|
||||
RNDR = 15 * 4,
|
||||
};
|
||||
};
|
||||
|
||||
class PFR0Reg final : public FeatureReg {
|
||||
public:
|
||||
FIELD_FETCHER(AA64_EL0, EL0, 0b0001);
|
||||
FIELD_FETCHER(AA32_EL0, EL0, 0b0010);
|
||||
|
||||
FIELD_FETCHER(AA64_EL1, EL1, 0b0001);
|
||||
FIELD_FETCHER(AA32_EL1, EL1, 0b0010);
|
||||
|
||||
FIELD_FETCHER(AA64_EL2, EL2, 0b0001);
|
||||
FIELD_FETCHER(AA32_EL2, EL2, 0b0010);
|
||||
|
||||
FIELD_FETCHER(AA64_EL3, EL3, 0b0001);
|
||||
FIELD_FETCHER(AA32_EL3, EL3, 0b0010);
|
||||
|
||||
bool SupportsFP() const {
|
||||
return GetField(FP) != 0b1111;
|
||||
}
|
||||
FIELD_FETCHER(HP, FP, 0b0001);
|
||||
|
||||
bool SupportsAdvSIMD() const {
|
||||
return GetField(AdvSIMD) != 0b1111;
|
||||
}
|
||||
FIELD_FETCHER(ASIMDHP, AdvSIMD, 0b0001);
|
||||
|
||||
FIELD_FETCHER(GIC4_0, GIC, 0b0001);
|
||||
FIELD_FETCHER(GIC4_1, GIC, 0b0011);
|
||||
|
||||
FIELD_FETCHER(RAS, RAS, 0b0001);
|
||||
FIELD_FETCHER(RAS1_1, RAS, 0b0010);
|
||||
FIELD_FETCHER(RAS2, RAS, 0b0011);
|
||||
|
||||
FIELD_FETCHER(SVE, SVE, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SEL2, SEL2, 0b0001);
|
||||
|
||||
uint64_t MPAM_Major() const {
|
||||
return GetField(MPAM);
|
||||
}
|
||||
|
||||
FIELD_FETCHER(AMU1, AMU, 0b0001);
|
||||
FIELD_FETCHER(AMU1_1, AMU, 0b0010);
|
||||
|
||||
FIELD_FETCHER(DIT, DIT, 0b0001);
|
||||
|
||||
FIELD_FETCHER(RME, RME, 0b0001);
|
||||
|
||||
FIELD_FETCHER(CSV2, CSV2, 0b0001);
|
||||
FIELD_FETCHER(CSV2_2, CSV2, 0b0010);
|
||||
FIELD_FETCHER(CSV2_3, CSV2, 0b0011);
|
||||
|
||||
FIELD_FETCHER(CSV3, CSV3, 0b0001);
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
EL0 = 0 * 4,
|
||||
EL1 = 1 * 4,
|
||||
EL2 = 2 * 4,
|
||||
EL3 = 3 * 4,
|
||||
FP = 4 * 4,
|
||||
AdvSIMD = 5 * 4,
|
||||
GIC = 6 * 4,
|
||||
RAS = 7 * 4,
|
||||
SVE = 8 * 4,
|
||||
SEL2 = 9 * 4,
|
||||
MPAM = 10 * 4,
|
||||
AMU = 11 * 4,
|
||||
DIT = 12 * 4,
|
||||
RME = 13 * 4,
|
||||
CSV2 = 14 * 4,
|
||||
CSV3 = 15 * 4,
|
||||
};
|
||||
};
|
||||
|
||||
class PFR1Reg final : public FeatureReg {
|
||||
public:
|
||||
FIELD_FETCHER(BTI, BT, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SSBS, SSBS, 0b0001);
|
||||
FIELD_FETCHER(SSBS2, SSBS, 0b0010);
|
||||
|
||||
FIELD_FETCHER(MTE, MTE, 0b0001);
|
||||
FIELD_FETCHER(MTE2, MTE, 0b0010);
|
||||
FIELD_FETCHER(MTE3, MTE, 0b0011);
|
||||
|
||||
uint64_t RAS_Minor() const {
|
||||
return GetField(RAS_frac);
|
||||
}
|
||||
uint64_t MPAM_Minor() const {
|
||||
return GetField(MPAM_frac);
|
||||
}
|
||||
|
||||
FIELD_FETCHER(SME, SME, 0b0001);
|
||||
FIELD_FETCHER(SME2, SME, 0b0010);
|
||||
|
||||
FIELD_FETCHER(RNDR_trap, RNDR_trap, 0b0001);
|
||||
|
||||
uint64_t CSV2_Minor() const {
|
||||
return GetField(CSV2_frac);
|
||||
}
|
||||
|
||||
FIELD_FETCHER(NMI, NMI, 0b0001);
|
||||
|
||||
uint64_t MTE_Minor() const {
|
||||
return GetField(MTE_frac);
|
||||
}
|
||||
|
||||
FIELD_FETCHER(GCS, GCS, 0b0001);
|
||||
|
||||
FIELD_FETCHER(THE, THE, 0b0001);
|
||||
|
||||
FIELD_FETCHER(MTEX, MTEX, 0b0001);
|
||||
|
||||
FIELD_FETCHER(DoubleFault2, DF2, 0b0001);
|
||||
|
||||
FIELD_FETCHER(PFAR, PFAR, 0b0001);
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
BT = 0 * 4,
|
||||
SSBS = 1 * 4,
|
||||
MTE = 2 * 4,
|
||||
RAS_frac = 3 * 4,
|
||||
MPAM_frac = 4 * 4,
|
||||
RES0 = 5 * 4,
|
||||
SME = 6 * 4,
|
||||
RNDR_trap = 7 * 4,
|
||||
CSV2_frac = 8 * 4,
|
||||
NMI = 9 * 4,
|
||||
MTE_frac = 10 * 4,
|
||||
GCS = 11 * 4,
|
||||
THE = 12 * 4,
|
||||
MTEX = 13 * 4,
|
||||
DF2 = 14 * 4,
|
||||
PFAR = 15 * 4,
|
||||
};
|
||||
};
|
||||
|
||||
class MIDRReg final : public FeatureReg {
|
||||
public:
|
||||
uint64_t GetRevision() const {
|
||||
return GetField(Revision);
|
||||
}
|
||||
uint64_t GetPartNum() const {
|
||||
return (Reg >> 4) & 0xFFF;
|
||||
}
|
||||
uint64_t GetArchitecture() const {
|
||||
return GetField(Architecture);
|
||||
}
|
||||
uint64_t GetVariant() const {
|
||||
return GetField(Variant);
|
||||
}
|
||||
uint64_t GetImplementer() const {
|
||||
return (Reg >> 24) & 0xFFFF;
|
||||
}
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
Revision = 0 * 4,
|
||||
// Partnum is 3 fields [15:4]
|
||||
Architecture = 4 * 4,
|
||||
Variant = 5 * 4,
|
||||
// Implementer is 2 fields [31:24]
|
||||
// Upper 32-bits is entirely reserved
|
||||
};
|
||||
};
|
||||
|
||||
class ISAR1Reg final : public FeatureReg {
|
||||
public:
|
||||
FIELD_FETCHER(DPB, DPB, 0b0001);
|
||||
FIELD_FETCHER(DPB2, DPB, 0b0010);
|
||||
|
||||
// Ignoring APA and API
|
||||
|
||||
FIELD_FETCHER(JSCVT, JSCVT, 0b0001);
|
||||
|
||||
FIELD_FETCHER(FCMA, FCMA, 0b0001);
|
||||
|
||||
FIELD_FETCHER(LRCPC, LRCPC, 0b0001);
|
||||
FIELD_FETCHER(LRCPC2, LRCPC, 0b0010);
|
||||
FIELD_FETCHER(LRCPC3, LRCPC, 0b0011);
|
||||
|
||||
// Ignoring GPA and GPI
|
||||
|
||||
FIELD_FETCHER(FRINTTS, FRINTTS, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SB, SB, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SPECRES, SPECRES, 0b0001);
|
||||
FIELD_FETCHER(SPECRES2, SPECRES, 0b0010);
|
||||
|
||||
FIELD_FETCHER(BF16, BF16, 0b0001);
|
||||
FIELD_FETCHER(SME_F64F64, BF16, 0b0010);
|
||||
|
||||
FIELD_FETCHER(DGH, DGH, 0b0001);
|
||||
|
||||
FIELD_FETCHER(I8MM, I8MM, 0b0001);
|
||||
|
||||
FIELD_FETCHER(XS, XS, 0b0001);
|
||||
|
||||
FIELD_FETCHER(LS64, LS64, 0b0001);
|
||||
FIELD_FETCHER(LS64_V, LS64, 0b0010);
|
||||
FIELD_FETCHER(LS64_ACCDATA, LS64, 0b0011);
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
DPB = 0 * 4,
|
||||
APA = 1 * 4,
|
||||
API = 2 * 4,
|
||||
JSCVT = 3 * 4,
|
||||
FCMA = 4 * 4,
|
||||
LRCPC = 5 * 4,
|
||||
GPA = 6 * 4,
|
||||
GPI = 7 * 4,
|
||||
FRINTTS = 8 * 4,
|
||||
SB = 9 * 4,
|
||||
SPECRES = 10 * 4,
|
||||
BF16 = 11 * 4,
|
||||
DGH = 12 * 4,
|
||||
I8MM = 13 * 4,
|
||||
XS = 14 * 4,
|
||||
LS64 = 15 * 4,
|
||||
};
|
||||
};
|
||||
|
||||
class MMFR0Reg final : public FeatureReg {
|
||||
public:
|
||||
FIELD_FETCHER(ECV, ECV, 0b0010);
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
PARange = 0 * 4,
|
||||
ASIDBits = 1 * 4,
|
||||
BigEnd = 2 * 4,
|
||||
SNSMem = 3 * 4,
|
||||
BigEndEL0 = 4 * 4,
|
||||
TGran16 = 5 * 4,
|
||||
TGran64 = 6 * 4,
|
||||
TGran4 = 7 * 4,
|
||||
TGran16_2 = 8 * 4,
|
||||
TGran64_2 = 9 * 4,
|
||||
TGran4_2 = 10 * 4,
|
||||
ExS = 11 * 4,
|
||||
RES0 = 12 * 4,
|
||||
RES1 = 13 * 4,
|
||||
FGT = 14 * 4,
|
||||
ECV = 15 * 4,
|
||||
};
|
||||
};
|
||||
|
||||
class MMFR2Reg final : public FeatureReg {
|
||||
public:
|
||||
FIELD_FETCHER(LSE2, AT, 0b0001);
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
CnP = 0 * 4,
|
||||
UAO = 1 * 4,
|
||||
LSM = 2 * 4,
|
||||
IESB = 3 * 4,
|
||||
VARange = 4 * 4,
|
||||
CCIDX = 5 * 4,
|
||||
NV = 6 * 4,
|
||||
ST = 7 * 4,
|
||||
AT = 8 * 4,
|
||||
IDS = 9 * 4,
|
||||
FWB = 10 * 4,
|
||||
RES0 = 11 * 4,
|
||||
TTL = 12 * 4,
|
||||
BBM = 13 * 4,
|
||||
EVT = 14 * 4,
|
||||
E0PD = 15 * 4,
|
||||
};
|
||||
};
|
||||
|
||||
class ZFR0Reg final : public FeatureReg {
|
||||
public:
|
||||
FIELD_FETCHER(SVE2, SVEver, 0b0001);
|
||||
FIELD_FETCHER(SVE2_1, SVEver, 0b0010);
|
||||
|
||||
FIELD_FETCHER(SVE_AES, AES, 0b0001);
|
||||
FIELD_FETCHER(SVE_PMULL128, AES, 0b0010);
|
||||
|
||||
FIELD_FETCHER(SVE_BitPerm, BitPerm, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SVE_BF16, BF16, 0b0001);
|
||||
FIELD_FETCHER(SME_F64F64, BF16, 0b0010);
|
||||
|
||||
FIELD_FETCHER(SVE_B16B16, B16B16, 0b0010);
|
||||
|
||||
FIELD_FETCHER(SVE_SHA3, SHA3, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SVE_SM4, SM4, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SVE_I8MM, I8MM, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SVE_F32MM, F32MM, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SVE_F64MM, F64MM, 0b0001);
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
SVEver = 0 * 4,
|
||||
AES = 1 * 4,
|
||||
RES0 = 2 * 4,
|
||||
RES1 = 3 * 4,
|
||||
BitPerm = 4 * 4,
|
||||
BF16 = 5 * 4,
|
||||
B16B16 = 6 * 4,
|
||||
RES2 = 7 * 4,
|
||||
SHA3 = 8 * 4,
|
||||
RES3 = 9 * 4,
|
||||
SM4 = 10 * 4,
|
||||
I8MM = 11 * 4,
|
||||
RES4 = 12 * 4,
|
||||
F32MM = 13 * 4,
|
||||
F64MM = 14 * 4,
|
||||
RES5 = 15 * 4,
|
||||
};
|
||||
};
|
||||
|
||||
class MMFR1Reg final : public FeatureReg {
|
||||
public:
|
||||
FIELD_FETCHER(AFP, AFP, 0b0001);
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
HAFDBS = 0 * 4,
|
||||
VMIDBits = 1 * 4,
|
||||
VH = 2 * 4,
|
||||
HPDS = 3 * 4,
|
||||
LO = 4 * 4,
|
||||
PAN = 5 * 4,
|
||||
SpecSEI = 6 * 4,
|
||||
XNX = 7 * 4,
|
||||
TWED = 8 * 4,
|
||||
ETS = 9 * 4,
|
||||
HCX = 10 * 4,
|
||||
AFP = 11 * 4,
|
||||
nTLBPA = 12 * 4,
|
||||
TIDCP1 = 13 * 4,
|
||||
CMOW = 14 * 4,
|
||||
ECBHB = 15 * 4,
|
||||
};
|
||||
};
|
||||
|
||||
class ISAR2Reg final : public FeatureReg {
|
||||
public:
|
||||
FIELD_FETCHER(WFxt, WFxt, 0b0010);
|
||||
|
||||
FIELD_FETCHER(RPRES, RPRES, 0b0001);
|
||||
|
||||
FIELD_FETCHER(PACQARMA3, GPA3, 0b0001);
|
||||
|
||||
FIELD_FETCHER(MOPS, MOPS, 0b0001);
|
||||
|
||||
FIELD_FETCHER(HBC, BC, 0b0001);
|
||||
|
||||
uint64_t PAC_Minor() const {
|
||||
return GetField(PAC_frac);
|
||||
}
|
||||
|
||||
FIELD_FETCHER(CLRBHB, CLRBHB, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SYSREG128, SYSREG_128, 0b0001);
|
||||
|
||||
FIELD_FETCHER(SYSINSTR128, SYSINSTR_128, 0b0001);
|
||||
|
||||
FIELD_FETCHER(PRFMSLC, PRFMSLC, 0b0001);
|
||||
|
||||
FIELD_FETCHER(RPRFM, RPRFM, 0b0001);
|
||||
|
||||
FIELD_FETCHER(CSSC, CSSC, 0b0001);
|
||||
|
||||
private:
|
||||
enum Field {
|
||||
WFxt = 0 * 4,
|
||||
RPRES = 1 * 4,
|
||||
GPA3 = 2 * 4,
|
||||
APA3 = 3 * 4,
|
||||
MOPS = 4 * 4,
|
||||
BC = 5 * 4,
|
||||
PAC_frac = 6 * 4,
|
||||
CLRBHB = 7 * 4,
|
||||
SYSREG_128 = 8 * 4,
|
||||
SYSINSTR_128 = 9 * 4,
|
||||
PRFMSLC = 10 * 4,
|
||||
RES0 = 11 * 4,
|
||||
RPRFM = 12 * 4,
|
||||
CSSC = 13 * 4,
|
||||
RES1 = 14 * 4,
|
||||
ATS1A = 15 * 4,
|
||||
};
|
||||
};
|
||||
#undef FIELD_FETCHER
|
||||
|
||||
ISAR0Reg ISAR0;
|
||||
PFR0Reg PFR0;
|
||||
PFR1Reg PFR1;
|
||||
MIDRReg MIDR;
|
||||
ISAR1Reg ISAR1;
|
||||
MMFR0Reg MMFR0;
|
||||
ZFR0Reg ZFR0;
|
||||
MMFR2Reg MMFR2;
|
||||
MMFR1Reg MMFR1;
|
||||
ISAR2Reg ISAR2;
|
||||
|
||||
uint64_t FeatureBits[(FEXCore::ToUnderlying(Feature::MAX) / (sizeof(uint64_t) * 8)) + 1] {};
|
||||
|
||||
void SetFeature(Feature feat) {
|
||||
const size_t DWordSelect = FEXCore::ToUnderlying(feat) / (sizeof(uint64_t) * 8);
|
||||
const size_t BitSelect = FEXCore::ToUnderlying(feat) - (DWordSelect * (sizeof(uint64_t) * 8));
|
||||
FeatureBits[DWordSelect] |= 1ULL << BitSelect;
|
||||
}
|
||||
};
|
||||
|
||||
void FillMIDRInformationViaLinux(FEXCore::HostFeatures* Features);
|
||||
|
||||
FEXCore::HostFeatures FetchHostFeatures(FEX::CPUFeatures& Features, bool SupportsCacheMaintenanceOps, uint64_t CTR, uint64_t MIDR);
|
||||
FEXCore::HostFeatures FetchHostFeatures();
|
||||
}
|
||||
} // namespace FEX
|
||||
@@ -1,10 +1,16 @@
|
||||
add_subdirectory(CommonTools)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
if (BUILD_FEXCONFIG)
|
||||
if (USE_FEXCONFIG_TOOLKIT STREQUAL "imgui")
|
||||
add_subdirectory(FEXConfig/)
|
||||
endif()
|
||||
elseif (USE_FEXCONFIG_TOOLKIT STREQUAL "qt")
|
||||
find_package(Qt6 COMPONENTS Qml Quick Widgets QUIET)
|
||||
if (NOT Qt6_FOUND)
|
||||
find_package(Qt5 COMPONENTS Qml Quick Widgets REQUIRED)
|
||||
endif()
|
||||
|
||||
add_subdirectory(FEXQonfig/)
|
||||
endif()
|
||||
|
||||
if (ENABLE_GDB_SYMBOLS)
|
||||
add_subdirectory(FEXGDBReader/)
|
||||
|
||||
@@ -22,7 +22,7 @@ else()
|
||||
target_link_libraries(${NAME} PRIVATE ${SDL2_LIBRARIES})
|
||||
endif()
|
||||
|
||||
target_link_libraries(${NAME} PRIVATE Common pthread epoxy X11 EGL imgui json-maker)
|
||||
target_link_libraries(${NAME} PRIVATE Common pthread epoxy X11 EGL imgui)
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
target_link_options(${NAME}
|
||||
|
||||
@@ -556,11 +556,13 @@ void FillHackConfig() {
|
||||
auto Value = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED);
|
||||
auto VectorTSO = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED);
|
||||
auto MemcpyTSO = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED);
|
||||
auto StrictSplitLockAtomics = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_STRICTINPROCESSSPLITLOCKS);
|
||||
auto HalfBarrierTSO = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_HALFBARRIERTSOENABLED);
|
||||
|
||||
bool TSOEnabled = Value.has_value() && **Value == "1";
|
||||
bool VectorTSOEnabled = VectorTSO.has_value() && **VectorTSO == "1";
|
||||
bool MemcpyTSOEnabled = MemcpyTSO.has_value() && **MemcpyTSO == "1";
|
||||
bool StrictSplitLockAtomicsEnabled = StrictSplitLockAtomics.has_value() && **StrictSplitLockAtomics == "1";
|
||||
bool HalfBarrierTSOEnabled = HalfBarrierTSO.has_value() && **HalfBarrierTSO == "1";
|
||||
|
||||
if (ImGui::Checkbox("TSO Emulation Enabled", &TSOEnabled)) {
|
||||
@@ -590,6 +592,16 @@ void FillHackConfig() {
|
||||
ImGui::EndTooltip();
|
||||
}
|
||||
|
||||
if (ImGui::Checkbox("Strict in-process split-lock atomics Enabled", &StrictSplitLockAtomicsEnabled)) {
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_STRICTINPROCESSSPLITLOCKS, StrictSplitLockAtomicsEnabled ? "1" : "0");
|
||||
ConfigChanged = true;
|
||||
}
|
||||
if (ImGui::IsItemHovered()) {
|
||||
ImGui::BeginTooltip();
|
||||
ImGui::Text("Disables strict in-process split-lock atomics using a mutex");
|
||||
ImGui::EndTooltip();
|
||||
}
|
||||
|
||||
if (ImGui::Checkbox("Unaligned Half-Barrier TSO Emulation Enabled", &HalfBarrierTSOEnabled)) {
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_HALFBARRIERTSOENABLED, HalfBarrierTSOEnabled ? "1" : "0");
|
||||
ConfigChanged = true;
|
||||
@@ -980,7 +992,7 @@ bool DrawUI() {
|
||||
} // namespace
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
FEX::Config::InitializeConfigs();
|
||||
FEX::Config::InitializeConfigs(FEX::Config::PortableInformation {});
|
||||
|
||||
fextl::string ImGUIConfig = FEXCore::Config::GetConfigDirectory(false) + "FEXConfig_imgui.ini";
|
||||
auto [window, gl_context] = FEX::GUI::SetupIMGui("#FEXConfig", ImGUIConfig);
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
include(GNUInstallDirs)
|
||||
set(NAME FEXGDBReader)
|
||||
set(SRCS FEXGDBReader.cpp)
|
||||
|
||||
@@ -5,7 +6,7 @@ add_library(${NAME} SHARED ${SRCS})
|
||||
|
||||
install(TARGETS ${NAME}
|
||||
RUNTIME
|
||||
LIBRARY DESTINATION lib/gdb
|
||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}/gdb
|
||||
COMPONENT LIBRARY)
|
||||
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
@@ -107,12 +107,12 @@ TSOEmulationFacts GetTSOEmulationFacts() {
|
||||
} // namespace
|
||||
|
||||
int main(int argc, char** argv, char** envp) {
|
||||
FEX::Config::InitializeConfigs(FEX::Config::PortableInformation {});
|
||||
FEXCore::Config::Initialize();
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateGlobalMainLayer());
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateMainLayer());
|
||||
// No FEX arguments passed through command line
|
||||
FEXCore::Config::AddLayer(FEX::Config::CreateEnvironmentLayer(envp));
|
||||
FEXCore::Config::Load();
|
||||
|
||||
// Load the arguments
|
||||
optparse::OptionParser Parser = optparse::OptionParser().description("Simple application to get a couple of FEX options");
|
||||
@@ -145,6 +145,8 @@ int main(int argc, char** argv, char** envp) {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Config::Load();
|
||||
|
||||
// Reload the meta layer
|
||||
FEXCore::Config::ReloadMetaLayer();
|
||||
|
||||
@@ -212,8 +214,9 @@ int main(int argc, char** argv, char** envp) {
|
||||
fprintf(stdout, "\tMemory atomics emulation method: %s\n", TSOFacts.LSE ? "\e[32mLSE\e[0m" : "\e[31mLL/SC\e[0m");
|
||||
fprintf(stdout, "\tUnaligned atomic memory granularity: %s\n", TSOFacts.LSE2 ? "\e[32m16-byte\e[0m" : "\e[31mNatural alignment\e[0m");
|
||||
///< TODO: Once TME is supported by hardware this can change.
|
||||
fprintf(stdout, "\tUnaligned memory atomic emulation: %s\n", TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m");
|
||||
fprintf(stdout, "\tUnaligned memory loadstore emulation: %s\n", UnalignedMemoryLoadStoreTSOEmulation);
|
||||
fprintf(stdout, "\t16-Byte split-lock atomic emulation: %s\n", TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m");
|
||||
fprintf(stdout, "\t64-Byte split-lock atomic emulation: %s\n", TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m");
|
||||
fprintf(stdout, "\tGPR memory model emulation: %s\n", GPRMemoryTSOEmulation);
|
||||
fprintf(stdout, "\tMemcpy memory model emulation: %s\n", MemcpyMemoryTSOEmulation);
|
||||
fprintf(stdout, "\tVector memory model emulation: %s\n", VectorMemoryTSOEmulation);
|
||||
@@ -222,12 +225,16 @@ int main(int argc, char** argv, char** envp) {
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
fprintf(stderr, "Strict: %d\n", StrictInProcessSplitLocks());
|
||||
|
||||
fprintf(stdout, "\nConfiguration:\n");
|
||||
fprintf(stdout, "\tTSO Emulation: %s\n", TSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\tMemcpy TSO Emulation: %s\n", TSOEnabled() && MemcpySetTSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\tVector TSO Emulation: %s\n", TSOEnabled() && VectorTSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\tHalf-barrier unaligned TSO emulation: %s\n", TSOEnabled() && HalfBarrierTSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\t16-Byte strict split-lock emulation: %s\n", StrictInProcessSplitLocks() ? "In-process mutex" : "Tearing");
|
||||
fprintf(stdout, "\t64-Byte strict split-lock emulation: %s\n", StrictInProcessSplitLocks() ? "In-process mutex" : "Tearing");
|
||||
}
|
||||
|
||||
return 0;
|
||||
|
||||
@@ -50,103 +50,108 @@ endfunction()
|
||||
GenerateInterpreter(FEXLoader 0)
|
||||
GenerateInterpreter(FEXInterpreter 1)
|
||||
|
||||
install(PROGRAMS "${PROJECT_SOURCE_DIR}/Scripts/FEXUpdateAOTIRCache.sh" DESTINATION bin RENAME FEXUpdateAOTIRCache)
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
# Check for conflicting binfmt before installing
|
||||
set (CONFLICTING_BINFMTS_32
|
||||
${CMAKE_INSTALL_PREFIX}/share/binfmts/qemu-i386
|
||||
${CMAKE_INSTALL_PREFIX}/share/binfmts/box86)
|
||||
set (CONFLICTING_BINFMTS_64
|
||||
${CMAKE_INSTALL_PREFIX}/share/binfmts/qemu-x86_64
|
||||
${CMAKE_INSTALL_PREFIX}/share/binfmts/box64)
|
||||
|
||||
find_program(UPDATE_BINFMTS_PROGRAM update-binfmts)
|
||||
if (UPDATE_BINFMTS_PROGRAM)
|
||||
add_custom_target(binfmt_misc_32
|
||||
echo "Attempting to install FEX-x86 misc now."
|
||||
COMMAND "${CMAKE_SOURCE_DIR}/Scripts/CheckBinfmtNotInstall.sh" ${CONFLICTING_BINFMTS_32}
|
||||
COMMAND "update-binfmts" "--importdir=${CMAKE_INSTALL_PREFIX}/share/binfmts/" "--import" "FEX-x86"
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "binfmt_misc FEX-x86 installed"
|
||||
if (NOT USE_LEGACY_BINFMTMISC)
|
||||
# Just restart the systemd service
|
||||
add_custom_target(binfmt_misc
|
||||
echo "Restarting systemd service now."
|
||||
COMMAND "service" "systemd-binfmt" "restart"
|
||||
)
|
||||
|
||||
add_custom_target(binfmt_misc_64
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "Attempting to install FEX-x86_64 misc now."
|
||||
COMMAND "${CMAKE_SOURCE_DIR}/Scripts/CheckBinfmtNotInstall.sh" ${CONFLICTING_BINFMTS_64}
|
||||
COMMAND "update-binfmts" "--importdir=${CMAKE_INSTALL_PREFIX}/share/binfmts/" "--import" "FEX-x86_64"
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "binfmt_misc FEX-x86_64 installed"
|
||||
)
|
||||
if(TARGET uninstall)
|
||||
add_custom_target(uninstall_binfmt_misc_32
|
||||
COMMAND update-binfmts --unimport FEX-x86 || (exit 0)
|
||||
)
|
||||
add_custom_target(uninstall_binfmt_misc_64
|
||||
COMMAND update-binfmts --unimport FEX-x86_64 || (exit 0)
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_32)
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_64)
|
||||
endif()
|
||||
else()
|
||||
set (SUPPORTED_BINFMT_MISC_FLAGS "POCF")
|
||||
execute_process(COMMAND uname -r OUTPUT_VARIABLE UNAME_VERSION OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
string(REGEX MATCH "[0-9]+.[0-9]+" KERNEL_VERSION ${UNAME_VERSION})
|
||||
message(STATUS "Kernel version: ${KERNEL_VERSION}")
|
||||
# Check for conflicting binfmt before installing
|
||||
set (CONFLICTING_BINFMTS_32
|
||||
${CMAKE_INSTALL_PREFIX}/share/binfmts/qemu-i386
|
||||
${CMAKE_INSTALL_PREFIX}/share/binfmts/box86)
|
||||
set (CONFLICTING_BINFMTS_64
|
||||
${CMAKE_INSTALL_PREFIX}/share/binfmts/qemu-x86_64
|
||||
${CMAKE_INSTALL_PREFIX}/share/binfmts/box64)
|
||||
|
||||
if (KERNEL_VERSION VERSION_GREATER_EQUAL 9999.0)
|
||||
# New binfmt_misc flag for exposing the interpreter was added in version '9999.0'
|
||||
# Only enable it if the host kernel is at least that.
|
||||
set (SUPPORTED_BINFMT_MISC_FLAGS "POCFI")
|
||||
endif()
|
||||
|
||||
# In the case of update-binfmts not being available (Arch for example) then we need to install manually
|
||||
add_custom_target(binfmt_misc_32
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "Attempting to remove FEX-x86 misc prior to install. Ignore permission denied"
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo -1 > /proc/sys/fs/binfmt_misc/FEX-x86 || (exit 0)
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
find_program(UPDATE_BINFMTS_PROGRAM update-binfmts)
|
||||
if (UPDATE_BINFMTS_PROGRAM)
|
||||
add_custom_target(binfmt_misc_32
|
||||
echo "Attempting to install FEX-x86 misc now."
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo
|
||||
':FEX-x86:M:0:\\x7fELF\\x01\\x01\\x01\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x02\\x00\\x03\\x00:\\xff\\xff\\xff\\xff\\xff\\xfe\\xfe\\x00\\x00\\x00\\x00\\xff\\xff\\xff\\xff\\xff\\xfe\\xff\\xff\\xff:${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter:${SUPPORTED_BINFMT_MISC_FLAGS}' > /proc/sys/fs/binfmt_misc/register
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
COMMAND "${CMAKE_SOURCE_DIR}/Scripts/CheckBinfmtNotInstall.sh" ${CONFLICTING_BINFMTS_32}
|
||||
COMMAND "update-binfmts" "--importdir=${CMAKE_INSTALL_PREFIX}/share/binfmts/" "--import" "FEX-x86"
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "binfmt_misc FEX-x86 installed"
|
||||
)
|
||||
add_custom_target(binfmt_misc_64
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "Attempting to remove FEX-x86_64 misc prior to install. Ignore permission denied"
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo -1 > /proc/sys/fs/binfmt_misc/FEX-x86_64 || (exit 0)
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
|
||||
add_custom_target(binfmt_misc_64
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "Attempting to install FEX-x86_64 misc now."
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo
|
||||
':FEX-x86_64:M:0:\\x7fELF\\x02\\x01\\x01\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x02\\x00\\x3e\\x00:\\xff\\xff\\xff\\xff\\xff\\xfe\\xfe\\x00\\x00\\x00\\x00\\xff\\xff\\xff\\xff\\xff\\xfe\\xff\\xff\\xff:${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter:${SUPPORTED_BINFMT_MISC_FLAGS}' > /proc/sys/fs/binfmt_misc/register
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
COMMAND "${CMAKE_SOURCE_DIR}/Scripts/CheckBinfmtNotInstall.sh" ${CONFLICTING_BINFMTS_64}
|
||||
COMMAND "update-binfmts" "--importdir=${CMAKE_INSTALL_PREFIX}/share/binfmts/" "--import" "FEX-x86_64"
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "binfmt_misc FEX-x86_64 installed"
|
||||
)
|
||||
if(TARGET uninstall)
|
||||
add_custom_target(uninstall_binfmt_misc_32
|
||||
COMMAND update-binfmts --unimport FEX-x86 || (exit 0)
|
||||
)
|
||||
add_custom_target(uninstall_binfmt_misc_64
|
||||
COMMAND update-binfmts --unimport FEX-x86_64 || (exit 0)
|
||||
)
|
||||
|
||||
if(TARGET uninstall)
|
||||
add_custom_target(uninstall_binfmt_misc_32
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_32)
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_64)
|
||||
endif()
|
||||
else()
|
||||
set (SUPPORTED_BINFMT_MISC_FLAGS "POCF")
|
||||
execute_process(COMMAND uname -r OUTPUT_VARIABLE UNAME_VERSION OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
string(REGEX MATCH "[0-9]+.[0-9]+" KERNEL_VERSION ${UNAME_VERSION})
|
||||
message(STATUS "Kernel version: ${KERNEL_VERSION}")
|
||||
|
||||
if (KERNEL_VERSION VERSION_GREATER_EQUAL 9999.0)
|
||||
# New binfmt_misc flag for exposing the interpreter was added in version '9999.0'
|
||||
# Only enable it if the host kernel is at least that.
|
||||
set (SUPPORTED_BINFMT_MISC_FLAGS "POCFI")
|
||||
endif()
|
||||
|
||||
# In the case of update-binfmts not being available (Arch for example) then we need to install manually
|
||||
add_custom_target(binfmt_misc_32
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "Attempting to remove FEX-x86 misc prior to install. Ignore permission denied"
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo -1 > /proc/sys/fs/binfmt_misc/FEX-x86 || (exit 0)
|
||||
)
|
||||
add_custom_target(uninstall_binfmt_misc_64
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "Attempting to install FEX-x86 misc now."
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo
|
||||
':FEX-x86:M:0:\\x7fELF\\x01\\x01\\x01\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x02\\x00\\x03\\x00:\\xff\\xff\\xff\\xff\\xff\\xfe\\xfe\\x00\\x00\\x00\\x00\\xff\\xff\\xff\\xff\\xff\\xfe\\xff\\xff\\xff:${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter:${SUPPORTED_BINFMT_MISC_FLAGS}' > /proc/sys/fs/binfmt_misc/register
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "binfmt_misc FEX-x86 installed"
|
||||
)
|
||||
add_custom_target(binfmt_misc_64
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "Attempting to remove FEX-x86_64 misc prior to install. Ignore permission denied"
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo -1 > /proc/sys/fs/binfmt_misc/FEX-x86_64 || (exit 0)
|
||||
)
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "Attempting to install FEX-x86_64 misc now."
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo
|
||||
':FEX-x86_64:M:0:\\x7fELF\\x02\\x01\\x01\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x00\\x02\\x00\\x3e\\x00:\\xff\\xff\\xff\\xff\\xff\\xfe\\xfe\\x00\\x00\\x00\\x00\\xff\\xff\\xff\\xff\\xff\\xfe\\xff\\xff\\xff:${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter:${SUPPORTED_BINFMT_MISC_FLAGS}' > /proc/sys/fs/binfmt_misc/register
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "binfmt_misc FEX-x86_64 installed"
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_32)
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_64)
|
||||
if(TARGET uninstall)
|
||||
add_custom_target(uninstall_binfmt_misc_32
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo -1 > /proc/sys/fs/binfmt_misc/FEX-x86 || (exit 0)
|
||||
)
|
||||
add_custom_target(uninstall_binfmt_misc_64
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo -1 > /proc/sys/fs/binfmt_misc/FEX-x86_64 || (exit 0)
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_32)
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_64)
|
||||
endif()
|
||||
endif()
|
||||
add_custom_target(binfmt_misc
|
||||
DEPENDS binfmt_misc_32
|
||||
DEPENDS binfmt_misc_64
|
||||
)
|
||||
endif()
|
||||
|
||||
add_custom_target(binfmt_misc
|
||||
DEPENDS binfmt_misc_32
|
||||
DEPENDS binfmt_misc_64
|
||||
)
|
||||
endif()
|
||||
@@ -64,7 +64,6 @@ $end_info$
|
||||
namespace {
|
||||
static bool SilentLog;
|
||||
static int OutputFD {-1};
|
||||
static bool ExecutedWithFD {false};
|
||||
|
||||
void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
|
||||
if (SilentLog) {
|
||||
@@ -194,14 +193,55 @@ void RootFSRedirect(fextl::string* Filename, const fextl::string& RootFS) {
|
||||
}
|
||||
}
|
||||
|
||||
bool RanAsInterpreter(const char* Program) {
|
||||
FEX::Config::PortableInformation ReadPortabilityInformation() {
|
||||
const FEX::Config::PortableInformation BadResult {false, {}};
|
||||
const char* PortableConfig = getenv("FEX_PORTABLE");
|
||||
if (!PortableConfig) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
uint32_t Value {};
|
||||
std::string_view PortableView {PortableConfig};
|
||||
|
||||
if (std::from_chars(PortableView.data(), PortableView.data() + PortableView.size(), Value).ec != std::errc {} || Value == 0) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
// Read the FEXInterpreter path from `/proc/self/exe` which is always a symlink to the absolute path of the executable running.
|
||||
// This way we can get the parent path that the application is executing from.
|
||||
char SelfPath[PATH_MAX];
|
||||
auto Result = readlink("/proc/self/exe", SelfPath, PATH_MAX);
|
||||
if (Result == -1) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
std::string_view SelfPathView {SelfPath, std::min<size_t>(PATH_MAX, Result)};
|
||||
|
||||
// Extract the absolute path from the FEXInterpreter path
|
||||
return {true, fextl::string {SelfPathView.substr(0, SelfPathView.find_last_of('/') + 1)}};
|
||||
}
|
||||
|
||||
bool RanAsInterpreter(bool ExecutedWithFD) {
|
||||
return ExecutedWithFD || FEXLOADER_AS_INTERPRETER;
|
||||
}
|
||||
|
||||
bool IsInterpreterInstalled() {
|
||||
// The interpreter is installed if both the binfmt_misc handlers are available
|
||||
// Or if we were originally executed with FD. Which means the interpreter is installed
|
||||
/**
|
||||
* @brief Queries if FEX is installed as a binfmt_misc interpreter
|
||||
*
|
||||
* @param ExecutedWithFD If FEXInterpreter was executed using a binfmt_misc FD handle from the kernel
|
||||
* @param Portable Portability information about FEX being run in portable mode
|
||||
*
|
||||
* @return true if the binfmt_misc handlers are installed and being used
|
||||
*/
|
||||
bool QueryInterpreterInstalled(bool ExecutedWithFD, const FEX::Config::PortableInformation& Portable) {
|
||||
if (Portable.IsPortable) {
|
||||
// Don't use binfmt interpreter even if it's installed
|
||||
return false;
|
||||
}
|
||||
|
||||
// Check if FEX's binfmt_misc handlers are both installed.
|
||||
// The explicit check can be omitted if FEX was executed from an FD,
|
||||
// since this only happens if the kernel launched FEX through binfmt_misc
|
||||
return ExecutedWithFD || (access("/proc/sys/fs/binfmt_misc/FEX-x86", F_OK) == 0 && access("/proc/sys/fs/binfmt_misc/FEX-x86_64", F_OK) == 0);
|
||||
}
|
||||
|
||||
@@ -267,10 +307,14 @@ static int StealFEXFDFromEnv(const char* Env) {
|
||||
int main(int argc, char** argv, char** const envp) {
|
||||
auto SBRKPointer = FEXCore::Allocator::DisableSBRKAllocations();
|
||||
FEXCore::Allocator::GLIBCScopedFault GLIBFaultScope;
|
||||
const bool IsInterpreter = RanAsInterpreter(argv[0]);
|
||||
|
||||
ExecutedWithFD = getauxval(AT_EXECFD) != 0;
|
||||
const bool ExecutedWithFD = getauxval(AT_EXECFD) != 0;
|
||||
const bool IsInterpreter = RanAsInterpreter(ExecutedWithFD);
|
||||
const auto PortableInfo = ReadPortabilityInformation();
|
||||
const bool InterpreterInstalled = QueryInterpreterInstalled(ExecutedWithFD, PortableInfo);
|
||||
|
||||
int FEXFD {StealFEXFDFromEnv("FEX_EXECVEFD")};
|
||||
int FEXSeccompFD {StealFEXFDFromEnv("FEX_SECCOMPFD")};
|
||||
|
||||
LogMan::Throw::InstallHandler(AssertHandler);
|
||||
LogMan::Msg::InstallHandler(MsgHandler);
|
||||
@@ -280,17 +324,18 @@ int main(int argc, char** argv, char** const envp) {
|
||||
argc, argv);
|
||||
auto Args = ArgsLoader->Get();
|
||||
auto ParsedArgs = ArgsLoader->GetParsedArgs();
|
||||
auto Program = FEX::Config::LoadConfig(std::move(ArgsLoader), true, envp, ExecutedWithFD, FEXFD);
|
||||
|
||||
auto Program = FEX::Config::GetApplicationNames(Args, ExecutedWithFD, FEXFD);
|
||||
if (Program.ProgramPath.empty() && FEXFD == -1) {
|
||||
// Early exit if we weren't passed an argument
|
||||
return 0;
|
||||
}
|
||||
|
||||
FEX::Config::LoadConfig(std::move(ArgsLoader), Program.ProgramName, envp, PortableInfo);
|
||||
|
||||
// Reload the meta layer
|
||||
FEXCore::Config::ReloadMetaLayer();
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS_INTERPRETER, IsInterpreter ? "1" : "0");
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_INTERPRETER_INSTALLED, IsInterpreterInstalled() ? "1" : "0");
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_INTERPRETER_INSTALLED, InterpreterInstalled ? "1" : "0");
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// If running under the vixl simulator, ensure that indirect runtime calls are enabled.
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_DISABLE_VIXL_INDIRECT_RUNTIME_CALLS, "0");
|
||||
@@ -467,6 +512,9 @@ int main(int argc, char** argv, char** const envp) {
|
||||
auto SyscallHandler = Loader.Is64BitMode() ? FEX::HLE::x64::CreateHandler(CTX.get(), SignalDelegation.get()) :
|
||||
FEX::HLE::x32::CreateHandler(CTX.get(), SignalDelegation.get(), std::move(Allocator));
|
||||
|
||||
// Load VDSO in to memory prior to mapping our ELFs.
|
||||
auto VDSOMapping = FEX::VDSO::LoadVDSOThunks(Loader.Is64BitMode(), SyscallHandler.get());
|
||||
|
||||
// Now that we have the syscall handler. Track some FDs that are FEX owned.
|
||||
if (OutputFD != -1) {
|
||||
SyscallHandler->FM.TrackFEXFD(OutputFD);
|
||||
@@ -477,9 +525,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
|
||||
{
|
||||
// Load VDSO in to memory prior to mapping our ELFs.
|
||||
void* VDSOBase = FEX::VDSO::LoadVDSOThunks(Loader.Is64BitMode(), SyscallHandler.get());
|
||||
Loader.SetVDSOBase(VDSOBase);
|
||||
Loader.SetVDSOBase(VDSOMapping.VDSOBase);
|
||||
Loader.CalculateHWCaps(CTX.get());
|
||||
|
||||
if (!Loader.MapMemory(SyscallHandler.get())) {
|
||||
@@ -516,6 +562,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
CTX->AppendThunkDefinitions(FEX::VDSO::GetVDSOThunkDefinitions());
|
||||
SignalDelegation->SetVDSOSigReturn();
|
||||
|
||||
SyscallHandler->DeserializeSeccompFD(ParentThread, FEXSeccompFD);
|
||||
|
||||
FEXCore::Context::ExitReason ShutdownReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
|
||||
|
||||
// There might already be an exit handler, leave it installed
|
||||
@@ -591,6 +639,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
SignalDelegation->UninstallTLSState(ParentThread);
|
||||
SyscallHandler->TM.DestroyThread(ParentThread);
|
||||
|
||||
FEX::VDSO::UnloadVDSOMapping(VDSOMapping);
|
||||
|
||||
DebugServer.reset();
|
||||
SyscallHandler.reset();
|
||||
SignalDelegation.reset();
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
set(CMAKE_AUTOMOC ON)
|
||||
|
||||
add_executable(FEXConfig)
|
||||
target_sources(FEXConfig PRIVATE Main.cpp Main.h)
|
||||
target_include_directories(FEXConfig PRIVATE ${CMAKE_SOURCE_DIR}/Source/)
|
||||
target_link_libraries(FEXConfig PRIVATE Common)
|
||||
if (Qt6_FOUND)
|
||||
qt_add_resources(QT_RESOURCES qml6.qrc)
|
||||
target_link_libraries(FEXConfig PRIVATE Qt6::Qml Qt6::Quick Qt6::Widgets)
|
||||
else()
|
||||
qt_add_resources(QT_RESOURCES qml5.qrc)
|
||||
target_link_libraries(FEXConfig PRIVATE Qt5::Qml Qt5::Quick Qt5::Widgets)
|
||||
endif()
|
||||
target_sources(FEXConfig PRIVATE ${QT_RESOURCES})
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
target_link_options(FEXConfig
|
||||
PRIVATE
|
||||
"LINKER:--gc-sections"
|
||||
"LINKER:--strip-all"
|
||||
"LINKER:--as-needed"
|
||||
)
|
||||
endif()
|
||||
|
||||
install(TARGETS FEXConfig
|
||||
RUNTIME
|
||||
DESTINATION bin
|
||||
COMPONENT runtime)
|
||||
@@ -0,0 +1,372 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Main.h"
|
||||
|
||||
#include <Common/Config.h>
|
||||
#include <Common/FileFormatCheck.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <QApplication>
|
||||
#include <QMessageBox>
|
||||
#include <QQmlApplicationEngine>
|
||||
#include <QQuickWindow>
|
||||
|
||||
#include <sys/inotify.h>
|
||||
|
||||
namespace fextl {
|
||||
// Helper to convert a std::filesystem::path to a fextl::string.
|
||||
inline fextl::string string_from_path(const std::filesystem::path& Path) {
|
||||
return Path.string().c_str();
|
||||
}
|
||||
} // namespace fextl
|
||||
|
||||
static fextl::unique_ptr<FEXCore::Config::Layer> LoadedConfig {};
|
||||
static fextl::map<FEXCore::Config::ConfigOption, std::pair<std::string, std::string_view>> ConfigToNameLookup;
|
||||
static fextl::map<std::string, FEXCore::Config::ConfigOption> NameToConfigLookup;
|
||||
|
||||
ConfigModel::ConfigModel() {
|
||||
setItemRoleNames(QHash<int, QByteArray> {{Qt::DisplayRole, "display"}, {Qt::UserRole + 1, "optionType"}, {Qt::UserRole + 2, "optionValue"}});
|
||||
Reload();
|
||||
}
|
||||
|
||||
void ConfigModel::Reload() {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
|
||||
beginResetModel();
|
||||
removeRows(0, rowCount());
|
||||
for (auto& Option : Options) {
|
||||
if (!LoadedConfig->OptionExists(Option.first)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
auto& [Name, TypeId] = ConfigToNameLookup.find(Option.first)->second;
|
||||
auto Item = new QStandardItem(QString::fromStdString(Name));
|
||||
|
||||
const char* OptionType = TypeId.data();
|
||||
Item->setData(OptionType, Qt::UserRole + 1);
|
||||
Item->setData(QString::fromStdString(Option.second.front().c_str()), Qt::UserRole + 2);
|
||||
appendRow(Item);
|
||||
}
|
||||
endResetModel();
|
||||
}
|
||||
|
||||
bool ConfigModel::has(const QString& Name, bool) const {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
return LoadedConfig->OptionExists(NameToConfigLookup.at(Name.toStdString()));
|
||||
}
|
||||
|
||||
void ConfigModel::erase(const QString& Name) {
|
||||
assert(has(Name, false));
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
LoadedConfig->Erase(NameToConfigLookup.at(Name.toStdString()));
|
||||
Reload();
|
||||
}
|
||||
|
||||
bool ConfigModel::getBool(const QString& Name, bool) const {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
|
||||
auto ret = LoadedConfig->Get(NameToConfigLookup.at(Name.toStdString()));
|
||||
if (!ret || !*ret) {
|
||||
throw std::runtime_error("Could not find setting");
|
||||
}
|
||||
return **ret == "1";
|
||||
}
|
||||
|
||||
void ConfigModel::setBool(const QString& Name, bool Value) {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
LoadedConfig->EraseSet(NameToConfigLookup.at(Name.toStdString()), Value ? "1" : "0");
|
||||
Reload();
|
||||
}
|
||||
|
||||
void ConfigModel::setString(const QString& Name, const QString& Value) {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
LoadedConfig->EraseSet(NameToConfigLookup.at(Name.toStdString()), Value.toStdString());
|
||||
Reload();
|
||||
}
|
||||
|
||||
void ConfigModel::setStringList(const QString& Name, const QStringList& Values) {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
|
||||
const auto& Option = NameToConfigLookup.at(Name.toStdString());
|
||||
LoadedConfig->Erase(Option);
|
||||
for (auto& Value : Values) {
|
||||
LoadedConfig->Set(Option, Value.toStdString().c_str());
|
||||
}
|
||||
Reload();
|
||||
}
|
||||
|
||||
void ConfigModel::setInt(const QString& Name, int Value) {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
LoadedConfig->EraseSet(NameToConfigLookup.at(Name.toStdString()), std::to_string(Value));
|
||||
Reload();
|
||||
}
|
||||
|
||||
QString ConfigModel::getString(const QString& Name, bool) const {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
|
||||
auto ret = LoadedConfig->Get(NameToConfigLookup.at(Name.toStdString()));
|
||||
if (!ret || !*ret) {
|
||||
throw std::runtime_error("Could not find setting");
|
||||
}
|
||||
return QString::fromUtf8((*ret)->c_str());
|
||||
}
|
||||
|
||||
QStringList ConfigModel::getStringList(const QString& Name, bool) const {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
|
||||
auto Values = LoadedConfig->All(NameToConfigLookup.at(Name.toStdString()));
|
||||
if (!Values || !*Values) {
|
||||
return {};
|
||||
}
|
||||
QStringList Ret;
|
||||
for (auto& Value : **Values) {
|
||||
Ret.append(Value.c_str());
|
||||
}
|
||||
return Ret;
|
||||
}
|
||||
|
||||
int ConfigModel::getInt(const QString& Name, bool) const {
|
||||
auto Options = LoadedConfig->GetOptionMap();
|
||||
|
||||
auto ret = LoadedConfig->Get(NameToConfigLookup.at(Name.toStdString()));
|
||||
if (!ret || !*ret) {
|
||||
throw std::runtime_error("Could not find setting");
|
||||
}
|
||||
int value;
|
||||
auto res = std::from_chars(&*(*ret)->begin(), &*(*ret)->end(), value);
|
||||
if (res.ptr != &*(*ret)->end()) {
|
||||
throw std::runtime_error("Could not parse integer");
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
static void LoadDefaultSettings() {
|
||||
LoadedConfig = fextl::make_unique<FEX::Config::EmptyMapper>();
|
||||
#define OPT_BASE(type, group, enum, json, default) LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_##enum, std::to_string(default));
|
||||
#define OPT_STR(group, enum, json, default) LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_##enum, default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) // Do nothing
|
||||
#define OPT_STRENUM(group, enum, json, default) \
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_##enum, std::to_string(FEXCore::ToUnderlying(default)));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
|
||||
// Erase unnamed options which shouldn't be set
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS_INTERPRETER);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_INTERPRETER_INSTALLED);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_FILENAME);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_CONFIG_NAME);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS64BIT_MODE);
|
||||
}
|
||||
|
||||
static void ConfigInit(fextl::string ConfigFilename) {
|
||||
#define OPT_BASE(type, group, enum, json, default) \
|
||||
ConfigToNameLookup[FEXCore::Config::ConfigOption::CONFIG_##enum].first = #json; \
|
||||
ConfigToNameLookup[FEXCore::Config::ConfigOption::CONFIG_##enum].second = #type; \
|
||||
NameToConfigLookup[#json] = FEXCore::Config::ConfigOption::CONFIG_##enum;
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
#undef OPT_BASE
|
||||
|
||||
// Ensure config and RootFS directories exist
|
||||
std::error_code ec {};
|
||||
std::filesystem::path Dirs[] = {std::filesystem::absolute(ConfigFilename).parent_path(),
|
||||
std::filesystem::absolute(FEXCore::Config::GetDataDirectory()) / "RootFS/"};
|
||||
for (auto& Dir : Dirs) {
|
||||
bool created = std::filesystem::create_directories(Dir, ec);
|
||||
if (created) {
|
||||
qInfo() << "Created folder" << Dir.c_str();
|
||||
}
|
||||
if (ec) {
|
||||
QMessageBox err(QMessageBox::Critical, "Failed to create directory", QString("Failed to create \"%1\" folder").arg(Dir.c_str()),
|
||||
QMessageBox::Ok);
|
||||
err.exec();
|
||||
std::exit(EXIT_FAILURE);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
RootFSModel::RootFSModel() {
|
||||
INotifyFD = inotify_init1(IN_NONBLOCK | IN_CLOEXEC);
|
||||
|
||||
fextl::string RootFS = FEXCore::Config::GetDataDirectory() + "RootFS/";
|
||||
FolderFD = inotify_add_watch(INotifyFD, RootFS.c_str(), IN_CREATE | IN_DELETE);
|
||||
if (FolderFD != -1) {
|
||||
Thread = std::thread {&RootFSModel::INotifyThreadFunc, this};
|
||||
} else {
|
||||
qWarning() << "Could not set up inotify. RootFS folder won't be monitored for changes.";
|
||||
}
|
||||
|
||||
// Load initial data
|
||||
Reload();
|
||||
}
|
||||
|
||||
RootFSModel::~RootFSModel() {
|
||||
close(INotifyFD);
|
||||
INotifyFD = -1;
|
||||
|
||||
ExitRequest.count_down();
|
||||
Thread.join();
|
||||
}
|
||||
|
||||
void RootFSModel::Reload() {
|
||||
beginResetModel();
|
||||
removeRows(0, rowCount());
|
||||
|
||||
fextl::string RootFS = FEXCore::Config::GetDataDirectory() + "RootFS/";
|
||||
std::vector<QString> NamedRootFS {};
|
||||
for (auto& it : std::filesystem::directory_iterator(RootFS)) {
|
||||
if (it.is_directory()) {
|
||||
NamedRootFS.push_back(QString::fromStdString(it.path().filename()));
|
||||
} else if (it.is_regular_file()) {
|
||||
// If it is a regular file then we need to check if it is a valid archive
|
||||
if (it.path().extension() == ".sqsh" && FEX::FormatCheck::IsSquashFS(fextl::string_from_path(it.path()))) {
|
||||
NamedRootFS.push_back(QString::fromStdString(it.path().filename()));
|
||||
} else if (it.path().extension() == ".ero" && FEX::FormatCheck::IsEroFS(fextl::string_from_path(it.path()))) {
|
||||
NamedRootFS.push_back(QString::fromStdString(it.path().filename()));
|
||||
}
|
||||
}
|
||||
}
|
||||
std::sort(NamedRootFS.begin(), NamedRootFS.end(), [](const QString& a, const QString& b) { return QString::localeAwareCompare(a, b) < 0; });
|
||||
for (auto& Entry : NamedRootFS) {
|
||||
appendRow(new QStandardItem(Entry));
|
||||
}
|
||||
|
||||
endResetModel();
|
||||
}
|
||||
|
||||
bool RootFSModel::hasItem(const QString& Name) const {
|
||||
return !findItems(Name, Qt::MatchExactly).empty();
|
||||
}
|
||||
|
||||
QUrl RootFSModel::getBaseUrl() const {
|
||||
return QUrl::fromLocalFile(QString::fromStdString(FEXCore::Config::GetDataDirectory().c_str()) + "RootFS/");
|
||||
}
|
||||
|
||||
void RootFSModel::INotifyThreadFunc() {
|
||||
while (!ExitRequest.try_wait()) {
|
||||
constexpr size_t DATA_SIZE = (16 * (sizeof(struct inotify_event) + NAME_MAX + 1));
|
||||
char buf[DATA_SIZE];
|
||||
int Ret {};
|
||||
do {
|
||||
fd_set Set {};
|
||||
FD_ZERO(&Set);
|
||||
FD_SET(INotifyFD, &Set);
|
||||
struct timeval tv {};
|
||||
// 50 ms
|
||||
tv.tv_usec = 50000;
|
||||
Ret = select(INotifyFD + 1, &Set, nullptr, nullptr, &tv);
|
||||
} while (Ret == 0 && INotifyFD != -1);
|
||||
|
||||
if (Ret == -1 || INotifyFD == -1) {
|
||||
// Just return on error
|
||||
return;
|
||||
}
|
||||
|
||||
// Spin through the events, we don't actually care what they are
|
||||
while (read(INotifyFD, buf, DATA_SIZE) > 0)
|
||||
;
|
||||
|
||||
// Queue update to the data model
|
||||
QMetaObject::invokeMethod(this, "Reload");
|
||||
}
|
||||
}
|
||||
|
||||
// Returns true on success
|
||||
static bool OpenFile(fextl::string Filename) {
|
||||
std::error_code ec {};
|
||||
if (!std::filesystem::exists(Filename, ec)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
LoadedConfig = FEX::Config::CreateMainLayer(&Filename);
|
||||
LoadedConfig->Load();
|
||||
|
||||
// Load default options and only overwrite only if the option didn't exist
|
||||
#define OPT_BASE(type, group, enum, json, default) \
|
||||
if (!LoadedConfig->OptionExists(FEXCore::Config::ConfigOption::CONFIG_##enum)) { \
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_##enum, std::to_string(default)); \
|
||||
}
|
||||
#define OPT_STR(group, enum, json, default) \
|
||||
if (!LoadedConfig->OptionExists(FEXCore::Config::ConfigOption::CONFIG_##enum)) { \
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_##enum, default); \
|
||||
}
|
||||
#define OPT_STRARRAY(group, enum, json, default) // Do nothing
|
||||
#define OPT_STRENUM(group, enum, json, default) \
|
||||
if (!LoadedConfig->OptionExists(FEXCore::Config::ConfigOption::CONFIG_##enum)) { \
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_##enum, std::to_string(FEXCore::ToUnderlying(default))); \
|
||||
}
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
|
||||
// Erase unnamed options which shouldn't be set
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS_INTERPRETER);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_INTERPRETER_INSTALLED);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_FILENAME);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_CONFIG_NAME);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS64BIT_MODE);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
ConfigRuntime::ConfigRuntime(const QString& ConfigFilename) {
|
||||
qmlRegisterSingletonInstance<ConfigModel>("FEX.ConfigModel", 1, 0, "ConfigModel", &ConfigModelInst);
|
||||
qmlRegisterSingletonInstance<RootFSModel>("FEX.RootFSModel", 1, 0, "RootFSModel", &RootFSList);
|
||||
Engine.load(QUrl("qrc:/main.qml"));
|
||||
|
||||
Window = qobject_cast<QQuickWindow*>(Engine.rootObjects().first());
|
||||
if (!ConfigFilename.isEmpty()) {
|
||||
Window->setProperty("configFilename", QUrl::fromLocalFile(ConfigFilename));
|
||||
} else {
|
||||
Window->setProperty("configFilename", QUrl::fromLocalFile(FEXCore::Config::GetConfigFileLocation().c_str()));
|
||||
Window->setProperty("configDirty", true);
|
||||
Window->setProperty("loadedDefaults", true);
|
||||
}
|
||||
|
||||
ConfigRuntime::connect(Window, SIGNAL(selectedConfigFile(const QUrl&)), this, SLOT(onLoad(const QUrl&)));
|
||||
ConfigRuntime::connect(Window, SIGNAL(triggeredSave(const QUrl&)), this, SLOT(onSave(const QUrl&)));
|
||||
ConfigRuntime::connect(&ConfigModelInst, SIGNAL(modelReset()), Window, SLOT(refreshUI()));
|
||||
}
|
||||
|
||||
void ConfigRuntime::onSave(const QUrl& Filename) {
|
||||
qInfo() << "Saving to" << Filename.toLocalFile().toStdString().c_str();
|
||||
FEX::Config::SaveLayerToJSON(Filename.toLocalFile().toStdString().c_str(), LoadedConfig.get());
|
||||
}
|
||||
|
||||
void ConfigRuntime::onLoad(const QUrl& Filename) {
|
||||
// TODO: Distinguish between "load" and "overlay".
|
||||
// Currently, the new configuration is overlaid on top of the previous one.
|
||||
|
||||
if (!OpenFile(Filename.toLocalFile().toStdString().c_str())) {
|
||||
// This basically never happens because OpenFile performs no actual syntax checks.
|
||||
// Treat as fatal since the UI state wouldn't be consistent after ignoring the error.
|
||||
QMessageBox err(QMessageBox::Critical, tr("Could not load config file"), tr("Failed to load \"%1\"").arg(Filename.toLocalFile()),
|
||||
QMessageBox::Ok);
|
||||
err.exec();
|
||||
QApplication::exit();
|
||||
return;
|
||||
}
|
||||
|
||||
ConfigModelInst.Reload();
|
||||
RootFSList.Reload();
|
||||
|
||||
QMetaObject::invokeMethod(Window, "refreshUI");
|
||||
}
|
||||
|
||||
int main(int Argc, char** Argv) {
|
||||
QApplication App(Argc, Argv);
|
||||
|
||||
FEX::Config::InitializeConfigs(FEX::Config::PortableInformation {});
|
||||
fextl::string ConfigFilename = Argc > 1 ? Argv[1] : FEXCore::Config::GetConfigFileLocation();
|
||||
ConfigInit(ConfigFilename);
|
||||
|
||||
qInfo() << "Opening" << ConfigFilename.c_str();
|
||||
if (!OpenFile(ConfigFilename)) {
|
||||
// Load defaults if not found
|
||||
ConfigFilename.clear();
|
||||
LoadDefaultSettings();
|
||||
}
|
||||
|
||||
ConfigRuntime Runtime(ConfigFilename.c_str());
|
||||
|
||||
return App.exec();
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <QStandardItemModel>
|
||||
#include <QQmlApplicationEngine>
|
||||
|
||||
#include <latch>
|
||||
#include <thread>
|
||||
|
||||
class QQuickWindow;
|
||||
|
||||
class ConfigModel : public QStandardItemModel {
|
||||
Q_OBJECT
|
||||
QML_ELEMENT
|
||||
QML_SINGLETON
|
||||
|
||||
public:
|
||||
ConfigModel();
|
||||
|
||||
void Reload();
|
||||
|
||||
public slots:
|
||||
bool has(const QString&, bool unused) const;
|
||||
void erase(const QString&);
|
||||
|
||||
bool getBool(const QString&, bool unused) const;
|
||||
QString getString(const QString&, bool unused) const;
|
||||
QStringList getStringList(const QString&, bool unused) const;
|
||||
int getInt(const QString&, bool unused) const;
|
||||
|
||||
void setBool(const QString&, bool);
|
||||
void setString(const QString&, const QString&);
|
||||
void setStringList(const QString&, const QStringList&);
|
||||
void setInt(const QString&, int value);
|
||||
};
|
||||
|
||||
class RootFSModel : public QStandardItemModel {
|
||||
Q_OBJECT
|
||||
QML_ELEMENT
|
||||
QML_SINGLETON
|
||||
|
||||
std::thread Thread;
|
||||
std::latch ExitRequest {1};
|
||||
|
||||
int INotifyFD;
|
||||
int FolderFD;
|
||||
|
||||
void INotifyThreadFunc();
|
||||
|
||||
public:
|
||||
RootFSModel();
|
||||
~RootFSModel();
|
||||
|
||||
public slots:
|
||||
void Reload();
|
||||
|
||||
bool hasItem(const QString&) const;
|
||||
|
||||
QUrl getBaseUrl() const;
|
||||
};
|
||||
|
||||
class ConfigRuntime : public QObject {
|
||||
Q_OBJECT
|
||||
|
||||
QQmlApplicationEngine Engine;
|
||||
QQuickWindow* Window = nullptr;
|
||||
RootFSModel RootFSList;
|
||||
ConfigModel ConfigModelInst;
|
||||
|
||||
public:
|
||||
ConfigRuntime(const QString& ConfigFilename);
|
||||
|
||||
public slots:
|
||||
void onSave(const QUrl&);
|
||||
void onLoad(const QUrl&);
|
||||
};
|
||||
@@ -0,0 +1,925 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
import QtQuick 2.15
|
||||
import QtQuick.Controls 2.15
|
||||
import QtQuick.Layouts 1.15
|
||||
|
||||
import FEX.ConfigModel 1.0
|
||||
import FEX.RootFSModel 1.0
|
||||
|
||||
// Qt 6 changed the API of the Dialogs module slightly.
|
||||
// The differences are abstracted away in this import:
|
||||
import "qrc:/dialogs"
|
||||
|
||||
ApplicationWindow {
|
||||
id: root
|
||||
|
||||
visible: true
|
||||
width: 540
|
||||
height: 585
|
||||
minimumWidth: 500
|
||||
minimumHeight: 450
|
||||
title: configDirty ? qsTr("FEX configuration *") : qsTr("FEX configuration")
|
||||
|
||||
property url configFilename
|
||||
|
||||
property bool configDirty: false
|
||||
property bool loadedDefaults: false
|
||||
property bool closeConfirmed: false
|
||||
|
||||
signal selectedConfigFile(name: url)
|
||||
signal triggeredSave(name: url)
|
||||
|
||||
// Property used to force reloading any elements that read ConfigModel
|
||||
property bool refreshCache: false
|
||||
|
||||
onConfigDirtyChanged: {
|
||||
if (!configDirty) {
|
||||
// We either just saved or loaded a file
|
||||
loadedDefaults = false
|
||||
}
|
||||
}
|
||||
|
||||
function refreshUI() {
|
||||
refreshCache = !refreshCache
|
||||
}
|
||||
|
||||
function urlToLocalFile(theurl: url): string {
|
||||
var str = theurl.toString()
|
||||
if (str.startsWith("file://")) {
|
||||
return decodeURIComponent(str.substring(7))
|
||||
}
|
||||
if (str.startsWith("file:")) {
|
||||
return decodeURIComponent(str.substring(5))
|
||||
}
|
||||
|
||||
return str;
|
||||
}
|
||||
|
||||
FileDialog {
|
||||
id: openFileDialog
|
||||
property bool isSaving: false
|
||||
|
||||
title: isSaving ? qsTr("Save FEX configuration") : qsTr("Open FEX configuration")
|
||||
nameFilters: [ qsTr("Config files(*.json)"), qsTr("All files(*)") ]
|
||||
|
||||
selectExisting: !isSaving
|
||||
|
||||
property var onNextAccept: null
|
||||
|
||||
// Prompts the user for an existing file and calls the callback on completion
|
||||
function loadAndThen(callback) {
|
||||
isSaving = false
|
||||
console.assert(!onNextAccept, "Tried to open dialog multiple times")
|
||||
onNextAccept = callback
|
||||
open()
|
||||
}
|
||||
|
||||
// Prompts the user for a new or existing file and calls the callback on completion
|
||||
function saveAndThen(callback) {
|
||||
isSaving = true
|
||||
console.assert(!onNextAccept, "Tried to open dialog multiple times")
|
||||
onNextAccept = callback
|
||||
open()
|
||||
}
|
||||
|
||||
onAccepted: {
|
||||
if (!isSaving) {
|
||||
root.selectedConfigFile(selectedFile)
|
||||
}
|
||||
configFilename = selectedFile
|
||||
configDirty = false
|
||||
if (onNextAccept) {
|
||||
onNextAccept()
|
||||
onNextAccept = null
|
||||
}
|
||||
}
|
||||
|
||||
onRejected: onNextAccept = null
|
||||
}
|
||||
|
||||
MessageDialog {
|
||||
id: confirmCloseDialog
|
||||
title: qsTr("Save changes")
|
||||
text: configFilename.toString() === "" ? qsTr("Save changes before quitting?") : qsTr("Save changes to %1 before quitting?").arg(urlToLocalFile(configFilename))
|
||||
buttons: buttonSave | buttonDiscard | buttonCancel
|
||||
|
||||
onButtonClicked: (button) => {
|
||||
switch (button) {
|
||||
case buttonSave:
|
||||
if (configFilename.toString() === "") {
|
||||
// Filename not yet set => trigger "Save As" dialog
|
||||
openFileDialog.saveAndThen(() => {
|
||||
save(configFilename)
|
||||
root.close()
|
||||
});
|
||||
return
|
||||
}
|
||||
save(configFilename)
|
||||
root.close()
|
||||
break
|
||||
|
||||
case buttonDiscard:
|
||||
closeConfirmed = true
|
||||
root.close()
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
onClosing: (close) => {
|
||||
if (configDirty) {
|
||||
close.accepted = closeConfirmed
|
||||
onTriggered: confirmCloseDialog.open()
|
||||
}
|
||||
}
|
||||
|
||||
function save(filename: url) {
|
||||
if (filename.toString() === "") {
|
||||
filename = configFilename
|
||||
}
|
||||
|
||||
if (filename.toString() === "") {
|
||||
// Filename not yet set => trigger "Save As" dialog
|
||||
openFileDialog.saveAndThen(() => {
|
||||
save(configFilename)
|
||||
});
|
||||
return
|
||||
}
|
||||
|
||||
triggeredSave(filename)
|
||||
configDirty = false
|
||||
}
|
||||
|
||||
menuBar: MenuBar {
|
||||
Menu {
|
||||
title: qsTr("&File")
|
||||
Action {
|
||||
text: qsTr("&Open...")
|
||||
shortcut: StandardKey.Open
|
||||
// TODO: Ask to discard pending changes first
|
||||
onTriggered: openFileDialog.loadAndThen(() => {})
|
||||
}
|
||||
Action {
|
||||
text: qsTr("&Save")
|
||||
shortcut: StandardKey.Save
|
||||
onTriggered: root.save("")
|
||||
}
|
||||
Action {
|
||||
text: qsTr("Save &as...")
|
||||
shortcut: StandardKey.SaveAs
|
||||
onTriggered: {
|
||||
openFileDialog.saveAndThen(() => {
|
||||
root.save(configFilename)
|
||||
});
|
||||
}
|
||||
}
|
||||
MenuSeparator {}
|
||||
Action {
|
||||
text: qsTr("&Quit")
|
||||
shortcut: StandardKey.Quit
|
||||
onTriggered: close()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
header: TabBar {
|
||||
id: tabBar
|
||||
currentIndex: 0
|
||||
|
||||
TabButton {
|
||||
text: qsTr("General")
|
||||
}
|
||||
TabButton {
|
||||
text: qsTr("Emulation")
|
||||
}
|
||||
TabButton {
|
||||
text: qsTr("CPU")
|
||||
}
|
||||
TabButton {
|
||||
text: qsTr("Advanced")
|
||||
}
|
||||
}
|
||||
|
||||
component ConfigCheckBox: CheckBox {
|
||||
property string config
|
||||
property string tooltip
|
||||
property bool invert: false
|
||||
|
||||
ToolTip.visible: (visualFocus || hovered) && tooltip !== ""
|
||||
ToolTip.text: tooltip
|
||||
|
||||
onToggled: {
|
||||
configDirty = true
|
||||
ConfigModel.setBool(config, checked ^ invert)
|
||||
}
|
||||
|
||||
checkState: config === "" ? Qt.PartiallyChecked
|
||||
: !ConfigModel.has(config, refreshCache) ? Qt.PartiallyChecked
|
||||
: (ConfigModel.getBool(config, refreshCache) ^ invert) ? Qt.Checked
|
||||
: Qt.Unchecked
|
||||
}
|
||||
|
||||
component ConfigSpinBox: SpinBox {
|
||||
property string config
|
||||
|
||||
textFromValue: (val) => {
|
||||
if (valueFromConfig === "") {
|
||||
return qsTr("(not set)");
|
||||
}
|
||||
|
||||
return val.toString()
|
||||
}
|
||||
|
||||
onValueModified: {
|
||||
configDirty = true
|
||||
ConfigModel.setInt(config, value)
|
||||
}
|
||||
|
||||
property string valueFromConfig: config === "" ? 0 : ConfigModel.has(config, refreshCache) ? ConfigModel.getInt(config, refreshCache).toString() : ""
|
||||
|
||||
value: valueFromConfig
|
||||
from: 0
|
||||
to: 1 << 30
|
||||
}
|
||||
|
||||
component ConfigTextField: TextField {
|
||||
property string config
|
||||
property bool hasData: config !== "" && ConfigModel.has(config, refreshCache)
|
||||
text: hasData ? ConfigModel.getString(config, refreshCache) : "(none set)"
|
||||
enabled: hasData
|
||||
|
||||
onTextEdited: {
|
||||
configDirty = true
|
||||
ConfigModel.setString(config, text)
|
||||
}
|
||||
}
|
||||
|
||||
component ConfigTextFieldForPath: RowLayout {
|
||||
property string text
|
||||
property string config
|
||||
|
||||
property var dialog: FileDialog {}
|
||||
|
||||
FileDialog { id: fileSelectorDialog }
|
||||
|
||||
Label { text: parent.text }
|
||||
ConfigTextField {
|
||||
Layout.fillWidth: true
|
||||
config: parent.config
|
||||
readOnly: true
|
||||
}
|
||||
|
||||
Button {
|
||||
icon.name: "search"
|
||||
onClicked: dialog.open()
|
||||
}
|
||||
|
||||
Component.onCompleted: {
|
||||
dialog.accepted.connect(() => {
|
||||
var selectedPath = (dialog instanceof FileDialog ? dialog.selectedFile : dialog.selectedFolder)
|
||||
|
||||
configDirty = true
|
||||
ConfigModel.setString(config, urlToLocalFile(selectedPath))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
StackLayout {
|
||||
anchors.left: parent.left
|
||||
anchors.right: parent.right
|
||||
anchors.top: tabBar.bottom
|
||||
anchors.bottom: parent.bottom
|
||||
|
||||
currentIndex: tabBar.currentIndex
|
||||
|
||||
component ScrollablePage: ScrollView {
|
||||
id: outer
|
||||
|
||||
readonly property var visibleScrollbarWidth: ScrollBar.vertical.visible ? ScrollBar.vertical.width : 0
|
||||
|
||||
// Children given by the user will be moved into the inner Column
|
||||
default property alias content: inner.children
|
||||
|
||||
property alias itemSpacing: inner.spacing
|
||||
|
||||
Column {
|
||||
id: inner
|
||||
|
||||
spacing: 8
|
||||
padding: 8
|
||||
|
||||
// This must be explicitly set via the id, since parent doesn't seem to be recognized within Column
|
||||
width: outer.width - outer.visibleScrollbarWidth
|
||||
}
|
||||
}
|
||||
|
||||
// Environment settings
|
||||
ScrollablePage {
|
||||
GroupBox {
|
||||
id: rootfsGroupBox
|
||||
title: qsTr("RootFS:")
|
||||
width: parent.width - parent.padding * 2
|
||||
|
||||
ColumnLayout {
|
||||
width: rootfsGroupBox.width - rootfsGroupBox.padding * 2
|
||||
ScrollView {
|
||||
Layout.fillWidth: true
|
||||
Layout.maximumHeight: 150
|
||||
clip: true
|
||||
|
||||
Column {
|
||||
id: rootfsList
|
||||
|
||||
property string selectedItem
|
||||
property string explicitEntry
|
||||
|
||||
spacing: 4
|
||||
|
||||
Component.onCompleted: {
|
||||
var initState = (ref) => {
|
||||
selectedItem = ConfigModel.has("RootFS", ref) ? ConfigModel.getString("RootFS", ref) : ""
|
||||
|
||||
// RootFSModel only lists entries in the $FEX_HOME/RootFS/ folder.
|
||||
// If a custom path is selected, add it as a dedicated entry
|
||||
if (selectedItem !== "" && !RootFSModel.hasItem(selectedItem)) {
|
||||
explicitEntry = selectedItem
|
||||
|
||||
// Make visible once needed.
|
||||
// Conversely, if the user selects something else after, keep the old option visible to allow easy undoing
|
||||
fallbackRootfsEntry.visible = true
|
||||
}
|
||||
}
|
||||
|
||||
initState(false)
|
||||
root.refreshCacheChanged.connect(initState)
|
||||
}
|
||||
|
||||
function updateRootFS(fileOrFolder: url) {
|
||||
configDirty = true
|
||||
var base = urlToLocalFile(RootFSModel.getBaseUrl())
|
||||
var file = urlToLocalFile(fileOrFolder)
|
||||
if (file.startsWith(base)) {
|
||||
file = file.substring(base.length)
|
||||
}
|
||||
|
||||
ConfigModel.setString("RootFS", file)
|
||||
refreshUI()
|
||||
}
|
||||
|
||||
component RootFSRadioDelegate: RadioButton {
|
||||
property var name
|
||||
|
||||
text: name
|
||||
checked: rootfsList.selectedItem === name
|
||||
|
||||
onToggled: {
|
||||
configDirty = true;
|
||||
ConfigModel.setString("RootFS", name)
|
||||
}
|
||||
}
|
||||
|
||||
RootFSRadioDelegate {
|
||||
id: fallbackRootfsEntry
|
||||
visible: false
|
||||
name: rootfsList.explicitEntry
|
||||
}
|
||||
Repeater {
|
||||
model: RootFSModel
|
||||
delegate: RootFSRadioDelegate { name: model.display }
|
||||
}
|
||||
}
|
||||
}
|
||||
RowLayout {
|
||||
FileDialog {
|
||||
id: addRootfsFileDialog
|
||||
title: qsTr("Select RootFS file")
|
||||
nameFilters: [ qsTr("SquashFS and EroFS (*.sqsh *.ero)"), qsTr("All files(*)") ]
|
||||
currentFolder: RootFSModel.getBaseUrl()
|
||||
onAccepted: rootfsList.updateRootFS(fileUrl)
|
||||
}
|
||||
|
||||
FolderDialog {
|
||||
id: addRootfsFolderDialog
|
||||
title: qsTr("Select RootFS folder")
|
||||
currentFolder: RootFSModel.getBaseUrl()
|
||||
onAccepted: rootfsList.updateRootFS(selectedFolder)
|
||||
}
|
||||
|
||||
Button {
|
||||
text: qsTr("Add archive")
|
||||
icon.name: "document-open"
|
||||
onClicked: addRootfsFileDialog.open()
|
||||
}
|
||||
Button {
|
||||
text: qsTr("Add folder")
|
||||
icon.name: "folder"
|
||||
onClicked: addRootfsFolderDialog.open()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
GroupBox {
|
||||
title: qsTr("Library forwarding:")
|
||||
width: parent.width - parent.padding * 2
|
||||
|
||||
ColumnLayout {
|
||||
anchors.left: parent ? parent.left : undefined
|
||||
anchors.right: parent ? parent.right : undefined
|
||||
|
||||
id: libfwdConfig
|
||||
|
||||
property url configDir: (() => {
|
||||
var configPath = urlToLocalFile(configFilename)
|
||||
var slashIndex = configPath.lastIndexOf('/')
|
||||
if (slashIndex === -1) {
|
||||
return ""
|
||||
}
|
||||
return "file://" + configPath.substr(0, slashIndex)
|
||||
})()
|
||||
|
||||
ConfigTextFieldForPath {
|
||||
text: qsTr("Configuration:")
|
||||
config: "ThunkConfig"
|
||||
dialog: FileDialog {
|
||||
title: qsTr("Select library forwarding configuration")
|
||||
nameFilters: [ qsTr("JSON files (*.json)"), qsTr("All files(*)") ]
|
||||
currentFolder: libfwdConfig.configDir
|
||||
}
|
||||
}
|
||||
ConfigTextFieldForPath {
|
||||
text: qsTr("Host library folder:")
|
||||
config: "ThunkHostLibs"
|
||||
dialog: FolderDialog {
|
||||
title: qsTr("Select path for host libraries")
|
||||
currentFolder: libfwdConfig.configDir
|
||||
}
|
||||
}
|
||||
ConfigTextFieldForPath {
|
||||
text: qsTr("Guest library folder:")
|
||||
config: "ThunkGuestLibs"
|
||||
dialog: FolderDialog {
|
||||
title: qsTr("Select path for guest libraries")
|
||||
currentFolder: libfwdConfig.configDir
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
GroupBox {
|
||||
title: qsTr("Logging:")
|
||||
width: parent.width - parent.padding * 2
|
||||
|
||||
label: ConfigCheckBox {
|
||||
id: loggingEnabledCheckBox
|
||||
config: "SilentLog"
|
||||
text: qsTr("Logging:")
|
||||
invert: true
|
||||
}
|
||||
|
||||
ColumnLayout {
|
||||
enabled: loggingEnabledCheckBox.checked
|
||||
|
||||
anchors.left: parent ? parent.left : undefined
|
||||
anchors.right: parent ? parent.right : undefined
|
||||
|
||||
RowLayout {
|
||||
Label { text: qsTr("Log to:") }
|
||||
|
||||
ComboBox {
|
||||
id: loggingComboBox
|
||||
property string configValue: ConfigModel.has("OutputLog", refreshCache) ? ConfigModel.getString("OutputLog", refreshCache) : ""
|
||||
|
||||
currentIndex: configValue === "" ? -1 : configValue == "server" ? 0 : configValue == "stderr" ? 1 : configValue == "stdout" ? 2 : 3
|
||||
|
||||
onActivated: {
|
||||
configDirty = true
|
||||
var configNames = [ "server", "stderr", "stdout" ]
|
||||
if (currentIndex != -1 && currentIndex < 3) {
|
||||
ConfigModel.setString("OutputLog", configNames[currentIndex])
|
||||
} else {
|
||||
// Set by text field below
|
||||
}
|
||||
}
|
||||
|
||||
model: ListModel {
|
||||
ListElement { text: "FEXServer" }
|
||||
ListElement { text: "stderr" }
|
||||
ListElement { text: "stdout" }
|
||||
ListElement { text: qsTr("File...") }
|
||||
}
|
||||
}
|
||||
|
||||
ConfigTextFieldForPath {
|
||||
visible: loggingComboBox.currentIndex === 3
|
||||
config: "OutputLog"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Emulation settings
|
||||
ScrollablePage {
|
||||
RowLayout {
|
||||
Label { text: qsTr("SMC detection:") }
|
||||
ComboBox {
|
||||
currentIndex: ConfigModel.has("SMCChecks", refreshCache) ? ConfigModel.getInt("SMCChecks", refreshCache) : -1
|
||||
|
||||
onActivated: {
|
||||
configDirty = true
|
||||
ConfigModel.setInt("SMCChecks", currentIndex)
|
||||
}
|
||||
|
||||
model: ListModel {
|
||||
ListElement { text: qsTr("None") }
|
||||
ListElement { text: qsTr("MTrack") }
|
||||
ListElement { text: qsTr("Full") }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
GroupBox {
|
||||
title: qsTr("Memory model:")
|
||||
width: parent.width - parent.padding * 2
|
||||
|
||||
ColumnLayout {
|
||||
anchors.left: parent ? parent.left : undefined
|
||||
anchors.right: parent ? parent.right : undefined
|
||||
|
||||
ButtonGroup {
|
||||
id: tsoButtonGroup
|
||||
buttons: [tso1, tso2, tso3]
|
||||
// Trying to be too clever here will trigger property binding loops,
|
||||
// so require both TSOEnabled and ParanoidTSO to be listed in the config.
|
||||
// If they are not, the state will be displayed as undetermined.
|
||||
checkedButton: !(ConfigModel.has("TSOEnabled", refreshCache) && (ConfigModel.has("ParanoidTSO", refreshCache))) ? null
|
||||
: ConfigModel.getBool("ParanoidTSO", refreshCache) ? tso3
|
||||
: ConfigModel.getBool("TSOEnabled", refreshCache) ? tso2 : tso1
|
||||
|
||||
property int pendingItemChange: -1
|
||||
|
||||
function onClickedButton(index: int) {
|
||||
pendingItemChange = index;
|
||||
|
||||
configDirty = true;
|
||||
|
||||
var newIndex = pendingItemChange
|
||||
var TSOEnabled = newIndex === 1
|
||||
var ParanoidTSO = newIndex === 2
|
||||
ConfigModel.setBool("ParanoidTSO", ParanoidTSO)
|
||||
ConfigModel.setBool("TSOEnabled", TSOEnabled)
|
||||
|
||||
pendingItemChange = -1;
|
||||
}
|
||||
|
||||
onClicked: {
|
||||
if (pendingItemChange !== -1) {
|
||||
return;
|
||||
}
|
||||
pendingItemChange = tso1.checked ? 0 : tso2.checked ? 1 : tso3.checked ? 2 : -1;
|
||||
if (pendingItemChange) {
|
||||
// Undetermined state, leave as is
|
||||
return;
|
||||
}
|
||||
|
||||
var newIndex = pendingItemChange
|
||||
var TSOEnabled = newIndex === 1
|
||||
var ParanoidTSO = newIndex === 2
|
||||
ConfigModel.setBool("ParanoidTSO", ParanoidTSO)
|
||||
ConfigModel.setBool("TSOEnabled", TSOEnabled)
|
||||
|
||||
pendingItemChange = -1;
|
||||
}
|
||||
}
|
||||
|
||||
ColumnLayout {
|
||||
RadioButton {
|
||||
id: tso1
|
||||
text: qsTr("Inaccurate")
|
||||
onToggled: tsoButtonGroup.onClickedButton(0)
|
||||
}
|
||||
|
||||
ColumnLayout {
|
||||
RadioButton {
|
||||
id: tso2
|
||||
text: qsTr("Accurate (TSO)")
|
||||
onToggled: tsoButtonGroup.onClickedButton(1)
|
||||
}
|
||||
|
||||
ColumnLayout {
|
||||
visible: tso2.checked
|
||||
|
||||
ConfigCheckBox {
|
||||
text: qsTr("... for vector instructions")
|
||||
tooltip: qsTr("Controls TSO emulation on vector load/store instructions")
|
||||
config: "VectorTSOEnabled"
|
||||
}
|
||||
ConfigCheckBox {
|
||||
text: qsTr("... for memcpy instructions")
|
||||
tooltip: qsTr("Controls TSO emulation on memcpy/memset instructions")
|
||||
config: "MemcpySetTSOEnabled"
|
||||
}
|
||||
ConfigCheckBox {
|
||||
text: qsTr("... for unaligned half-barriers")
|
||||
tooltip: qsTr("Controls half-barrier TSO emulation on unaligned load/store instructions")
|
||||
config: "HalfBarrierTSOEnabled"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
RadioButton {
|
||||
id: tso3
|
||||
text: qsTr("Overly accurate (paranoid TSO)")
|
||||
onToggled: tsoButtonGroup.onClickedButton(2)
|
||||
}
|
||||
}
|
||||
|
||||
ConfigCheckBox {
|
||||
topPadding: 4
|
||||
text: qsTr("Enable non-tearing split-lock atomics")
|
||||
config: "StrictInProcessSplitLocks"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
component EnvVarList: GroupBox {
|
||||
width: parent.width - parent.padding * 2
|
||||
|
||||
property bool ofHost: false
|
||||
|
||||
ColumnLayout {
|
||||
anchors.left: parent ? parent.left : undefined
|
||||
anchors.right: parent ? parent.right : undefined
|
||||
|
||||
spacing: 0
|
||||
|
||||
id: envGroup
|
||||
property var values: ConfigModel.getStringList(ofHost ? "HostEnv" : "Env", refreshCache)
|
||||
|
||||
property int editedIndex: -1
|
||||
Repeater {
|
||||
model: parent.values
|
||||
Layout.fillWidth: true
|
||||
|
||||
RowLayout {
|
||||
property bool isEditing: envGroup.editedIndex === index
|
||||
|
||||
ItemDelegate {
|
||||
text: modelData;
|
||||
visible: !parent.isEditing
|
||||
onClicked: envGroup.editedIndex = index
|
||||
|
||||
}
|
||||
TextField {
|
||||
id: envVarEditTextField
|
||||
visible: parent.isEditing;
|
||||
text: modelData
|
||||
|
||||
onEditingFinished: {
|
||||
envGroup.editedIndex = -1
|
||||
if (text === modelData) {
|
||||
return
|
||||
}
|
||||
|
||||
var newValues = envGroup.values
|
||||
newValues[model.index] = text
|
||||
configDirty = true
|
||||
ConfigModel.setStringList(ofHost ? "HostEnv" : "Env", newValues)
|
||||
}
|
||||
}
|
||||
Button {
|
||||
visible: parent.isEditing
|
||||
icon.name: "list-remove"
|
||||
onClicked: {
|
||||
envGroup.editedIndex = -1
|
||||
var newValues = []
|
||||
for (var i = 0; i < envGroup.values.length; ++i) {
|
||||
if (i != index) {
|
||||
newValues.push(envGroup.values[i])
|
||||
}
|
||||
}
|
||||
|
||||
configDirty = true
|
||||
ConfigModel.setStringList(ofHost ? "HostEnv" : "Env", newValues)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
RowLayout {
|
||||
TextField {
|
||||
id: envVarTextField
|
||||
Layout.fillWidth: true
|
||||
|
||||
onAccepted: {
|
||||
var newValues = envGroup.values
|
||||
newValues.push(envVarTextField.text)
|
||||
configDirty = true
|
||||
ConfigModel.setStringList(ofHost ? "HostEnv" : "Env", newValues)
|
||||
text = ""
|
||||
}
|
||||
}
|
||||
Button {
|
||||
icon.name : "list-add"
|
||||
enabled: envVarTextField.text !== ""
|
||||
onClicked: envVarTextField.onAccepted()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
EnvVarList {
|
||||
title: qsTr("Guest environment variables:")
|
||||
}
|
||||
|
||||
EnvVarList {
|
||||
title: qsTr("Host environment variables:")
|
||||
ofHost: true
|
||||
}
|
||||
}
|
||||
|
||||
// CPU settings
|
||||
ScrollablePage {
|
||||
ConfigCheckBox {
|
||||
text: qsTr("Multiblock")
|
||||
config: "Multiblock"
|
||||
}
|
||||
|
||||
RowLayout {
|
||||
Layout.fillWidth: true
|
||||
|
||||
Label { text: qsTr("Block size:") }
|
||||
ConfigSpinBox {
|
||||
config: "MaxInst"
|
||||
from: 0
|
||||
to: 1 << 30
|
||||
}
|
||||
}
|
||||
|
||||
GroupBox {
|
||||
title: qsTr("JIT caches:")
|
||||
width: parent.width - parent.padding * 2
|
||||
|
||||
ColumnLayout {
|
||||
anchors.left: parent ? parent.left : undefined
|
||||
anchors.right: parent ? parent.right : undefined
|
||||
|
||||
ConfigCheckBox {
|
||||
text: qsTr("Generate AOT")
|
||||
config: "AOTIRGenerate"
|
||||
}
|
||||
ConfigCheckBox {
|
||||
text: qsTr("Capture AOT")
|
||||
config: "AOTIRCapture"
|
||||
}
|
||||
ConfigCheckBox {
|
||||
text: qsTr("Load AOT")
|
||||
config: "AOTIRLoad"
|
||||
}
|
||||
|
||||
RowLayout {
|
||||
Label { text: qsTr("Cache object code:") }
|
||||
|
||||
ButtonGroup {
|
||||
buttons: cacheObjCodeRadios.children
|
||||
|
||||
checkedButton: ConfigModel.has("CacheObjectCodeCompilation", refreshCache)
|
||||
? cacheObjCodeRadios.children[ConfigModel.getInt("CacheObjectCodeCompilation", refreshCache)]
|
||||
: null
|
||||
|
||||
onClicked: (button) => {
|
||||
configDirty = true
|
||||
for (var idx in buttons) {
|
||||
if (button === buttons[idx]) {
|
||||
ConfigModel.setInt("CacheObjectCodeCompilation", idx)
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
RowLayout {
|
||||
id: cacheObjCodeRadios
|
||||
RadioButton {
|
||||
text: qsTr("Off")
|
||||
}
|
||||
RadioButton {
|
||||
text: qsTr("Read-only")
|
||||
}
|
||||
RadioButton {
|
||||
text: qsTr("Read & write")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
ConfigCheckBox {
|
||||
text: qsTr("Reduced x87 precision")
|
||||
config: "X87ReducedPrecision"
|
||||
}
|
||||
|
||||
ConfigCheckBox {
|
||||
text: qsTr("Unsafe local flags optimization")
|
||||
config: "ABILocalFlags"
|
||||
}
|
||||
|
||||
ConfigCheckBox {
|
||||
text: qsTr("Disable JIT optimization passes")
|
||||
config: "O0"
|
||||
}
|
||||
}
|
||||
|
||||
// Advanced settings
|
||||
// NOTE: This is wrapped in a Loader that dynamically instantiates/destroys the page contents whenever the tab is selected.
|
||||
// This avoids costly UI updates for its UI elements.
|
||||
// TODO: Options contained multiple times in JSON aren't listed (neither are they in old FEXConfig though)
|
||||
Loader { sourceComponent: tabBar.currentIndex === 3 ? advancedSettingsPage : null }
|
||||
Component {
|
||||
id: advancedSettingsPage
|
||||
ScrollablePage {
|
||||
itemSpacing: 0
|
||||
Frame {
|
||||
width: parent.width - parent.padding * 2
|
||||
id: frame
|
||||
Column {
|
||||
Repeater {
|
||||
model: ConfigModel
|
||||
delegate: RowLayout {
|
||||
width: frame.width - frame.padding * 2
|
||||
|
||||
Label {
|
||||
id: label
|
||||
text: display
|
||||
}
|
||||
|
||||
ConfigCheckBox {
|
||||
visible: optionType == "bool"
|
||||
config: visible ? label.text : ""
|
||||
}
|
||||
|
||||
ConfigTextField {
|
||||
Layout.fillWidth: true
|
||||
visible: optionType == "fextl::string"
|
||||
config: visible ? label.text : ""
|
||||
}
|
||||
|
||||
ConfigSpinBox {
|
||||
visible: optionType.startsWith("int") || optionType.startsWith("uint")
|
||||
config: visible ? label.text : ""
|
||||
from: 0
|
||||
to: 1 << 30
|
||||
}
|
||||
|
||||
// Spacing
|
||||
Item {
|
||||
Layout.fillWidth: true
|
||||
}
|
||||
|
||||
Button {
|
||||
icon.name: "list-remove"
|
||||
onClicked: {
|
||||
ConfigModel.erase(label.text)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
footer: Pane {
|
||||
anchors.left: parent.left
|
||||
anchors.right: parent.right
|
||||
|
||||
padding: 0
|
||||
|
||||
ColumnLayout {
|
||||
anchors.left: parent.left
|
||||
anchors.right: parent.right
|
||||
spacing: 0
|
||||
|
||||
ToolSeparator {
|
||||
Layout.fillWidth: true
|
||||
orientation: Qt.Horizontal
|
||||
|
||||
// Override padding from theme.
|
||||
// Some themes use verticalPadding, others topPadding/bottomPadding, so we set them all.
|
||||
verticalPadding: 0
|
||||
bottomPadding: 0
|
||||
topPadding: 0
|
||||
}
|
||||
|
||||
Label {
|
||||
Layout.alignment: Qt.AlignHCenter
|
||||
enabled: false
|
||||
text: loadedDefaults
|
||||
? qsTr("Config.json not found — loaded defaults")
|
||||
: qsTr("Editing %1").arg(urlToLocalFile(configFilename))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
<RCC>
|
||||
<qresource prefix="/">
|
||||
<file>main.qml</file>
|
||||
</qresource>
|
||||
<qresource prefix="/dialogs">
|
||||
<file alias="FileDialog.qml">qt5/FileDialog.qml</file>
|
||||
<file alias="FolderDialog.qml">qt5/FolderDialog.qml</file>
|
||||
<file alias="MessageDialog.qml">qt5/MessageDialog.qml</file>
|
||||
</qresource>
|
||||
</RCC>
|
||||
@@ -0,0 +1,10 @@
|
||||
<RCC>
|
||||
<qresource prefix="/">
|
||||
<file>main.qml</file>
|
||||
</qresource>
|
||||
<qresource prefix="/dialogs">
|
||||
<file alias="FileDialog.qml">qt6/FileDialog.qml</file>
|
||||
<file alias="FolderDialog.qml">qt6/FolderDialog.qml</file>
|
||||
<file alias="MessageDialog.qml">qt6/MessageDialog.qml</file>
|
||||
</qresource>
|
||||
</RCC>
|
||||
@@ -0,0 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
import QtQuick.Dialogs 1.3 as FromQt
|
||||
|
||||
FromQt.FileDialog {
|
||||
property url selectedFile
|
||||
property url currentFolder
|
||||
|
||||
folder: currentFolder
|
||||
onAccepted: selectedFile = fileUrl
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
import QtQuick.Dialogs 1.3 as FromQt
|
||||
|
||||
FromQt.FileDialog {
|
||||
property url currentFolder
|
||||
property url selectedFolder
|
||||
|
||||
folder: currentFolder
|
||||
|
||||
selectFolder: true
|
||||
|
||||
onAccepted: selectedFolder = fileUrl
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
import QtQuick 2.15
|
||||
import QtQuick.Dialogs 1.3 as FromQt
|
||||
|
||||
Item {
|
||||
id: dialogParent
|
||||
property alias text: child.text
|
||||
property alias title: child.title
|
||||
|
||||
readonly property int buttonSave: FromQt.Dialog.Save
|
||||
readonly property int buttonDiscard: FromQt.Dialog.Discard
|
||||
readonly property int buttonCancel: FromQt.Dialog.Cancel
|
||||
|
||||
property int buttons
|
||||
|
||||
signal buttonClicked(button: int)
|
||||
|
||||
property bool pendingResult: false
|
||||
|
||||
function open() {
|
||||
// Workaround for QTBUG-91650, due to which signals may get emitted twice
|
||||
pendingResult = true
|
||||
child.open()
|
||||
}
|
||||
|
||||
FromQt.MessageDialog {
|
||||
id: child
|
||||
|
||||
standardButtons: buttons
|
||||
|
||||
onAccepted: {
|
||||
if (pendingResult) {
|
||||
dialogParent.buttonClicked(buttonSave)
|
||||
pendingResult = false
|
||||
}
|
||||
}
|
||||
onDiscard: {
|
||||
if (pendingResult) {
|
||||
dialogParent.buttonClicked(buttonDiscard)
|
||||
pendingResult = false
|
||||
}
|
||||
}
|
||||
onRejected: {
|
||||
if (pendingResult) {
|
||||
dialogParent.buttonClicked(buttonCancel)
|
||||
pendingResult = false
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
import QtQuick.Dialogs as FromQt
|
||||
|
||||
FromQt.FileDialog {
|
||||
property bool selectExisting: true
|
||||
property bool selectMultiple: false
|
||||
fileMode: selectMultiple ? FileDialog.OpenFiles : selectExisting ? FileDialog.OpenFile : FileDialog.SaveFile
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
import QtQuick.Dialogs as FromQt
|
||||
|
||||
FromQt.FolderDialog {
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
import QtQuick.Dialogs as FromQt
|
||||
|
||||
FromQt.MessageDialog {
|
||||
readonly property int buttonSave: MessageDialog.Save
|
||||
readonly property int buttonDiscard: MessageDialog.Discard
|
||||
readonly property int buttonCancel: MessageDialog.Cancel
|
||||
}
|
||||
@@ -44,6 +44,14 @@ fextl::vector<fextl::string> RemainingArgs;
|
||||
std::string DistroName {};
|
||||
std::string DistroVersion {};
|
||||
|
||||
enum class UIOverrideOption {
|
||||
Default,
|
||||
TTY,
|
||||
Zenity,
|
||||
};
|
||||
|
||||
UIOverrideOption UIOption {UIOverrideOption::Default};
|
||||
|
||||
void ParseArguments(int argc, char** argv) {
|
||||
optparse::OptionParser Parser = optparse::OptionParser().description("Tool for fetching RootFS from FEXServers").add_help_option(true);
|
||||
|
||||
@@ -59,6 +67,8 @@ void ParseArguments(int argc, char** argv) {
|
||||
|
||||
Parser.add_option("--distro-list-first").action("store_true").help("When presented the distro-list option, automatically select the first distro if there isn't an exact match.");
|
||||
|
||||
Parser.add_option("--force-ui").choices({"default", "tty", "zenity"}).set_default("default").help("Override which UI to use for selection");
|
||||
|
||||
optparse::Values Options = Parser.parse_args(argc, argv);
|
||||
|
||||
if (Options.is_set_by_user("assume_yes")) {
|
||||
@@ -85,6 +95,15 @@ void ParseArguments(int argc, char** argv) {
|
||||
DistroVersion = Options["distro_version"];
|
||||
}
|
||||
|
||||
if (Options.is_set_by_user("force_ui")) {
|
||||
auto Option = Options["force_ui"];
|
||||
if (Option == "tty") {
|
||||
UIOption = UIOverrideOption::TTY;
|
||||
} else if (Option == "zenity") {
|
||||
UIOption = UIOverrideOption::Zenity;
|
||||
}
|
||||
}
|
||||
|
||||
RemainingArgs = Parser.args();
|
||||
}
|
||||
} // namespace ArgOptions
|
||||
@@ -933,8 +952,6 @@ bool ValidateDownloadSelection(const WebFileFetcher::FileTargets& Target) {
|
||||
} // namespace TTY
|
||||
|
||||
namespace {
|
||||
bool IsTTY {};
|
||||
|
||||
std::function<bool(const fextl::string& Question)> _AskForConfirmation;
|
||||
std::function<void(const fextl::string& Text)> _ExecWithInfo;
|
||||
std::function<int32_t(const fextl::string& Text, const std::vector<fextl::string>& List)> _AskForConfirmationList;
|
||||
@@ -943,7 +960,12 @@ std::function<bool(const WebFileFetcher::FileTargets& Target)> _ValidateCheckExi
|
||||
std::function<bool(const WebFileFetcher::FileTargets& Target)> _ValidateDownloadSelection;
|
||||
|
||||
void CheckTTY() {
|
||||
IsTTY = isatty(STDOUT_FILENO);
|
||||
bool IsTTY {};
|
||||
if (ArgOptions::UIOption == ArgOptions::UIOverrideOption::Default) {
|
||||
IsTTY = isatty(STDOUT_FILENO);
|
||||
} else {
|
||||
IsTTY = ArgOptions::UIOption == ArgOptions::UIOverrideOption::TTY;
|
||||
}
|
||||
|
||||
if (IsTTY) {
|
||||
_AskForConfirmation = TTY::AskForConfirmation;
|
||||
@@ -1068,16 +1090,16 @@ bool ExtractEroFS(const fextl::string& Path, const fextl::string& RootFS, const
|
||||
} // namespace UnSquash
|
||||
|
||||
int main(int argc, char** argv, char** const envp) {
|
||||
CheckTTY();
|
||||
|
||||
auto ArgsLoader = fextl::make_unique<FEX::ArgLoader::ArgLoader>(FEX::ArgLoader::ArgLoader::LoadType::WITHOUT_FEXLOADER_PARSER, argc, argv);
|
||||
FEX::Config::LoadConfig(std::move(ArgsLoader), false, envp, false, {});
|
||||
FEX::Config::LoadConfig(std::move(ArgsLoader), {}, envp);
|
||||
|
||||
// Reload the meta layer
|
||||
FEXCore::Config::ReloadMetaLayer();
|
||||
|
||||
ArgOptions::ParseArguments(argc, argv);
|
||||
|
||||
CheckTTY();
|
||||
|
||||
if (ArgOptions::RemainingArgs.size()) {
|
||||
auto Res = XXFileHash::HashFile(ArgOptions::RemainingArgs[0]);
|
||||
if (Res.first) {
|
||||
|
||||
@@ -117,7 +117,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
|
||||
auto ArgsLoader = fextl::make_unique<FEX::ArgLoader::ArgLoader>(FEX::ArgLoader::ArgLoader::LoadType::WITHOUT_FEXLOADER_PARSER, argc, argv);
|
||||
FEX::Config::LoadConfig(std::move(ArgsLoader), false, envp, false, {});
|
||||
FEX::Config::LoadConfig(std::move(ArgsLoader), {}, envp);
|
||||
|
||||
// Reload the meta layer
|
||||
FEXCore::Config::ReloadMetaLayer();
|
||||
|
||||
@@ -4,15 +4,19 @@ set (SRCS
|
||||
VDSO_Emulation.cpp
|
||||
LinuxSyscalls/GdbServer.cpp
|
||||
LinuxSyscalls/EmulatedFiles/EmulatedFiles.cpp
|
||||
LinuxSyscalls/FaultSafeMemcpy.cpp
|
||||
LinuxSyscalls/FaultSafeUserMemAccess.cpp
|
||||
LinuxSyscalls/FileManagement.cpp
|
||||
LinuxSyscalls/LinuxAllocator.cpp
|
||||
LinuxSyscalls/NetStream.cpp
|
||||
LinuxSyscalls/Seccomp/SeccompEmulator.cpp
|
||||
LinuxSyscalls/Seccomp/BPFEmitter.cpp
|
||||
LinuxSyscalls/Seccomp/Dumper.cpp
|
||||
LinuxSyscalls/SignalDelegator.cpp
|
||||
LinuxSyscalls/Syscalls.cpp
|
||||
LinuxSyscalls/SyscallsSMCTracking.cpp
|
||||
LinuxSyscalls/SyscallsVMATracking.cpp
|
||||
LinuxSyscalls/ThreadManager.cpp
|
||||
LinuxSyscalls/SignalDelegator/GuestFramesManagement.cpp
|
||||
LinuxSyscalls/Utils/Threads.cpp
|
||||
LinuxSyscalls/x32/Syscalls.cpp
|
||||
LinuxSyscalls/x32/EPoll.cpp
|
||||
|
||||
@@ -1,69 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "LinuxSyscalls/Syscalls.h"
|
||||
|
||||
namespace FEX::HLE::FaultSafeMemcpy {
|
||||
#ifdef _M_ARM_64
|
||||
__attribute__((naked)) size_t CopyFromUser(void* Dest, const void* Src, size_t Size) {
|
||||
__asm volatile(R"(
|
||||
// Early exit if a memcpy of size zero.
|
||||
cbz x2, 2f;
|
||||
|
||||
1:
|
||||
.globl CopyFromUser_FaultInst
|
||||
CopyFromUser_FaultInst:
|
||||
ldrb w3, [x1], 1; // <- This line can fault.
|
||||
strb w3, [x0], 1;
|
||||
sub x2, x2, 1;
|
||||
cbnz x2, 1b;
|
||||
2:
|
||||
mov x0, 0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) size_t CopyToUser(void* Dest, const void* Src, size_t Size) {
|
||||
__asm volatile(R"(
|
||||
// Early exit if a memcpy of size zero.
|
||||
cbz x2, 2f;
|
||||
|
||||
1:
|
||||
ldrb w3, [x1], 1;
|
||||
.globl CopyToUser_FaultInst
|
||||
CopyToUser_FaultInst:
|
||||
strb w3, [x0], 1; // <- This line can fault.
|
||||
sub x2, x2, 1;
|
||||
cbnz x2, 1b;
|
||||
2:
|
||||
mov x0, 0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
extern "C" uint64_t CopyFromUser_FaultInst;
|
||||
void* const CopyFromUser_FaultLocation = &CopyFromUser_FaultInst;
|
||||
|
||||
extern "C" uint64_t CopyToUser_FaultInst;
|
||||
void* const CopyToUser_FaultLocation = &CopyToUser_FaultInst;
|
||||
|
||||
bool IsFaultLocation(uint64_t PC) {
|
||||
return reinterpret_cast<void*>(PC) == CopyFromUser_FaultLocation || reinterpret_cast<void*>(PC) == CopyToUser_FaultLocation;
|
||||
}
|
||||
|
||||
#else
|
||||
size_t CopyFromUser(void* Dest, const void* Src, size_t Size) {
|
||||
memcpy(Dest, Src, Size);
|
||||
return Size;
|
||||
}
|
||||
|
||||
size_t CopyToUser(void* Dest, const void* Src, size_t Size) {
|
||||
memcpy(Dest, Src, Size);
|
||||
return Size;
|
||||
}
|
||||
|
||||
bool IsFaultLocation(uint64_t PC) {
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
} // namespace FEX::HLE::FaultSafeMemcpy
|
||||
@@ -0,0 +1,183 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "LinuxSyscalls/Syscalls.h"
|
||||
|
||||
namespace FEX::HLE::FaultSafeUserMemAccess {
|
||||
#ifdef _M_ARM_64
|
||||
__attribute__((naked)) size_t CopyFromUser(void* Dest, const void* Src, size_t Size) {
|
||||
__asm volatile(R"(
|
||||
// Early exit if a memcpy of size zero.
|
||||
cbz x2, 2f;
|
||||
|
||||
1:
|
||||
.globl CopyFromUser_FaultInst
|
||||
CopyFromUser_FaultInst:
|
||||
ldrb w3, [x1], 1; // <- This line can fault.
|
||||
strb w3, [x0], 1;
|
||||
sub x2, x2, 1;
|
||||
cbnz x2, 1b;
|
||||
2:
|
||||
mov x0, 0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) size_t CopyToUser(void* Dest, const void* Src, size_t Size) {
|
||||
__asm volatile(R"(
|
||||
// Early exit if a memcpy of size zero.
|
||||
cbz x2, 2f;
|
||||
|
||||
1:
|
||||
ldrb w3, [x1], 1;
|
||||
.globl CopyToUser_FaultInst
|
||||
CopyToUser_FaultInst:
|
||||
strb w3, [x0], 1; // <- This line can fault.
|
||||
sub x2, x2, 1;
|
||||
cbnz x2, 1b;
|
||||
2:
|
||||
mov x0, 0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
extern "C" uint64_t CopyFromUser_FaultInst;
|
||||
void* const CopyFromUser_FaultLocation = &CopyFromUser_FaultInst;
|
||||
|
||||
extern "C" uint64_t CopyToUser_FaultInst;
|
||||
void* const CopyToUser_FaultLocation = &CopyToUser_FaultInst;
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED && defined(_M_ARM_64)
|
||||
__attribute__((naked)) bool VerifyIsReadableImpl(const void* Src, size_t Size) {
|
||||
__asm volatile(R"(
|
||||
// Early exit if size is zero.
|
||||
cbz x1, 2f;
|
||||
|
||||
1:
|
||||
.globl UserReadable_FaultInst
|
||||
UserReadable_FaultInst:
|
||||
ldrb wzr, [x0], 1; // <- This line can fault.
|
||||
sub x1, x1, 1;
|
||||
cbnz x1, 1b;
|
||||
|
||||
2:
|
||||
mov x0, 1;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) bool VerifyIsOnlyWritable(void* Src, size_t Size) {
|
||||
__asm volatile(R"(
|
||||
// Early exit if size is zero.
|
||||
cbz x1, 2f;
|
||||
|
||||
1:
|
||||
ldrb w2, [x0];
|
||||
.globl UserWritable_FaultInst
|
||||
UserWritable_FaultInst:
|
||||
strb w2, [x0], 1; // <- This line can fault.
|
||||
|
||||
sub x1, x1, 1;
|
||||
cbnz x1, 1b;
|
||||
|
||||
2:
|
||||
mov x0, 1;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) bool VerifyIsStringReadableMaxSizeImpl(const char* Src, size_t MaxSize) {
|
||||
__asm volatile(R"(
|
||||
1:
|
||||
cbz x1, 2f;
|
||||
|
||||
.globl UserStringReadable_FaultInst
|
||||
UserStringReadable_FaultInst:
|
||||
ldrb w2, [x0], 1; //< This line can fault.
|
||||
sub x1, x1, 1;
|
||||
cbnz x2, 1b;
|
||||
|
||||
2:
|
||||
mov x0, 1;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
void VerifyIsReadable(const void* Src, size_t Size) {
|
||||
LOGMAN_THROW_A_FMT(VerifyIsReadableImpl(Src, Size), "EFAULT needs readable!");
|
||||
}
|
||||
|
||||
void VerifyIsStringReadable(const char* Src) {
|
||||
LOGMAN_THROW_A_FMT(VerifyIsStringReadableMaxSizeImpl(Src, ~0ULL), "EFAULT needs string readable!");
|
||||
}
|
||||
|
||||
void VerifyIsStringReadableMaxSize(const char* Src, size_t MaxSize) {
|
||||
LOGMAN_THROW_A_FMT(VerifyIsStringReadableMaxSizeImpl(Src, MaxSize), "EFAULT needs string readable!");
|
||||
}
|
||||
|
||||
void VerifyIsReadableOrNull(const void* Src, size_t Size) {
|
||||
if (Src == nullptr) {
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(VerifyIsReadableImpl(Src, Size), "EFAULT needs readable!");
|
||||
}
|
||||
|
||||
void VerifyIsWritable(void* Src, size_t Size) {
|
||||
///< Checking if writable needs to check if readable first.
|
||||
VerifyIsReadable(Src, Size);
|
||||
|
||||
LOGMAN_THROW_A_FMT(VerifyIsOnlyWritable(Src, Size), "EFAULT needs writable!");
|
||||
}
|
||||
|
||||
void VerifyIsWritableOrNull(void* Src, size_t Size) {
|
||||
if (Src == nullptr) {
|
||||
return;
|
||||
}
|
||||
|
||||
///< Checking if writable needs to check if readable first.
|
||||
VerifyIsReadable(Src, Size);
|
||||
LOGMAN_THROW_A_FMT(VerifyIsOnlyWritable(Src, Size), "EFAULT needs writable!");
|
||||
}
|
||||
|
||||
extern "C" uint64_t UserReadable_FaultInst;
|
||||
void* const UserReadable_FaultLocation = &UserReadable_FaultInst;
|
||||
|
||||
extern "C" uint64_t UserWritable_FaultInst;
|
||||
void* const UserWritable_FaultLocation = &UserWritable_FaultInst;
|
||||
|
||||
extern "C" uint64_t UserStringReadable_FaultInst;
|
||||
void* const UserStringReadable_FaultLocation = &UserStringReadable_FaultInst;
|
||||
#endif
|
||||
|
||||
bool IsFaultLocation(uint64_t PC) {
|
||||
bool IsMemcpyFault = false;
|
||||
IsMemcpyFault |= reinterpret_cast<void*>(PC) == CopyToUser_FaultLocation;
|
||||
IsMemcpyFault |= reinterpret_cast<void*>(PC) == CopyFromUser_FaultLocation;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED && defined(_M_ARM_64)
|
||||
IsMemcpyFault |= reinterpret_cast<void*>(PC) == UserReadable_FaultLocation;
|
||||
IsMemcpyFault |= reinterpret_cast<void*>(PC) == UserWritable_FaultLocation;
|
||||
IsMemcpyFault |= reinterpret_cast<void*>(PC) == UserStringReadable_FaultLocation;
|
||||
#endif
|
||||
return IsMemcpyFault;
|
||||
}
|
||||
|
||||
#else
|
||||
size_t CopyFromUser(void* Dest, const void* Src, size_t Size) {
|
||||
memcpy(Dest, Src, Size);
|
||||
return Size;
|
||||
}
|
||||
|
||||
size_t CopyToUser(void* Dest, const void* Src, size_t Size) {
|
||||
memcpy(Dest, Src, Size);
|
||||
return Size;
|
||||
}
|
||||
|
||||
bool IsFaultLocation(uint64_t PC) {
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
} // namespace FEX::HLE::FaultSafeUserMemAccess
|
||||
@@ -0,0 +1,381 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: LinuxSyscalls|syscalls-shared
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "LinuxSyscalls/Seccomp/BPFEmitter.h"
|
||||
#include "LinuxSyscalls/Seccomp/SeccompEmulator.h"
|
||||
|
||||
#include <linux/bpf_common.h>
|
||||
#include <linux/filter.h>
|
||||
#include <linux/seccomp.h>
|
||||
|
||||
#define VALIDATE(cond) \
|
||||
do { \
|
||||
if (!(cond)) { \
|
||||
RETURN_ERROR(-EINVAL) \
|
||||
} \
|
||||
} while (0)
|
||||
namespace FEX::HLE {
|
||||
template<bool CalculateSize>
|
||||
uint64_t BPFEmitter::HandleLoad(uint32_t BPFIP, const sock_filter* Inst) {
|
||||
VALIDATE(BPF_SIZE(Inst->code) == BPF_W);
|
||||
[[maybe_unused]] size_t OpSize {};
|
||||
|
||||
const auto DestReg = BPF_CLASS(Inst->code) == BPF_LD ? REG_A : REG_X;
|
||||
|
||||
switch (BPF_MODE(Inst->code)) {
|
||||
case BPF_IMM: {
|
||||
auto Const = ConstPool.try_emplace(Inst->k, ARMEmitter::ForwardLabel {});
|
||||
EMIT_INST(ldr(DestReg, &Const.first->second));
|
||||
break;
|
||||
}
|
||||
case BPF_ABS: {
|
||||
// ABS has some restrictions
|
||||
// - Must be 4-byte aligned
|
||||
// - Must be less than the size of seccomp_data
|
||||
const auto Offset = Inst->k;
|
||||
|
||||
// Need to be 4-byte aligned.
|
||||
VALIDATE((Offset & 0b11) == 0);
|
||||
// Ensure accessing inside of seccomp_data.
|
||||
VALIDATE(Offset < sizeof(seccomp_data));
|
||||
|
||||
EMIT_INST(ldr(DestReg, REG_SECCOMP_DATA, Offset));
|
||||
break;
|
||||
}
|
||||
case BPF_MEM:
|
||||
// Must be smaller than scratch space size.
|
||||
VALIDATE(Inst->k < 16);
|
||||
|
||||
EMIT_INST(ldr(DestReg, REG_SECCOMP_DATA, offsetof(WorkingBuffer, ScratchMemory[Inst->k])));
|
||||
break;
|
||||
case BPF_LEN:
|
||||
// Just returns the length of seccomp_data.
|
||||
EMIT_INST(movz(DestReg, sizeof(seccomp_data)));
|
||||
break;
|
||||
case BPF_IND:
|
||||
case BPF_MSH:
|
||||
default: RETURN_ERROR(-EINVAL); // Unsupported
|
||||
}
|
||||
|
||||
RETURN_SUCCESS();
|
||||
}
|
||||
|
||||
template<bool CalculateSize>
|
||||
uint64_t BPFEmitter::HandleStore(uint32_t BPFIP, const sock_filter* Inst) {
|
||||
VALIDATE(BPF_SIZE(Inst->code) == BPF_W);
|
||||
|
||||
[[maybe_unused]] size_t OpSize {};
|
||||
|
||||
const auto SrcReg = BPF_CLASS(Inst->code) == BPF_LD ? REG_A : REG_X;
|
||||
// Must be smaller than scratch space size.
|
||||
VALIDATE(Inst->k < 16);
|
||||
|
||||
EMIT_INST(str(SrcReg, REG_SECCOMP_DATA, offsetof(WorkingBuffer, ScratchMemory[Inst->k])));
|
||||
|
||||
RETURN_SUCCESS();
|
||||
}
|
||||
|
||||
template<bool CalculateSize>
|
||||
uint64_t BPFEmitter::HandleALU(uint32_t BPFIP, const sock_filter* Inst) {
|
||||
[[maybe_unused]] size_t OpSize {};
|
||||
const auto SrcType = BPF_SRC(Inst->code);
|
||||
const auto Op = BPF_OP(Inst->code);
|
||||
|
||||
switch (Op) {
|
||||
case BPF_ADD:
|
||||
case BPF_SUB:
|
||||
case BPF_MUL:
|
||||
case BPF_DIV:
|
||||
case BPF_OR:
|
||||
case BPF_AND:
|
||||
case BPF_LSH:
|
||||
case BPF_RSH:
|
||||
case BPF_MOD:
|
||||
case BPF_XOR: {
|
||||
auto SrcReg = REG_X;
|
||||
if (SrcType == BPF_K) {
|
||||
SrcReg = REG_TMP;
|
||||
auto Const = ConstPool.try_emplace(Inst->k, ARMEmitter::ForwardLabel {});
|
||||
EMIT_INST(ldr(SrcReg, &Const.first->second));
|
||||
}
|
||||
|
||||
switch (Op) {
|
||||
case BPF_ADD: EMIT_INST(add(ARMEmitter::Size::i32Bit, REG_A, REG_A, SrcReg)); break;
|
||||
case BPF_SUB: EMIT_INST(sub(ARMEmitter::Size::i32Bit, REG_A, REG_A, SrcReg)); break;
|
||||
case BPF_MUL: EMIT_INST(mul(ARMEmitter::Size::i32Bit, REG_A, REG_A, SrcReg)); break;
|
||||
case BPF_DIV:
|
||||
// Specifically unsigned.
|
||||
EMIT_INST(udiv(ARMEmitter::Size::i32Bit, REG_A, REG_A, SrcReg));
|
||||
break;
|
||||
case BPF_OR: EMIT_INST(orr(ARMEmitter::Size::i32Bit, REG_A, REG_A, SrcReg)); break;
|
||||
case BPF_AND: EMIT_INST(and_(ARMEmitter::Size::i32Bit, REG_A, REG_A, SrcReg)); break;
|
||||
case BPF_LSH: EMIT_INST(lslv(ARMEmitter::Size::i32Bit, REG_A, REG_A, SrcReg)); break;
|
||||
case BPF_RSH: EMIT_INST(lsrv(ARMEmitter::Size::i32Bit, REG_A, REG_A, SrcReg)); break;
|
||||
case BPF_MOD:
|
||||
// Specifically unsigned.
|
||||
EMIT_INST(udiv(ARMEmitter::Size::i32Bit, REG_TMP2, REG_A, SrcReg));
|
||||
EMIT_INST(msub(ARMEmitter::Size::i32Bit, REG_A, REG_TMP2, SrcReg, REG_A));
|
||||
break;
|
||||
case BPF_XOR: EMIT_INST(eor(ARMEmitter::Size::i32Bit, REG_A, REG_A, SrcReg)); break;
|
||||
default: RETURN_ERROR(-EINVAL);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case BPF_NEG:
|
||||
// Only BPF_K supported on NEG.
|
||||
VALIDATE(SrcType == BPF_K);
|
||||
|
||||
EMIT_INST(neg(ARMEmitter::Size::i32Bit, REG_A, REG_A));
|
||||
break;
|
||||
|
||||
default: RETURN_ERROR(-EINVAL);
|
||||
}
|
||||
|
||||
RETURN_SUCCESS();
|
||||
}
|
||||
|
||||
template<bool CalculateSize>
|
||||
uint64_t BPFEmitter::HandleJmp(uint32_t BPFIP, uint32_t NumInst, const sock_filter* Inst) {
|
||||
[[maybe_unused]] size_t OpSize {};
|
||||
const auto SrcType = BPF_SRC(Inst->code);
|
||||
const auto Op = BPF_OP(Inst->code);
|
||||
|
||||
switch (Op) {
|
||||
case BPF_JA: {
|
||||
// Only BPF_K supported on JA.
|
||||
VALIDATE(SrcType == BPF_K);
|
||||
|
||||
// BPF IP register is effectively only 32-bit. Treat k constant like a signed integer.
|
||||
// This allows it to jump anywhere in the program.
|
||||
// But! Loops are EXPLICITLY disallowed inside of BPF programs.
|
||||
// This is to prevent DOS style attacks through BPF programs.
|
||||
uint64_t Target = BPFIP + Inst->k + 1;
|
||||
// Must not jump past the end.
|
||||
VALIDATE(Target < NumInst);
|
||||
|
||||
fextl::unordered_map<uint32_t, ARMEmitter::ForwardLabel>::iterator TargetLabel {};
|
||||
|
||||
if constexpr (!CalculateSize) {
|
||||
TargetLabel = JumpLabels.try_emplace(Target, ARMEmitter::ForwardLabel {}).first;
|
||||
}
|
||||
|
||||
EMIT_INST(b(&TargetLabel->second));
|
||||
break;
|
||||
}
|
||||
case BPF_JEQ:
|
||||
case BPF_JGT:
|
||||
case BPF_JGE:
|
||||
case BPF_JSET: {
|
||||
auto CompareSrcReg = REG_X;
|
||||
if (SrcType == BPF_K) {
|
||||
CompareSrcReg = REG_TMP;
|
||||
auto Const = ConstPool.try_emplace(Inst->k, ARMEmitter::ForwardLabel {});
|
||||
EMIT_INST(ldr(CompareSrcReg, &Const.first->second));
|
||||
}
|
||||
uint32_t TargetTrue = BPFIP + Inst->jt + 1;
|
||||
uint32_t TargetFalse = BPFIP + Inst->jf + 1;
|
||||
|
||||
// Must not jump past the end.
|
||||
VALIDATE(TargetTrue < NumInst && TargetFalse < NumInst);
|
||||
|
||||
ARMEmitter::Condition CompareResultOp;
|
||||
if (Op == BPF_JEQ) {
|
||||
CompareResultOp = ARMEmitter::Condition::CC_EQ;
|
||||
EMIT_INST(cmp(ARMEmitter::Size::i32Bit, REG_A, CompareSrcReg));
|
||||
} else if (Op == BPF_JGT) {
|
||||
CompareResultOp = ARMEmitter::Condition::CC_HI;
|
||||
EMIT_INST(cmp(ARMEmitter::Size::i32Bit, REG_A, CompareSrcReg));
|
||||
} else if (Op == BPF_JGE) {
|
||||
CompareResultOp = ARMEmitter::Condition::CC_HS;
|
||||
EMIT_INST(cmp(ARMEmitter::Size::i32Bit, REG_A, CompareSrcReg));
|
||||
} else if (Op == BPF_JSET) {
|
||||
CompareResultOp = ARMEmitter::Condition::CC_NE;
|
||||
EMIT_INST(tst(ARMEmitter::Size::i32Bit, REG_A, CompareSrcReg));
|
||||
} else {
|
||||
RETURN_ERROR(-EINVAL);
|
||||
}
|
||||
|
||||
fextl::unordered_map<uint32_t, ARMEmitter::ForwardLabel>::iterator TargetTrueLabel {};
|
||||
fextl::unordered_map<uint32_t, ARMEmitter::ForwardLabel>::iterator TargetFalseLabel {};
|
||||
|
||||
if constexpr (!CalculateSize) {
|
||||
TargetTrueLabel = JumpLabels.try_emplace(TargetTrue, ARMEmitter::ForwardLabel {}).first;
|
||||
TargetFalseLabel = JumpLabels.try_emplace(TargetFalse, ARMEmitter::ForwardLabel {}).first;
|
||||
}
|
||||
|
||||
EMIT_INST(b(CompareResultOp, &TargetTrueLabel->second));
|
||||
EMIT_INST(b(&TargetFalseLabel->second));
|
||||
break;
|
||||
}
|
||||
default: RETURN_ERROR(-EINVAL); // Unknown jump type
|
||||
}
|
||||
|
||||
RETURN_SUCCESS();
|
||||
}
|
||||
|
||||
template<bool CalculateSize>
|
||||
uint64_t BPFEmitter::HandleRet(uint32_t BPFIP, const sock_filter* Inst) {
|
||||
[[maybe_unused]] size_t OpSize {};
|
||||
const auto RValSrc = BPF_RVAL(Inst->code);
|
||||
switch (RValSrc) {
|
||||
case BPF_K: {
|
||||
auto Const = ConstPool.try_emplace(Inst->k, ARMEmitter::ForwardLabel {});
|
||||
EMIT_INST(ldr(ARMEmitter::WReg::w0, &Const.first->second));
|
||||
break;
|
||||
}
|
||||
case BPF_X: EMIT_INST(mov(ARMEmitter::WReg::w0, REG_X)); break;
|
||||
case BPF_A:
|
||||
// w0 is already REG_A
|
||||
static_assert(REG_A == ARMEmitter::WReg::w0, "This is expected to be the same");
|
||||
break;
|
||||
default: RETURN_ERROR(-EINVAL);
|
||||
}
|
||||
|
||||
EMIT_INST(ret());
|
||||
|
||||
RETURN_SUCCESS();
|
||||
}
|
||||
|
||||
template<bool CalculateSize>
|
||||
uint64_t BPFEmitter::HandleMisc(uint32_t BPFIP, const sock_filter* Inst) {
|
||||
[[maybe_unused]] size_t OpSize {};
|
||||
const auto MiscOp = BPF_MISCOP(Inst->code);
|
||||
switch (MiscOp) {
|
||||
case BPF_TAX: EMIT_INST(mov(REG_X, REG_A)); break;
|
||||
case BPF_TXA: EMIT_INST(mov(REG_A, REG_X)); break;
|
||||
default: RETURN_ERROR(-EINVAL) // Unsupported misc operation.
|
||||
}
|
||||
|
||||
RETURN_SUCCESS();
|
||||
}
|
||||
|
||||
template<bool CalculateSize, class Pred>
|
||||
uint64_t BPFEmitter::HandleEmission(uint32_t flags, const sock_fprog* prog) {
|
||||
constexpr Pred PredFunc;
|
||||
uint64_t CalculatedSize {};
|
||||
|
||||
for (uint32_t i = 0; i < prog->len; ++i) {
|
||||
if constexpr (!CalculateSize) {
|
||||
auto jump_label = JumpLabels.find(i);
|
||||
if (jump_label != JumpLabels.end()) {
|
||||
Bind(&jump_label->second);
|
||||
}
|
||||
}
|
||||
|
||||
bool HadError {};
|
||||
uint64_t Result {};
|
||||
|
||||
const sock_filter* Inst = &prog->filter[i];
|
||||
const uint16_t Code = Inst->code;
|
||||
const uint16_t Class = BPF_CLASS(Code);
|
||||
switch (Class) {
|
||||
case BPF_LD:
|
||||
case BPF_LDX: {
|
||||
Result = HandleLoad<CalculateSize>(i, Inst);
|
||||
break;
|
||||
}
|
||||
case BPF_ST:
|
||||
case BPF_STX: {
|
||||
Result = HandleStore<CalculateSize>(i, Inst);
|
||||
break;
|
||||
}
|
||||
case BPF_ALU: {
|
||||
Result = HandleALU<CalculateSize>(i, Inst);
|
||||
break;
|
||||
}
|
||||
case BPF_JMP: {
|
||||
Result = HandleJmp<CalculateSize>(i, prog->len, Inst);
|
||||
break;
|
||||
}
|
||||
case BPF_RET: {
|
||||
Result = HandleRet<CalculateSize>(i, Inst);
|
||||
break;
|
||||
}
|
||||
case BPF_MISC: {
|
||||
Result = HandleMisc<CalculateSize>(i, Inst);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
// We handle all instruction classes.
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
HadError = PredFunc(Result);
|
||||
|
||||
if constexpr (CalculateSize) {
|
||||
CalculatedSize += Result;
|
||||
}
|
||||
|
||||
if (HadError) {
|
||||
if constexpr (!CalculateSize) {
|
||||
// Had error, early return and free the memory.
|
||||
FEXCore::Allocator::munmap(GetBufferBase(), FuncSize);
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
|
||||
if constexpr (CalculateSize) {
|
||||
// Add the constant pool size.
|
||||
CalculatedSize += ConstPool.size() * 4;
|
||||
|
||||
// Size calculation could have added constants and jump labels. Erase them now.
|
||||
ConstPool.clear();
|
||||
JumpLabels.clear();
|
||||
|
||||
return CalculatedSize;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t BPFEmitter::JITFilter(uint32_t flags, const sock_fprog* prog) {
|
||||
FuncSize = HandleEmission<true, SizeErrorCheck>(flags, prog);
|
||||
|
||||
if (FuncSize == ~0ULL) {
|
||||
// Buffer size calculation found invalid code.
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
SetBuffer((uint8_t*)FEXCore::Allocator::mmap(nullptr, FuncSize, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0), FuncSize);
|
||||
|
||||
const auto CodeBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
uint64_t Result = HandleEmission<false, EmissionErrorCheck>(flags, prog);
|
||||
|
||||
if (Result != 0) {
|
||||
// Had error, early return and free the memory.
|
||||
FEXCore::Allocator::munmap(GetBufferBase(), FuncSize);
|
||||
return Result;
|
||||
}
|
||||
|
||||
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeBegin;
|
||||
|
||||
// Emit the constant pool.
|
||||
Align();
|
||||
for (auto& Const : ConstPool) {
|
||||
Bind(&Const.second);
|
||||
dc32(Const.first);
|
||||
}
|
||||
|
||||
ClearICache(CodeBegin, CodeOnlySize);
|
||||
::mprotect(CodeBegin, AllocationSize(), PROT_READ | PROT_EXEC);
|
||||
Func = CodeBegin;
|
||||
|
||||
if constexpr (false) {
|
||||
// Useful for debugging seccomp filters.
|
||||
LogMan::Msg::DFmt("JITFilter: disas 0x{:x},+{}", (uint64_t)CodeBegin, CodeOnlySize);
|
||||
}
|
||||
|
||||
ConstPool.clear();
|
||||
JumpLabels.clear();
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
} // namespace FEX::HLE
|
||||
@@ -0,0 +1,98 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: LinuxSyscalls|syscalls-shared
|
||||
$end_info$
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <linux/filter.h>
|
||||
#include <linux/seccomp.h>
|
||||
|
||||
struct sock_fprog;
|
||||
struct sock_filter;
|
||||
|
||||
namespace FEX::HLE {
|
||||
class BPFEmitter final : public ARMEmitter::Emitter {
|
||||
public:
|
||||
struct WorkingBuffer {
|
||||
struct seccomp_data Data;
|
||||
uint32_t ScratchMemory[BPF_MEMWORDS]; // Defined as 16 words.
|
||||
};
|
||||
|
||||
BPFEmitter() = default;
|
||||
|
||||
uint64_t JITFilter(uint32_t flags, const sock_fprog* prog);
|
||||
void* GetFunc() const {
|
||||
return Func;
|
||||
}
|
||||
|
||||
size_t AllocationSize() const {
|
||||
return FuncSize;
|
||||
}
|
||||
|
||||
private:
|
||||
template<bool CalculateSize>
|
||||
uint64_t HandleLoad(uint32_t BPFIP, const sock_filter* Inst);
|
||||
template<bool CalculateSize>
|
||||
uint64_t HandleStore(uint32_t BPFIP, const sock_filter* Inst);
|
||||
template<bool CalculateSize>
|
||||
uint64_t HandleALU(uint32_t BPFIP, const sock_filter* Inst);
|
||||
template<bool CalculateSize>
|
||||
uint64_t HandleJmp(uint32_t BPFIP, uint32_t NumInst, const sock_filter* Inst);
|
||||
template<bool CalculateSize>
|
||||
uint64_t HandleRet(uint32_t BPFIP, const sock_filter* Inst);
|
||||
template<bool CalculateSize>
|
||||
uint64_t HandleMisc(uint32_t BPFIP, const sock_filter* Inst);
|
||||
|
||||
#define EMIT_INST(x) \
|
||||
do { \
|
||||
if constexpr (CalculateSize) { \
|
||||
OpSize += 4; \
|
||||
} else { \
|
||||
x; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
#define RETURN_ERROR(x) \
|
||||
if constexpr (CalculateSize) { \
|
||||
return ~0ULL; \
|
||||
} else { \
|
||||
static_assert(x == -EINVAL, "Early return error evaluation only supports EINVAL"); \
|
||||
return x; \
|
||||
}
|
||||
|
||||
#define RETURN_SUCCESS() \
|
||||
do { \
|
||||
if constexpr (CalculateSize) { \
|
||||
return OpSize; \
|
||||
} else { \
|
||||
return 0; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
using SizeErrorCheck = decltype([](uint64_t Result) -> bool { return Result == ~0ULL; });
|
||||
using EmissionErrorCheck = decltype([](uint64_t Result) { return Result != 0; });
|
||||
|
||||
template<bool CalculateSize, class Pred>
|
||||
uint64_t HandleEmission(uint32_t flags, const sock_fprog* prog);
|
||||
|
||||
// Register selection comes from function signature.
|
||||
constexpr static auto REG_A = ARMEmitter::WReg::w0;
|
||||
constexpr static auto REG_X = ARMEmitter::WReg::w1;
|
||||
constexpr static auto REG_TMP = ARMEmitter::WReg::w2;
|
||||
constexpr static auto REG_TMP2 = ARMEmitter::WReg::w3;
|
||||
constexpr static auto REG_SECCOMP_DATA = ARMEmitter::XReg::x4;
|
||||
fextl::unordered_map<uint32_t, ARMEmitter::ForwardLabel> JumpLabels;
|
||||
fextl::unordered_map<uint32_t, ARMEmitter::ForwardLabel> ConstPool;
|
||||
|
||||
void* Func;
|
||||
size_t FuncSize;
|
||||
};
|
||||
|
||||
|
||||
} // namespace FEX::HLE
|
||||
@@ -0,0 +1,185 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: LinuxSyscalls|syscalls-shared
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "LinuxSyscalls/Seccomp/SeccompEmulator.h"
|
||||
|
||||
#include <linux/bpf_common.h>
|
||||
#include <linux/filter.h>
|
||||
#include <linux/seccomp.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
void SeccompEmulator::DumpProgram(const sock_fprog* prog) {
|
||||
auto Parse_Class_LD = [](uint32_t BPFIP, const sock_filter* Inst) {
|
||||
auto DestName = [](sock_filter const* Inst) {
|
||||
if (BPF_CLASS(Inst->code) == BPF_LD) {
|
||||
return "A";
|
||||
} else {
|
||||
return "X";
|
||||
}
|
||||
};
|
||||
|
||||
auto AccessSize = [](sock_filter const* Inst) {
|
||||
switch (BPF_SIZE(Inst->code)) {
|
||||
case BPF_W: return 32;
|
||||
case BPF_H: return 16;
|
||||
case BPF_B: return 8;
|
||||
case 0x18: /* BPF_DW */ return 64;
|
||||
}
|
||||
return 0;
|
||||
};
|
||||
|
||||
auto ModeType = [](sock_filter const* Inst) {
|
||||
switch (BPF_MODE(Inst->code)) {
|
||||
case BPF_IMM: return "IMM";
|
||||
case BPF_ABS: return "ABS";
|
||||
case BPF_IND: return "IND";
|
||||
case BPF_MEM: return "MEM";
|
||||
case BPF_LEN: return "LEN";
|
||||
case BPF_MSH: return "MSH";
|
||||
}
|
||||
return "Unknown";
|
||||
};
|
||||
|
||||
auto LoadName = [](sock_filter const* Inst) {
|
||||
using namespace std::string_view_literals;
|
||||
switch (BPF_MODE(Inst->code)) {
|
||||
case BPF_IMM: return fextl::fmt::format("#{}", Inst->k);
|
||||
case BPF_ABS: return fextl::fmt::format("seccomp_data + #{}", Inst->k);
|
||||
case BPF_IND: return fextl::fmt::format("Ind[X+#{}]", Inst->k);
|
||||
case BPF_MEM: return fextl::fmt::format("Mem[#{}]", Inst->k);
|
||||
case BPF_LEN: return fextl::fmt::format("len");
|
||||
case BPF_MSH: return fextl::fmt::format("msh");
|
||||
}
|
||||
return fextl::fmt::format("Unknown");
|
||||
};
|
||||
|
||||
LogMan::Msg::IFmt("0x{:04x}: {} <- LD.{} {} {}", BPFIP, DestName(Inst), AccessSize(Inst), ModeType(Inst), LoadName(Inst));
|
||||
};
|
||||
|
||||
auto Parse_Class_ST = [](uint32_t BPFIP, const sock_filter* Inst) {
|
||||
auto DestName = [](sock_filter const* Inst) {
|
||||
if (BPF_CLASS(Inst->code) == BPF_ST) {
|
||||
return "A";
|
||||
} else {
|
||||
return "X";
|
||||
}
|
||||
};
|
||||
|
||||
LogMan::Msg::IFmt("0x{:04x}: Mem[{}] <- ST.{}", BPFIP, Inst->k, DestName(Inst));
|
||||
};
|
||||
|
||||
auto Parse_Class_ALU = [](uint32_t BPFIP, const sock_filter* Inst) {
|
||||
auto GetOp = [](sock_filter const* Inst) {
|
||||
const auto Op = BPF_OP(Inst->code);
|
||||
|
||||
switch (Op) {
|
||||
case BPF_ADD: return "ADD";
|
||||
case BPF_SUB: return "SUB";
|
||||
case BPF_MUL: return "MUL";
|
||||
case BPF_DIV: return "DIV";
|
||||
case BPF_OR: return "OR";
|
||||
case BPF_AND: return "AND";
|
||||
case BPF_LSH: return "LSH";
|
||||
case BPF_RSH: return "RSH";
|
||||
case BPF_MOD: return "MOD";
|
||||
case BPF_XOR: return "XOR";
|
||||
case BPF_NEG: return "NEG";
|
||||
default: return "Unknown";
|
||||
}
|
||||
};
|
||||
|
||||
auto GetSrc = [](sock_filter const* Inst) {
|
||||
switch (BPF_SRC(Inst->code)) {
|
||||
case BPF_K: return fextl::fmt::format("0x{:x}", Inst->k);
|
||||
case BPF_X: return fextl::fmt::format("<X>");
|
||||
}
|
||||
return fextl::fmt::format("Unknown");
|
||||
};
|
||||
|
||||
LogMan::Msg::IFmt("0x{:04x}: {} <A>, {}", BPFIP, GetOp(Inst), GetSrc(Inst));
|
||||
};
|
||||
|
||||
auto Parse_Class_JMP = [](uint32_t BPFIP, const sock_filter* Inst) {
|
||||
auto GetOp = [](sock_filter const* Inst) {
|
||||
switch (BPF_OP(Inst->code)) {
|
||||
case BPF_JA: return "a";
|
||||
case BPF_JEQ: return "eq";
|
||||
case BPF_JGT: return "gt";
|
||||
case BPF_JGE: return "ge";
|
||||
case BPF_JSET: return "set";
|
||||
}
|
||||
return "Unknown";
|
||||
};
|
||||
|
||||
auto GetSrc = [](sock_filter const* Inst) {
|
||||
switch (BPF_SRC(Inst->code)) {
|
||||
case BPF_K: return fextl::fmt::format("0x{:x}", Inst->k);
|
||||
case BPF_X: return fextl::fmt::format("<X>");
|
||||
}
|
||||
return fextl::fmt::format("Unknown");
|
||||
};
|
||||
|
||||
LogMan::Msg::IFmt("0x{:04x}: JMP.{} {}, +{} (#0x{:x}), +{} (#0x{:x})", BPFIP, GetOp(Inst), GetSrc(Inst), Inst->jt, BPFIP + Inst->jt + 1,
|
||||
Inst->jf, BPFIP + Inst->jf + 1);
|
||||
};
|
||||
|
||||
auto Parse_Class_RET = [](uint32_t BPFIP, const sock_filter* Inst) {
|
||||
auto GetRetValue = [](sock_filter const* Inst) {
|
||||
switch (BPF_RVAL(Inst->code)) {
|
||||
case BPF_K: {
|
||||
uint32_t RetData = Inst->k & SECCOMP_RET_DATA;
|
||||
switch (Inst->k & SECCOMP_RET_ACTION_FULL) {
|
||||
case SECCOMP_RET_KILL_PROCESS: return fextl::fmt::format("KILL_PROCESS.{}", RetData);
|
||||
case SECCOMP_RET_KILL_THREAD: return fextl::fmt::format("KILL_THREAD.{}", RetData);
|
||||
case SECCOMP_RET_TRAP: return fextl::fmt::format("TRAP.{}", RetData);
|
||||
case SECCOMP_RET_ERRNO: return fextl::fmt::format("ERRNO.{}", RetData);
|
||||
case SECCOMP_RET_USER_NOTIF: return fextl::fmt::format("USER_NOTIF.{}", RetData);
|
||||
case SECCOMP_RET_TRACE: return fextl::fmt::format("TRACE.{}", RetData);
|
||||
case SECCOMP_RET_LOG: return fextl::fmt::format("LOG.{}", RetData);
|
||||
case SECCOMP_RET_ALLOW: return fextl::fmt::format("ALLOW.{}", RetData);
|
||||
default: break;
|
||||
}
|
||||
return fextl::fmt::format("<Unknown>.{}", RetData);
|
||||
}
|
||||
case BPF_X: return fextl::fmt::format("<X>");
|
||||
case BPF_A: return fextl::fmt::format("<A>");
|
||||
}
|
||||
|
||||
return fextl::fmt::format("Unknown");
|
||||
};
|
||||
|
||||
LogMan::Msg::IFmt("0x{:04x}: RET {}", BPFIP, GetRetValue(Inst));
|
||||
};
|
||||
|
||||
auto Parse_Class_MISC = [](uint32_t BPFIP, const sock_filter* Inst) {
|
||||
const auto MiscOp = BPF_MISCOP(Inst->code);
|
||||
switch (MiscOp) {
|
||||
case BPF_TAX: LogMan::Msg::IFmt("0x{:04x}: TAX", BPFIP); break;
|
||||
case BPF_TXA: LogMan::Msg::IFmt("0x{:04x}: TXA", BPFIP); break;
|
||||
default: LogMan::Msg::IFmt("0x{:04x}: Misc: Unknown", BPFIP); break;
|
||||
};
|
||||
};
|
||||
|
||||
LogMan::Msg::IFmt("BPF program: 0x{:x} instructions", prog->len);
|
||||
|
||||
for (size_t i = 0; i < prog->len; ++i) {
|
||||
const sock_filter* Inst = &prog->filter[i];
|
||||
const uint16_t Code = Inst->code;
|
||||
const uint16_t Class = BPF_CLASS(Code);
|
||||
switch (Class) {
|
||||
case BPF_LD:
|
||||
case BPF_LDX: Parse_Class_LD(i, Inst); break;
|
||||
case BPF_ST:
|
||||
case BPF_STX: Parse_Class_ST(i, Inst); break;
|
||||
case BPF_ALU: Parse_Class_ALU(i, Inst); break;
|
||||
case BPF_JMP: Parse_Class_JMP(i, Inst); break;
|
||||
case BPF_RET: Parse_Class_RET(i, Inst); break;
|
||||
case BPF_MISC: Parse_Class_MISC(i, Inst); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace FEX::HLE
|
||||
@@ -0,0 +1,718 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: LinuxSyscalls|syscalls-shared
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "LinuxSyscalls/Seccomp/BPFEmitter.h"
|
||||
#include "LinuxSyscalls/Seccomp/SeccompEmulator.h"
|
||||
|
||||
#include "LinuxSyscalls/x32/Syscalls.h"
|
||||
#include "LinuxSyscalls/x64/Syscalls.h"
|
||||
#include "LinuxSyscalls/SignalDelegator.h"
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <linux/audit.h>
|
||||
#include <linux/bpf_common.h>
|
||||
#include <linux/filter.h>
|
||||
#include <linux/seccomp.h>
|
||||
#include <sys/prctl.h>
|
||||
|
||||
// seccomp
|
||||
//
|
||||
// global
|
||||
// - kcmp - pass
|
||||
// - mode_strict_support - pass
|
||||
// - mode_strict_cannot_call_prctl - pass
|
||||
// - no_new_privs_support - pass
|
||||
// - mode_filter_support - pass
|
||||
// - mode_filter_without_nnp - pass
|
||||
// - filter_size_limits - pass
|
||||
// - filter_chain_limits - pass
|
||||
// - mode_filter_cannot_move_to_strict - pass
|
||||
// - mode_filter_get_seccomp - pass
|
||||
// - ALLOW_all - pass
|
||||
// - empty_prog - pass
|
||||
// - log_all - pass
|
||||
// - unknown_ret_is_kill_inside - pass
|
||||
// - unknown_ret_is_kill_above_allow - pass
|
||||
// - KILL_all - pass
|
||||
// - KILL_one - pass
|
||||
// - KILL_one_arg_one - pass
|
||||
// - KILL_one_arg_six - pass
|
||||
// - KILL_thread - FAIL (unrelated to bpf)
|
||||
// - KILL_process - FAIL (unrelated to bpf)
|
||||
// - KILL_unknown - FAIL (unrelated to bpf)
|
||||
// - arg_out_of_range - pass
|
||||
// - ERRNO_valid - pass
|
||||
// - ERRNO_zero - pass
|
||||
// - ERRNO_capped - pass
|
||||
// - ERRNO_order - pass
|
||||
// - seccomp_syscall - pass
|
||||
// - seccomp_syscall_mode_lock - pass
|
||||
// - detect_seccomp_filter_flags - pass
|
||||
// - TSYNC_first - pass
|
||||
// - syscall_restart - FAIL (PTRACE)
|
||||
// - filter_flag_log - pass
|
||||
// - get_action_avail - FAIL (ptrace and user-notif)
|
||||
// TSYNC
|
||||
// - siblings_fail_prctl - pass
|
||||
// - two_siblings_with_ancestor - FAIL (kill-thread not working quite right)
|
||||
// - two_sibling_want_nnp - pass
|
||||
// - two_siblings_with_one_divergence - pass
|
||||
// - two_siblings_with_one_divergence_no_tid_in_err - pass
|
||||
// - two_siblings_not_under_filter - FAIL (kill-thread not working quite right)
|
||||
// - two_siblings_with_no_filter - FAIL (kill-thread not working quite right)
|
||||
//
|
||||
// user-notif stuff
|
||||
// - get_metadata - SKIP (Needs root)
|
||||
// - user_notification_basic - FAIL (user-notif)
|
||||
// - user_notification_with_tsync - FAIL (user-notif)
|
||||
// - user_notification_kill_in_middle - FAIL (user-notif)
|
||||
// - user_notification_signal - FAIL (user-notif)
|
||||
// - user_notification_closed_listener - FAIL (user-notif)
|
||||
// - user_notification_child_pid_ns - FAIL (user-notif)
|
||||
// - user_notification_sibling_pid_ns - FAIL (user-notif)
|
||||
// - user_notification_fault_recv - FAIL (user-notif)
|
||||
// - seccomp_get_notif_sizes - pass
|
||||
// - user_notification_continue - FAIL (user-notif)
|
||||
// - user_notification_filter_empty - FAIL (user-notif)
|
||||
// - user_notification_filter_empty_threaded - FAIL (user-notif)
|
||||
// - user_notification_addfd - FAIL (user-notif)
|
||||
// - user_notification_addfd_rlimit - FAIL (user-notif)
|
||||
// - user_notification_sync - FAIL (user-notif)
|
||||
// - user_notification_fifo - FAIL (user-notif)
|
||||
// - user_notification_wait_killable_pre_notification - FAIL (user-notif)
|
||||
// - user_notification_wait_killable - FAIL (user-notif)
|
||||
// - user_notification_wait_killable_fatal - FAIL (user-notif)
|
||||
//
|
||||
// O_SUSPEND_SECCOMP
|
||||
// - setoptions - FAIL (ptrace)
|
||||
// - seize - FAIL (ptrace)
|
||||
// TRAP
|
||||
// - dfl - pass
|
||||
// - ign - pass
|
||||
// - handler - pass
|
||||
//
|
||||
// precedence
|
||||
// - allow_ok - pass
|
||||
// - kill_is_highest - pass
|
||||
// - kill_is_highest_in_any_order - pass
|
||||
// - trap_is_second - pass
|
||||
// - trap_is_second_in_any_order - pass
|
||||
// - errno_is_third - pass
|
||||
// - errno_is_third_in_any_order - pass
|
||||
// - trace_is_fourth - pass
|
||||
// - trace_is_fourth_in_any_order - pass
|
||||
// - log_is_fifth - pass
|
||||
// - log_is_fifth_in_any_order - pass
|
||||
//
|
||||
// TRACE_poke
|
||||
// - ptrace unsupported
|
||||
// TRACE_syscall
|
||||
// - ptrace unsupported
|
||||
|
||||
namespace FEX::HLE {
|
||||
uint64_t SeccompEmulator::Handle(FEXCore::Core::CpuStateFrame* Frame, uint32_t Op, uint32_t flags, void* arg) {
|
||||
// If seccomp isn't enabled then say so.
|
||||
if (!NeedsSeccomp) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
switch (Op) {
|
||||
case SECCOMP_SET_MODE_STRICT: return SetModeStrict(Frame, flags, arg);
|
||||
case SECCOMP_SET_MODE_FILTER: return SetModeFilter(Frame, flags, static_cast<const sock_fprog*>(arg));
|
||||
case SECCOMP_GET_ACTION_AVAIL: return GetActionAvail(flags, static_cast<const uint32_t*>(arg));
|
||||
case SECCOMP_GET_NOTIF_SIZES: return GetNotifSizes(flags, static_cast<struct seccomp_notif_sizes*>(arg));
|
||||
default:
|
||||
// operation is unknown or is not supported by this kernel version or configuration.
|
||||
return -EINVAL;
|
||||
}
|
||||
}
|
||||
|
||||
// Equivalent to prctl(PR_GET_SECCOMP)
|
||||
uint64_t SeccompEmulator::GetSeccomp(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
// If seccomp isn't enabled then say so.
|
||||
if (!NeedsSeccomp) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
return Thread->SeccompMode;
|
||||
}
|
||||
|
||||
void SeccompEmulator::InheritSeccompFilters(FEX::HLE::ThreadStateObject* Parent, FEX::HLE::ThreadStateObject* Child) {
|
||||
// Don't interrupt me while I'm copying.
|
||||
auto lk = FEXCore::MaskSignalsAndLockMutex(FilterMutex);
|
||||
|
||||
Child->Filters.resize(Parent->Filters.size());
|
||||
|
||||
for (size_t i = 0; i < Child->Filters.size(); ++i) {
|
||||
auto& ParentFilter = Parent->Filters[i];
|
||||
auto& ChildFilter = Child->Filters[i];
|
||||
ChildFilter = ParentFilter;
|
||||
std::atomic_ref<uint64_t>(ParentFilter->RefCount)++;
|
||||
}
|
||||
|
||||
// Copy the operating mode.
|
||||
Child->SeccompMode = Parent->SeccompMode;
|
||||
}
|
||||
|
||||
void SeccompEmulator::FreeSeccompFilters(FEX::HLE::ThreadStateObject* Thread) {
|
||||
// Don't talk to me when I'm busy deleting myself.
|
||||
auto lk = FEXCore::MaskSignalsAndLockMutex(FilterMutex);
|
||||
|
||||
bool HasFiltersToDelete {};
|
||||
for (auto& Filter : Thread->Filters) {
|
||||
auto RefCount = std::atomic_ref<uint64_t>(Filter->RefCount).fetch_sub(1);
|
||||
|
||||
if (RefCount == 1) {
|
||||
HasFiltersToDelete = true;
|
||||
}
|
||||
}
|
||||
Thread->Filters.clear();
|
||||
|
||||
if (HasFiltersToDelete) {
|
||||
// Garbage collect filters
|
||||
std::erase_if(Filters, [](auto& Filter) {
|
||||
if (std::atomic_ref<uint64_t>(Filter.RefCount).load(std::memory_order_relaxed) != 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(Filter.Func), Filter.MappedSize);
|
||||
return true;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
struct SerializedFilter {
|
||||
size_t CodeSize;
|
||||
uint32_t FilterInstructions;
|
||||
bool ShouldLog;
|
||||
char Code[];
|
||||
};
|
||||
|
||||
struct SerializationHeader {
|
||||
size_t NumberOfFilters;
|
||||
uint32_t SeccompMode;
|
||||
SerializedFilter Filters[];
|
||||
};
|
||||
|
||||
std::optional<int> SeccompEmulator::SerializeFilters(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
if (Thread->SeccompMode == SECCOMP_MODE_DISABLED) {
|
||||
// Didn't have seccomp enabled.
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
int FD = memfd_create("seccomp_filters", MFD_ALLOW_SEALING);
|
||||
if (FD == -1) {
|
||||
// Couldn't create memfd
|
||||
LogMan::Msg::EFmt("Couldn't create seccomp filter FD!");
|
||||
return -1;
|
||||
}
|
||||
|
||||
SerializationHeader Header {
|
||||
.NumberOfFilters = Thread->Filters.size(),
|
||||
.SeccompMode = Thread->SeccompMode,
|
||||
};
|
||||
|
||||
int Res = write(FD, &Header, sizeof(Header));
|
||||
if (Res == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't write header!");
|
||||
close(FD);
|
||||
return -1;
|
||||
}
|
||||
|
||||
for (auto& Filter : Thread->Filters) {
|
||||
SerializedFilter SFilter {
|
||||
.CodeSize = Filter->MappedSize,
|
||||
.FilterInstructions = Filter->FilterInstructions,
|
||||
.ShouldLog = Filter->ShouldLog,
|
||||
};
|
||||
|
||||
Res = write(FD, &SFilter, sizeof(SFilter));
|
||||
if (Res == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't write filter header!");
|
||||
close(FD);
|
||||
return -1;
|
||||
}
|
||||
|
||||
Res = write(FD, (const void*)Filter->Func, Filter->MappedSize);
|
||||
if (Res == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't write filter!");
|
||||
close(FD);
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
// Reset FD to start.
|
||||
lseek(FD, 0, SEEK_SET);
|
||||
|
||||
// Seal everything about this FD.
|
||||
fcntl(FD, F_ADD_SEALS, F_SEAL_SEAL | F_SEAL_SHRINK | F_SEAL_GROW | F_SEAL_WRITE | F_SEAL_FUTURE_WRITE);
|
||||
|
||||
return FD;
|
||||
}
|
||||
|
||||
void SeccompEmulator::DeserializeFilters(FEXCore::Core::CpuStateFrame* Frame, int FD) {
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
SerializationHeader Header;
|
||||
int Res = read(FD, &Header, sizeof(Header));
|
||||
if (Res == -1 || Res != sizeof(Header)) {
|
||||
LogMan::Msg::EFmt("Couldn't read Seccomp header!");
|
||||
close(FD);
|
||||
return;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < Header.NumberOfFilters; ++i) {
|
||||
SerializedFilter SFilter;
|
||||
|
||||
Res = read(FD, &SFilter, sizeof(SFilter));
|
||||
if (Res == -1 || Res != sizeof(SFilter)) {
|
||||
LogMan::Msg::EFmt("Couldn't read Seccomp Filter header!");
|
||||
close(FD);
|
||||
return;
|
||||
}
|
||||
auto Ptr = FEXCore::Allocator::mmap(nullptr, SFilter.CodeSize, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
if (Ptr == (void*)~0ULL) {
|
||||
LogMan::Msg::EFmt("Couldn't allocate ptr for filter!");
|
||||
close(FD);
|
||||
return;
|
||||
}
|
||||
|
||||
Res = read(FD, Ptr, SFilter.CodeSize);
|
||||
if (Res == -1 || Res != SFilter.CodeSize) {
|
||||
LogMan::Msg::EFmt("Couldn't read Seccomp Filter code!");
|
||||
close(FD);
|
||||
return;
|
||||
}
|
||||
|
||||
::mprotect(Ptr, SFilter.CodeSize, PROT_READ | PROT_EXEC);
|
||||
|
||||
auto& it = Filters.emplace_back(FilterInformation {(FilterFunc)Ptr, 1, SFilter.CodeSize, SFilter.FilterInstructions, SFilter.ShouldLog});
|
||||
TotalFilterInstructions += SFilter.FilterInstructions;
|
||||
|
||||
// Append the filter to the thread.
|
||||
Thread->Filters.emplace_back(&it);
|
||||
}
|
||||
|
||||
Thread->SeccompMode = Header.SeccompMode;
|
||||
close(FD);
|
||||
}
|
||||
|
||||
SeccompEmulator::ExecuteFilterResult
|
||||
SeccompEmulator::ExecuteFilter(FEXCore::Core::CpuStateFrame* Frame, uint64_t JITPC, FEXCore::HLE::SyscallArguments* Args) {
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
if (Thread->Filters.empty()) {
|
||||
// Seccomp not installed. Allow it.
|
||||
return {false, 0};
|
||||
}
|
||||
|
||||
// Reconstruct the RIP from the JITPC.
|
||||
const uint64_t RIP = Thread->Thread->CTX->RestoreRIPFromHostPC(Frame->Thread, JITPC);
|
||||
|
||||
const auto Arch = Is64BitMode() ? AUDIT_ARCH_X86_64 : AUDIT_ARCH_I386;
|
||||
bool ShouldLog {};
|
||||
uint32_t SeccompResult {};
|
||||
|
||||
{
|
||||
BPFEmitter::WorkingBuffer Data {
|
||||
.Data =
|
||||
{
|
||||
.nr = static_cast<int32_t>(Args->Argument[0]),
|
||||
.arch = Arch,
|
||||
.instruction_pointer = RIP,
|
||||
.args =
|
||||
{
|
||||
Args->Argument[1],
|
||||
Args->Argument[2],
|
||||
Args->Argument[3],
|
||||
Args->Argument[4],
|
||||
Args->Argument[5],
|
||||
Args->Argument[6],
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
bool HasResult {};
|
||||
// seccomp filters are executed from latest added to oldest.
|
||||
for (auto it = Thread->Filters.rbegin(); it != Thread->Filters.rend(); ++it) {
|
||||
// Explicitly zero scratch memory.
|
||||
memset(&Data.ScratchMemory, 0, sizeof(Data.ScratchMemory));
|
||||
|
||||
uint32_t CurrentResult = (*it)->Func(0, 0, 0, 0, &Data);
|
||||
|
||||
if (!HasResult) {
|
||||
SeccompResult = CurrentResult;
|
||||
ShouldLog = (*it)->ShouldLog;
|
||||
HasResult = true;
|
||||
continue;
|
||||
}
|
||||
|
||||
const int16_t CurrentAction = (CurrentResult & SECCOMP_RET_ACTION_FULL) >> 16;
|
||||
const int16_t Action = (SeccompResult & SECCOMP_RET_ACTION_FULL) >> 16;
|
||||
|
||||
// All actions are executed but the first highest precendent result is returned.
|
||||
// Precedent order from highest priority to lowest:
|
||||
// - SECCOMP_RET_KILL_PROCESS (0x8000, -32768)
|
||||
// - SECCOMP_RET_KILL_THREAD (0x0000, 0)
|
||||
// - SECCOMP_RET_TRAP (0x0003, 3)
|
||||
// - SECCOMP_RET_ERRNO (0x0005, 5)
|
||||
// - SECCOMP_RET_USER_NOTIF (0x7fc0, 32704)
|
||||
// - SECCOMP_RET_TRACE (0x7ff0, 32752)
|
||||
// - SECCOMP_RET_LOG (0x7ffc, 32764)
|
||||
// - SECCOMP_RET_ALLOW (0x7fff, 32767)
|
||||
if (CurrentAction < Action) {
|
||||
SeccompResult = CurrentResult;
|
||||
ShouldLog = (*it)->ShouldLog;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const auto ActionMasked = SeccompResult & SECCOMP_RET_ACTION_FULL;
|
||||
const auto DataMasked = SeccompResult & SECCOMP_RET_DATA;
|
||||
|
||||
// Logging rules
|
||||
// - Log if explicitly returning SECCOMP_RET_LOG
|
||||
// - Log if the filter enabled the logging flag and the action is something other than SECCOMP_RET_ALLOW.
|
||||
if ((ShouldLog && ActionMasked != SECCOMP_RET_ALLOW) || ActionMasked == SECCOMP_RET_LOG) {
|
||||
int Signal = 0;
|
||||
switch (ActionMasked) {
|
||||
case SECCOMP_RET_KILL_PROCESS:
|
||||
case SECCOMP_RET_KILL_THREAD: Signal = GetKillSignal(); break;
|
||||
case SECCOMP_RET_TRAP: Signal = SIGSYS; break;
|
||||
default: break;
|
||||
}
|
||||
|
||||
// With real secommp the logs go to dmesg. log through FEX since we can't use dmesg.
|
||||
// ex: `[13572.669277] audit: type=1326 audit(1715469332.533:62): auid=1000 uid=1000 gid=1000 ses=2 subj=unconfined pid=52546 comm="seccomp_bpf"
|
||||
// exe="/mnt/Work/Projects/work/linux/tools/testing/selftests/seccomp/seccomp_bpf" sig=0 arch=c000003e syscall=39 compat=0 ip=0x7d789352725d code=0x7ffc0000`
|
||||
timespec tp {};
|
||||
clock_gettime(CLOCK_MONOTONIC, &tp);
|
||||
LogMan::Msg::IFmt("audit: type={} audit({}.{:03}:{}): uid={} gid={} pid={} comm={} sig={} arch={:x} syscall={} ip=0x{:x} code=0x{:x}",
|
||||
AUDIT_SECCOMP, tp.tv_sec, tp.tv_nsec / 1'000'000, AuditSerialIncrement(), ::getuid(), ::getgid(), ::getpid(),
|
||||
Filename(), Signal, Arch, Args->Argument[0], RIP, SeccompResult);
|
||||
}
|
||||
|
||||
switch (ActionMasked) {
|
||||
// Unknown actions behave like RET_KILL_PROCESS.
|
||||
default:
|
||||
case SECCOMP_RET_KILL_PROCESS: {
|
||||
const int KillSignal = GetKillSignal();
|
||||
// Ignores signal handler and sigmask
|
||||
uint64_t Mask = 1 << (KillSignal - 1);
|
||||
SignalDelegation->GuestSigProcMask(Thread, SIG_UNBLOCK, &Mask, nullptr);
|
||||
SignalDelegation->UninstallHostHandler(KillSignal);
|
||||
kill(0, KillSignal);
|
||||
break;
|
||||
}
|
||||
case SECCOMP_RET_KILL_THREAD: {
|
||||
// Ignores signal handler and sigmask
|
||||
uint64_t Mask = 1 << (SIGSYS - 1);
|
||||
SignalDelegation->GuestSigProcMask(Thread, SIG_UNBLOCK, &Mask, nullptr);
|
||||
SignalDelegation->UninstallHostHandler(SIGSYS);
|
||||
tgkill(::getpid(), ::gettid(), SIGSYS);
|
||||
break;
|
||||
}
|
||||
case SECCOMP_RET_TRAP: {
|
||||
siginfo_t Info {
|
||||
.si_signo = SIGSYS,
|
||||
.si_errno = static_cast<int32_t>(DataMasked),
|
||||
.si_code = 1, // SYS_SECCOMP
|
||||
};
|
||||
|
||||
Info.si_call_addr = reinterpret_cast<void*>(RIP);
|
||||
Info.si_syscall = Args->Argument[0];
|
||||
Info.si_arch = Arch;
|
||||
|
||||
SignalDelegation->QueueSignal(::getpid(), ::gettid(), SIGSYS, &Info, true);
|
||||
break;
|
||||
}
|
||||
case SECCOMP_RET_ERRNO: {
|
||||
// errno return is clamped.
|
||||
return {true, -(std::min<uint64_t>(DataMasked, 4095))};
|
||||
}
|
||||
case SECCOMP_RET_TRACE: {
|
||||
// When no tracer attached, behave like RET_ERRNO returning ENOSYS.
|
||||
// TODO: Implement once FEX supports tracing.
|
||||
return {true, static_cast<uint64_t>(-ENOSYS)};
|
||||
}
|
||||
case SECCOMP_RET_USER_NOTIF:
|
||||
case SECCOMP_RET_LOG:
|
||||
case SECCOMP_RET_ALLOW: break;
|
||||
}
|
||||
|
||||
return {false, 0};
|
||||
}
|
||||
|
||||
// Equivalent to seccomp(SECCOMP_SET_MODE_STRICT, ...);
|
||||
uint64_t SeccompEmulator::SetModeStrict(FEXCore::Core::CpuStateFrame* Frame, uint32_t flags, const void* arg) {
|
||||
const auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
if (::prctl(PR_GET_NO_NEW_PRIVS, 0, 0, 0, 0) == 0) {
|
||||
// The caller did not have the CAP_SYS_ADMIN capability in its user namespace, or had not set no_new_privs before using SECCOMP_SET_MODE_FILTER.
|
||||
return -EACCES;
|
||||
}
|
||||
|
||||
if (flags != 0) {
|
||||
// The specified flags are invalid for the given operation.
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if (arg != nullptr) {
|
||||
// The specified arg are invalid for the given operation.
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if (Thread->SeccompMode == SECCOMP_MODE_FILTER) {
|
||||
// Filter mode cannot move to strict
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
#define syscall_nr (offsetof(struct seccomp_data, nr))
|
||||
#define ALLOW_SYSCALL(name) \
|
||||
BPF_JUMP(BPF_JMP + BPF_JEQ + BPF_K, FEX::HLE::x64::SYSCALL_x64_##name, 0, 1), BPF_STMT(BPF_RET + BPF_K, SECCOMP_RET_ALLOW)
|
||||
#define ALLOW_SYSCALL_x32(name) \
|
||||
BPF_JUMP(BPF_JMP + BPF_JEQ + BPF_K, FEX::HLE::x32::SYSCALL_x86_##name, 0, 1), BPF_STMT(BPF_RET + BPF_K, SECCOMP_RET_ALLOW)
|
||||
|
||||
constexpr static struct sock_filter strict_filter_x64[] = {
|
||||
// Load syscall number
|
||||
BPF_STMT(BPF_LD + BPF_W + BPF_ABS, syscall_nr),
|
||||
|
||||
// Allow read, write, exit, exit_group, and sigreturn
|
||||
ALLOW_SYSCALL(read),
|
||||
ALLOW_SYSCALL(write),
|
||||
ALLOW_SYSCALL(exit),
|
||||
ALLOW_SYSCALL(exit_group),
|
||||
ALLOW_SYSCALL(rt_sigreturn),
|
||||
BPF_STMT(BPF_RET + BPF_K, SECCOMP_RET_KILL_PROCESS),
|
||||
};
|
||||
|
||||
constexpr static struct sock_filter strict_filter_x32[] = {
|
||||
// Load syscall number
|
||||
BPF_STMT(BPF_LD + BPF_W + BPF_ABS, syscall_nr),
|
||||
|
||||
// Allow read, write, exit, exit_group, and sigreturn
|
||||
ALLOW_SYSCALL_x32(read),
|
||||
ALLOW_SYSCALL_x32(write),
|
||||
ALLOW_SYSCALL_x32(exit),
|
||||
ALLOW_SYSCALL_x32(exit_group),
|
||||
ALLOW_SYSCALL_x32(rt_sigreturn),
|
||||
ALLOW_SYSCALL_x32(sigreturn),
|
||||
BPF_STMT(BPF_RET + BPF_K, SECCOMP_RET_KILL_PROCESS),
|
||||
};
|
||||
|
||||
const sock_fprog prog_x64 {
|
||||
.len = (unsigned short)(sizeof(strict_filter_x64) / sizeof(strict_filter_x64[0])),
|
||||
.filter = const_cast<struct sock_filter*>(strict_filter_x64),
|
||||
};
|
||||
|
||||
const sock_fprog prog_x32 {
|
||||
.len = (unsigned short)(sizeof(strict_filter_x32) / sizeof(strict_filter_x32[0])),
|
||||
.filter = const_cast<struct sock_filter*>(strict_filter_x32),
|
||||
};
|
||||
CurrentKillSignal = SIGKILL;
|
||||
const sock_fprog* prog = Is64BitMode() ? &prog_x64 : &prog_x32;
|
||||
SetModeFilter(Frame, 0, prog);
|
||||
Thread->SeccompMode = SECCOMP_MODE_STRICT;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t SeccompEmulator::CanDoTSync(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
auto ParentThread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
auto Threads = SyscallHandler->TM.GetThreads();
|
||||
|
||||
for (auto& Thread : *Threads) {
|
||||
if (Thread == ParentThread) {
|
||||
// Skip same thread.
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Thread->SeccompMode == SECCOMP_MODE_DISABLED) {
|
||||
// Threads which have seccomp disabled are safe to TSync
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Thread->SeccompMode != ParentThread->SeccompMode) {
|
||||
/// If the seccomp mode differs between threads then it can't tsync.
|
||||
/// Strict versus filter mode aren't tsync compatible.
|
||||
return Thread->ThreadInfo.TID;
|
||||
}
|
||||
|
||||
if (Thread->Filters.size() != ParentThread->Filters.size()) {
|
||||
// If the filter count doesn't even match then it can't tsync.
|
||||
return Thread->ThreadInfo.TID;
|
||||
}
|
||||
|
||||
// Walk each filter and ensure the entry points are the same and in the same order.
|
||||
for (size_t i = 0; i < ParentThread->Filters.size(); ++i) {
|
||||
if (Thread->Filters[i]->Func != ParentThread->Filters[i]->Func) {
|
||||
/// Entry point mismatch, not the same filter.
|
||||
/// Not tsync compatible.
|
||||
return Thread->ThreadInfo.TID;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Everything matched. tsync compatible!
|
||||
return 0;
|
||||
}
|
||||
|
||||
void SeccompEmulator::TSyncFilters(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
auto ParentThread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
auto Threads = SyscallHandler->TM.GetThreads();
|
||||
|
||||
for (auto& Thread : *Threads) {
|
||||
if (Thread == ParentThread) {
|
||||
// Skip same thread.
|
||||
continue;
|
||||
}
|
||||
|
||||
Thread->Filters.clear();
|
||||
Thread->Filters = ParentThread->Filters;
|
||||
for (auto& Filter : ParentThread->Filters) {
|
||||
// Need to increment all the refcounters
|
||||
std::atomic_ref<uint64_t>(Filter->RefCount)++;
|
||||
}
|
||||
Thread->SeccompMode = ParentThread->SeccompMode;
|
||||
}
|
||||
}
|
||||
|
||||
// Equivalent to seccomp(SECCOMP_SET_MODE_FILTER, ...);
|
||||
uint64_t SeccompEmulator::SetModeFilter(FEXCore::Core::CpuStateFrame* Frame, uint32_t flags, const sock_fprog* prog) {
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
// Order of checks in this function matter
|
||||
// 1) Check flags
|
||||
// 2) Check if program is invalid
|
||||
uint32_t SUPPORTED_FLAGS = SECCOMP_FILTER_FLAG_TSYNC | // 1U << 0
|
||||
SECCOMP_FILTER_FLAG_LOG | // 1U << 1
|
||||
SECCOMP_FILTER_FLAG_SPEC_ALLOW | // 1U << 2
|
||||
// SECCOMP_FILTER_FLAG_NEW_LISTENER | // 1U << 3
|
||||
SECCOMP_FILTER_FLAG_TSYNC_ESRCH | // 1U << 4
|
||||
0;
|
||||
|
||||
const bool DoingTsync = flags & SECCOMP_FILTER_FLAG_TSYNC;
|
||||
|
||||
if (flags & ~SUPPORTED_FLAGS) {
|
||||
// Unknown flags passed in.
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if ((flags & SECCOMP_FILTER_FLAG_TSYNC) && (flags & SECCOMP_FILTER_FLAG_NEW_LISTENER) && !(flags & SECCOMP_FILTER_FLAG_TSYNC_ESRCH)) {
|
||||
/// If NEW_LISTENER and TSYNC are both used then TSYNC_ESRCH must also be set.
|
||||
/// Otherwise on error there would be no way to tell the difference between success and failure.
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if (!prog) {
|
||||
return -EFAULT;
|
||||
}
|
||||
|
||||
if (::prctl(PR_GET_NO_NEW_PRIVS, 0, 0, 0, 0) == 0) {
|
||||
// The caller did not have the CAP_SYS_ADMIN capability in its user namespace, or had not set no_new_privs before using SECCOMP_SET_MODE_FILTER.
|
||||
return -EACCES;
|
||||
}
|
||||
|
||||
if (prog->len > BPF_MAXINSNS || prog->len == 0) {
|
||||
// operation specified SECCOMP_SET_MODE_FILTER, but the filter program pointed to by args was not valid or the length of the filter
|
||||
// program was zero or exceeded BPF_MAXINSNS (4096) instructions.
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
// Don't interrupt me while I'm jitting.
|
||||
auto lk = FEXCore::MaskSignalsAndLockMutex(FilterMutex);
|
||||
|
||||
const size_t TotalFinalInstructions = TotalFilterInstructions + prog->len + Thread->Filters.size() * BPF_MULTIFILTERPENALTY;
|
||||
if (TotalFinalInstructions > BPF_MAX_INSNS_PER_PATH) {
|
||||
return -ENOMEM;
|
||||
}
|
||||
|
||||
if constexpr (false) {
|
||||
// Useful for debugging seccomp problems.
|
||||
DumpProgram(prog);
|
||||
}
|
||||
|
||||
if (DoingTsync) {
|
||||
auto TSyncThread = CanDoTSync(Frame);
|
||||
if (TSyncThread != 0) {
|
||||
if (flags & SECCOMP_FILTER_FLAG_TSYNC_ESRCH) {
|
||||
// This flag explicitly ensures that if TSYNC can't sync then it won't return a TID.
|
||||
return -ESRCH;
|
||||
} else {
|
||||
// Return the TID that caused a tsync problem.
|
||||
return TSyncThread;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
BPFEmitter emit {};
|
||||
const bool LoggingEnabled = flags & SECCOMP_FILTER_FLAG_LOG;
|
||||
auto Result = emit.JITFilter(flags, prog);
|
||||
if (Result == 0) {
|
||||
|
||||
auto& it = Filters.emplace_back(FilterInformation {(FilterFunc)emit.GetFunc(), 1, emit.AllocationSize(), prog->len, LoggingEnabled});
|
||||
TotalFilterInstructions += prog->len;
|
||||
|
||||
// Append the filter to the thread.
|
||||
Thread->Filters.emplace_back(&it);
|
||||
Thread->SeccompMode = SECCOMP_MODE_FILTER;
|
||||
if (flags & SECCOMP_FILTER_FLAG_TSYNC) {
|
||||
TSyncFilters(Frame);
|
||||
}
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
// Equivalent to seccomp(SECCOMP_GET_ACTION_AVAIL, ...);
|
||||
uint64_t SeccompEmulator::GetActionAvail(uint32_t flags, const uint32_t* action) {
|
||||
if (flags != 0) {
|
||||
// Unknown flags passed in
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
if (!action) {
|
||||
// Invalid action
|
||||
return -EFAULT;
|
||||
}
|
||||
switch (*action) {
|
||||
case SECCOMP_RET_KILL_PROCESS:
|
||||
case SECCOMP_RET_KILL_THREAD:
|
||||
case SECCOMP_RET_TRAP:
|
||||
case SECCOMP_RET_ERRNO:
|
||||
case SECCOMP_RET_LOG:
|
||||
case SECCOMP_RET_ALLOW: return 0;
|
||||
case SECCOMP_RET_USER_NOTIF:
|
||||
case SECCOMP_RET_TRACE:
|
||||
default: break;
|
||||
}
|
||||
|
||||
return -EOPNOTSUPP;
|
||||
}
|
||||
|
||||
// Equivalent to seccomp(SECCOMP_GET_NOTIF_SIZES, ...);
|
||||
uint64_t SeccompEmulator::GetNotifSizes(uint32_t flags, struct seccomp_notif_sizes* sizes) {
|
||||
if (flags != 0) {
|
||||
// Unknown flags passed in
|
||||
return -EINVAL;
|
||||
}
|
||||
sizes->seccomp_notif = sizeof(struct seccomp_notif);
|
||||
sizes->seccomp_notif_resp = sizeof(struct seccomp_notif_resp);
|
||||
sizes->seccomp_data = sizeof(struct seccomp_data);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
} // namespace FEX::HLE
|
||||
@@ -0,0 +1,112 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: LinuxSyscalls|syscalls-shared
|
||||
$end_info$
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
|
||||
#include <signal.h>
|
||||
|
||||
struct sock_fprog;
|
||||
struct seccomp_data;
|
||||
struct seccomp_notif_sizes;
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Core {
|
||||
struct CpuStateFrame;
|
||||
}
|
||||
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEX::HLE {
|
||||
|
||||
class SyscallHandler;
|
||||
class SignalDelegator;
|
||||
struct ThreadStateObject;
|
||||
|
||||
class SeccompEmulator final {
|
||||
public:
|
||||
SeccompEmulator(FEX::HLE::SyscallHandler* SyscallHandler, FEX::HLE::SignalDelegator* SignalDelegation)
|
||||
: SyscallHandler {SyscallHandler}
|
||||
, SignalDelegation {SignalDelegation} {}
|
||||
|
||||
uint64_t Handle(FEXCore::Core::CpuStateFrame* Frame, uint32_t Op, uint32_t flags, void* arg);
|
||||
|
||||
// Equivalent to prctl(PR_GET_SECCOMP)
|
||||
uint64_t GetSeccomp(FEXCore::Core::CpuStateFrame* Frame);
|
||||
|
||||
void InheritSeccompFilters(FEX::HLE::ThreadStateObject* Parent, FEX::HLE::ThreadStateObject* Child);
|
||||
void FreeSeccompFilters(FEX::HLE::ThreadStateObject* Thread);
|
||||
|
||||
struct ExecuteFilterResult {
|
||||
bool EarlyReturn {};
|
||||
uint64_t Result;
|
||||
};
|
||||
ExecuteFilterResult ExecuteFilter(FEXCore::Core::CpuStateFrame* Frame, uint64_t JITPC, FEXCore::HLE::SyscallArguments* Args);
|
||||
int GetKillSignal() const {
|
||||
return CurrentKillSignal;
|
||||
}
|
||||
|
||||
std::optional<int> SerializeFilters(FEXCore::Core::CpuStateFrame* Frame);
|
||||
void DeserializeFilters(FEXCore::Core::CpuStateFrame* Frame, int FD);
|
||||
|
||||
using FilterFunc = uint64_t (*)(uint32_t Acc, uint32_t Index, uint32_t Tmp, uint32_t Tmp2, void* Data);
|
||||
struct FilterInformation final {
|
||||
FilterFunc Func;
|
||||
uint64_t RefCount;
|
||||
size_t MappedSize;
|
||||
uint32_t FilterInstructions;
|
||||
bool ShouldLog;
|
||||
};
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(NeedsSeccomp, NEEDSSECCOMP);
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
FEX::HLE::SyscallHandler* SyscallHandler;
|
||||
FEX::HLE::SignalDelegator* SignalDelegation;
|
||||
|
||||
int CurrentKillSignal {SIGSYS};
|
||||
|
||||
// Equivalent to seccomp(SECCOMP_SET_MODE_STRICT, ...);
|
||||
uint64_t SetModeStrict(FEXCore::Core::CpuStateFrame* Frame, uint32_t flags, const void* arg);
|
||||
// Equivalent to seccomp(SECCOMP_SET_MODE_FILTER, ...);
|
||||
uint64_t SetModeFilter(FEXCore::Core::CpuStateFrame* Frame, uint32_t flags, const sock_fprog* prog);
|
||||
// Equivalent to seccomp(SECCOMP_GET_ACTION_AVAIL, ...);
|
||||
uint64_t GetActionAvail(uint32_t flags, const uint32_t* action);
|
||||
// Equivalent to seccomp(SECCOMP_GET_NOTIF_SIZES, ...);
|
||||
uint64_t GetNotifSizes(uint32_t flags, struct seccomp_notif_sizes* sizes);
|
||||
|
||||
// 0 on TSync possible
|
||||
/// TID for the first thread that breaks tsync.
|
||||
uint64_t CanDoTSync(FEXCore::Core::CpuStateFrame* Frame);
|
||||
void TSyncFilters(FEXCore::Core::CpuStateFrame* Frame);
|
||||
|
||||
static void DumpProgram(const sock_fprog* prog);
|
||||
|
||||
// Multiple filter instruction count penalty.
|
||||
// When multiple filters are installed there is a penalty per filter counted towards the maximum number of instructions.
|
||||
constexpr static size_t BPF_MULTIFILTERPENALTY = 4;
|
||||
// Maximum number of BPF instructions.
|
||||
constexpr static size_t BPF_MAX_INSNS_PER_PATH = 32768;
|
||||
uint64_t TotalFilterInstructions {};
|
||||
|
||||
FEXCore::ForkableUniqueMutex FilterMutex;
|
||||
fextl::list<FilterInformation> Filters {};
|
||||
|
||||
uint64_t AuditSerialIncrement() {
|
||||
return AuditSerial.fetch_add(1);
|
||||
}
|
||||
std::atomic<uint64_t> AuditSerial {};
|
||||
};
|
||||
} // namespace FEX::HLE
|
||||
File diff suppressed because it is too large.
Load diff
@@ -54,7 +54,7 @@ public:
|
||||
~SignalDelegator() override;
|
||||
|
||||
// Called from the signal trampoline function.
|
||||
void HandleSignal(int Signal, void* Info, void* UContext);
|
||||
void HandleSignal(FEX::HLE::ThreadStateObject* Thread, int Signal, void* Info, void* UContext);
|
||||
|
||||
void RegisterTLSState(FEX::HLE::ThreadStateObject* Thread);
|
||||
void UninstallTLSState(FEX::HLE::ThreadStateObject* Thread);
|
||||
@@ -82,11 +82,11 @@ public:
|
||||
*/
|
||||
uint64_t RegisterGuestSignalHandler(int Signal, const GuestSigAction* Action, struct GuestSigAction* OldAction);
|
||||
|
||||
uint64_t RegisterGuestSigAltStack(const stack_t* ss, stack_t* old_ss);
|
||||
uint64_t RegisterGuestSigAltStack(FEX::HLE::ThreadStateObject* Thread, const stack_t* ss, stack_t* old_ss);
|
||||
|
||||
uint64_t GuestSigProcMask(int how, const uint64_t* set, uint64_t* oldset);
|
||||
uint64_t GuestSigPending(uint64_t* set, size_t sigsetsize);
|
||||
uint64_t GuestSigSuspend(uint64_t* set, size_t sigsetsize);
|
||||
uint64_t GuestSigProcMask(FEX::HLE::ThreadStateObject* Thread, int how, const uint64_t* set, uint64_t* oldset);
|
||||
uint64_t GuestSigPending(FEX::HLE::ThreadStateObject* Thread, uint64_t* set, size_t sigsetsize);
|
||||
uint64_t GuestSigSuspend(FEX::HLE::ThreadStateObject* Thread, uint64_t* set, size_t sigsetsize);
|
||||
uint64_t GuestSigTimedWait(uint64_t* set, siginfo_t* info, const struct timespec* timeout, size_t sigsetsize);
|
||||
uint64_t GuestSignalFD(int fd, const uint64_t* set, size_t sigsetsize, int flags);
|
||||
/** @} */
|
||||
@@ -100,15 +100,13 @@ public:
|
||||
void CheckXIDHandler();
|
||||
|
||||
void UninstallHostHandler(int Signal);
|
||||
void QueueSignal(pid_t tgid, pid_t tid, int Signal, siginfo_t* info, bool IgnoreMask);
|
||||
|
||||
FEXCore::Context::Context* CTX;
|
||||
|
||||
void SetVDSOSigReturn() {
|
||||
// Get symbols from VDSO.
|
||||
VDSOPointers = FEX::VDSO::GetVDSOSymbols();
|
||||
|
||||
// Update VDSO to generated code.
|
||||
// TODO: Have the frontend generate the x86 sigreturn pointers.
|
||||
CTX->GetVDSOSigReturn(&VDSOPointers);
|
||||
}
|
||||
|
||||
[[noreturn]]
|
||||
@@ -134,10 +132,8 @@ public:
|
||||
|
||||
void SaveTelemetry();
|
||||
private:
|
||||
FEX::HLE::ThreadStateObject* GetTLSThread();
|
||||
|
||||
// Called from the thunk handler to handle the signal
|
||||
void HandleGuestSignal(FEXCore::Core::InternalThreadState* Thread, int Signal, void* Info, void* UContext);
|
||||
void HandleGuestSignal(FEX::HLE::ThreadStateObject* ThreadObject, int Signal, void* Info, void* UContext);
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
|
||||
@@ -0,0 +1,800 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: LinuxSyscalls|common
|
||||
desc: Handles host -> host and host -> guest signal routing, emulates procmask & co
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "LinuxSyscalls/SignalDelegator.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/ArchHelpers/Arm64.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/FPState.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
// Total number of layouts that siginfo supports.
|
||||
enum class SigInfoLayout {
|
||||
LAYOUT_KILL,
|
||||
LAYOUT_TIMER,
|
||||
LAYOUT_POLL,
|
||||
LAYOUT_FAULT,
|
||||
LAYOUT_FAULT_RIP,
|
||||
LAYOUT_CHLD,
|
||||
LAYOUT_RT,
|
||||
LAYOUT_SYS,
|
||||
};
|
||||
|
||||
// Calculate the siginfo layout based on Signal and si_code.
|
||||
static SigInfoLayout CalculateSigInfoLayout(int Signal, int si_code) {
|
||||
if (si_code > SI_USER && si_code < SI_KERNEL) {
|
||||
// For signals that are not considered RT.
|
||||
if (Signal == SIGSEGV || Signal == SIGBUS || Signal == SIGTRAP) {
|
||||
// Regular FAULT layout.
|
||||
return SigInfoLayout::LAYOUT_FAULT;
|
||||
} else if (Signal == SIGILL || Signal == SIGFPE) {
|
||||
// Fault layout but addr refers to RIP.
|
||||
return SigInfoLayout::LAYOUT_FAULT_RIP;
|
||||
} else if (Signal == SIGCHLD) {
|
||||
// Child layout
|
||||
return SigInfoLayout::LAYOUT_CHLD;
|
||||
} else if (Signal == SIGPOLL) {
|
||||
// Poll layout
|
||||
return SigInfoLayout::LAYOUT_POLL;
|
||||
} else if (Signal == SIGSYS) {
|
||||
// Sys layout
|
||||
return SigInfoLayout::LAYOUT_SYS;
|
||||
}
|
||||
} else {
|
||||
// Negative si_codes are kernel specific things.
|
||||
if (si_code == SI_TIMER) {
|
||||
return SigInfoLayout::LAYOUT_TIMER;
|
||||
} else if (si_code == SI_SIGIO) {
|
||||
return SigInfoLayout::LAYOUT_POLL;
|
||||
} else if (si_code < 0) {
|
||||
return SigInfoLayout::LAYOUT_RT;
|
||||
}
|
||||
}
|
||||
|
||||
return SigInfoLayout::LAYOUT_KILL;
|
||||
}
|
||||
|
||||
static uint32_t ConvertSignalToTrapNo(int Signal, siginfo_t* HostSigInfo) {
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
if (HostSigInfo->si_code == SEGV_MAPERR || HostSigInfo->si_code == SEGV_ACCERR) {
|
||||
// Protection fault
|
||||
return FEXCore::X86State::X86_TRAPNO_PF;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// Unknown mapping, fall back to old behaviour and just pass signal
|
||||
return Signal;
|
||||
}
|
||||
|
||||
static uint32_t ConvertSignalToError(void* ucontext, int Signal, siginfo_t* HostSigInfo) {
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
if (HostSigInfo->si_code == SEGV_MAPERR || HostSigInfo->si_code == SEGV_ACCERR) {
|
||||
// Protection fault
|
||||
// Always a user fault for us
|
||||
return ArchHelpers::Context::GetProtectFlags(ucontext);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// Not a page fault issue
|
||||
return 0;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static void SetXStateInfo(T* xstate, bool is_avx_enabled) {
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
fpstate->sw_reserved.magic1 = FEXCore::x86_64::fpx_sw_bytes::FP_XSTATE_MAGIC;
|
||||
fpstate->sw_reserved.extended_size = is_avx_enabled ? sizeof(T) : 0;
|
||||
|
||||
fpstate->sw_reserved.xfeatures |= FEXCore::x86_64::fpx_sw_bytes::FEATURE_FP | FEXCore::x86_64::fpx_sw_bytes::FEATURE_SSE;
|
||||
if (is_avx_enabled) {
|
||||
fpstate->sw_reserved.xfeatures |= FEXCore::x86_64::fpx_sw_bytes::FEATURE_YMM;
|
||||
}
|
||||
|
||||
fpstate->sw_reserved.xstate_size = fpstate->sw_reserved.extended_size;
|
||||
|
||||
if (is_avx_enabled) {
|
||||
xstate->xstate_hdr.xfeatures = 0;
|
||||
}
|
||||
}
|
||||
|
||||
void SignalDelegator::RestoreFrame_x64(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* Context,
|
||||
FEXCore::Core::CpuStateFrame* Frame, void* ucontext) {
|
||||
const bool IsAVXEnabled = Config.SupportsAVX;
|
||||
|
||||
auto* guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(Context->UContextLocation);
|
||||
[[maybe_unused]] auto* guest_siginfo = reinterpret_cast<siginfo_t*>(Context->SigInfoLocation);
|
||||
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] || Context->FaultToTopAndGeneratedException) {
|
||||
|
||||
// Restore previous `InSyscallInfo` structure.
|
||||
Frame->InSyscallInfo = Context->InSyscallInfo;
|
||||
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, Config.AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
|
||||
// XXX: Full context setting
|
||||
CTX->SetFlagsFromCompactedEFLAGS(Thread, guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL]);
|
||||
|
||||
#define COPY_REG(x) Frame->State.gregs[FEXCore::X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x];
|
||||
COPY_REG(R8);
|
||||
COPY_REG(R9);
|
||||
COPY_REG(R10);
|
||||
COPY_REG(R11);
|
||||
COPY_REG(R12);
|
||||
COPY_REG(R13);
|
||||
COPY_REG(R14);
|
||||
COPY_REG(R15);
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
auto* xstate = reinterpret_cast<FEXCore::x86_64::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.AbridgedFTW = fpstate->ftw;
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
|
||||
void SignalDelegator::RestoreFrame_ia32(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* Context,
|
||||
FEXCore::Core::CpuStateFrame* Frame, void* ucontext) {
|
||||
const bool IsAVXEnabled = Config.SupportsAVX;
|
||||
|
||||
SigFrame_i32* guest_uctx = reinterpret_cast<SigFrame_i32*>(Context->UContextLocation);
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->sc.ip || Context->FaultToTopAndGeneratedException) {
|
||||
// Restore previous `InSyscallInfo` structure.
|
||||
Frame->InSyscallInfo = Context->InSyscallInfo;
|
||||
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, Config.AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
// XXX: Full context setting
|
||||
CTX->SetFlagsFromCompactedEFLAGS(Thread, guest_uctx->sc.flags);
|
||||
|
||||
Frame->State.rip = guest_uctx->sc.ip;
|
||||
Frame->State.cs_idx = guest_uctx->sc.cs;
|
||||
Frame->State.ds_idx = guest_uctx->sc.ds;
|
||||
Frame->State.es_idx = guest_uctx->sc.es;
|
||||
Frame->State.fs_idx = guest_uctx->sc.fs;
|
||||
Frame->State.gs_idx = guest_uctx->sc.gs;
|
||||
Frame->State.ss_idx = guest_uctx->sc.ss;
|
||||
|
||||
Frame->State.cs_cached = Frame->State.gdt[Frame->State.cs_idx >> 3].base;
|
||||
Frame->State.ds_cached = Frame->State.gdt[Frame->State.ds_idx >> 3].base;
|
||||
Frame->State.es_cached = Frame->State.gdt[Frame->State.es_idx >> 3].base;
|
||||
Frame->State.fs_cached = Frame->State.gdt[Frame->State.fs_idx >> 3].base;
|
||||
Frame->State.gs_cached = Frame->State.gdt[Frame->State.gs_idx >> 3].base;
|
||||
Frame->State.ss_cached = Frame->State.gdt[Frame->State.ss_idx >> 3].base;
|
||||
|
||||
#define COPY_REG(x, y) Frame->State.gregs[FEXCore::X86State::REG_##x] = guest_uctx->sc.y;
|
||||
COPY_REG(RDI, di);
|
||||
COPY_REG(RSI, si);
|
||||
COPY_REG(RBP, bp);
|
||||
COPY_REG(RBX, bx);
|
||||
COPY_REG(RDX, dx);
|
||||
COPY_REG(RAX, ax);
|
||||
COPY_REG(RCX, cx);
|
||||
COPY_REG(RSP, sp);
|
||||
#undef COPY_REG
|
||||
auto* xstate = reinterpret_cast<FEXCore::x86::xstate*>(guest_uctx->sc.fpstate);
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
if (IsAVXEnabled) {
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.AbridgedFTW = FEXCore::FPState::ConvertToAbridgedFTW(fpstate->ftw);
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
|
||||
void SignalDelegator::RestoreRTFrame_ia32(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* Context,
|
||||
FEXCore::Core::CpuStateFrame* Frame, void* ucontext) {
|
||||
const bool IsAVXEnabled = Config.SupportsAVX;
|
||||
|
||||
RTSigFrame_i32* guest_uctx = reinterpret_cast<RTSigFrame_i32*>(Context->UContextLocation);
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] || Context->FaultToTopAndGeneratedException) {
|
||||
|
||||
// Restore previous `InSyscallInfo` structure.
|
||||
Frame->InSyscallInfo = Context->InSyscallInfo;
|
||||
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, Config.AbsoluteLoopTopAddressFillSRA);
|
||||
// Set our state register to point to our guest thread data
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
// XXX: Full context setting
|
||||
CTX->SetFlagsFromCompactedEFLAGS(Thread, guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL]);
|
||||
|
||||
Frame->State.rip = guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
|
||||
Frame->State.cs_idx = guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds_idx = guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es_idx = guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs_idx = guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs_idx = guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss_idx = guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
|
||||
Frame->State.cs_cached = Frame->State.gdt[Frame->State.cs_idx >> 3].base;
|
||||
Frame->State.ds_cached = Frame->State.gdt[Frame->State.ds_idx >> 3].base;
|
||||
Frame->State.es_cached = Frame->State.gdt[Frame->State.es_idx >> 3].base;
|
||||
Frame->State.fs_cached = Frame->State.gdt[Frame->State.fs_idx >> 3].base;
|
||||
Frame->State.gs_cached = Frame->State.gdt[Frame->State.gs_idx >> 3].base;
|
||||
Frame->State.ss_cached = Frame->State.gdt[Frame->State.ss_idx >> 3].base;
|
||||
|
||||
#define COPY_REG(x) Frame->State.gregs[FEXCore::X86State::REG_##x] = guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
auto* xstate = reinterpret_cast<FEXCore::x86::xstate*>(guest_uctx->uc.uc_mcontext.fpregs);
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
if (IsAVXEnabled) {
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.AbridgedFTW = FEXCore::FPState::ConvertToAbridgedFTW(fpstate->ftw);
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t SignalDelegator::SetupFrame_x64(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* ContextBackup,
|
||||
FEXCore::Core::CpuStateFrame* Frame, int Signal, siginfo_t* HostSigInfo, void* ucontext,
|
||||
GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags) {
|
||||
|
||||
// Back up past the redzone, which is 128bytes
|
||||
// 32-bit doesn't have a redzone
|
||||
NewGuestSP -= 128;
|
||||
|
||||
const bool IsAVXEnabled = Config.SupportsAVX;
|
||||
|
||||
// On 64-bit the kernel sets up the siginfo_t and ucontext_t regardless of SA_SIGINFO set.
|
||||
// This allows the application to /always/ get the siginfo and ucontext even if it didn't set this flag.
|
||||
//
|
||||
// Signal frame layout on stack needs to be as follows
|
||||
// void* ReturnPointer
|
||||
// ucontext_t
|
||||
// siginfo_t
|
||||
// FP state
|
||||
// Host stack location
|
||||
NewGuestSP -= sizeof(uint64_t);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(uint64_t));
|
||||
|
||||
uint64_t HostStackLocation = NewGuestSP;
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::xstate);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(FEXCore::x86_64::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::_libc_fpstate);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(FEXCore::x86_64::_libc_fpstate));
|
||||
}
|
||||
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(siginfo_t);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(siginfo_t));
|
||||
uint64_t SigInfoLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::ucontext_t);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(FEXCore::x86_64::ucontext_t));
|
||||
uint64_t UContextLocation = NewGuestSP;
|
||||
|
||||
ContextBackup->FPStateLocation = FPStateLocation;
|
||||
ContextBackup->UContextLocation = UContextLocation;
|
||||
ContextBackup->SigInfoLocation = SigInfoLocation;
|
||||
|
||||
FEXCore::x86_64::ucontext_t* guest_uctx = reinterpret_cast<FEXCore::x86_64::ucontext_t*>(UContextLocation);
|
||||
siginfo_t* guest_siginfo = reinterpret_cast<siginfo_t*>(SigInfoLocation);
|
||||
// Store where the host context lives in the guest stack.
|
||||
*(uint64_t*)HostStackLocation = (uint64_t)ContextBackup;
|
||||
|
||||
// We have extended float information
|
||||
guest_uctx->uc_flags = FEXCore::x86_64::UC_FP_XSTATE | FEXCore::x86_64::UC_SIGCONTEXT_SS | FEXCore::x86_64::UC_STRICT_RESTORE_SS;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
auto* xstate = reinterpret_cast<FEXCore::x86_64::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = ContextBackup->OriginalRIP;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = eflags;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CSGSFS] = 0;
|
||||
|
||||
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
|
||||
// SI_USER could also potentially have random data in it, needs to be bit perfect
|
||||
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
|
||||
// Overwrite si_code
|
||||
guest_siginfo->si_code = Thread->CurrentFrame->SynchronousFaultData.si_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
} else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(ucontext, Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_OLDMASK] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CR2] = 0;
|
||||
|
||||
#define COPY_REG(x) guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x] = Frame->State.gregs[FEXCore::X86State::REG_##x];
|
||||
COPY_REG(R8);
|
||||
COPY_REG(R9);
|
||||
COPY_REG(R10);
|
||||
COPY_REG(R11);
|
||||
COPY_REG(R12);
|
||||
COPY_REG(R13);
|
||||
COPY_REG(R14);
|
||||
COPY_REG(R15);
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
fpstate->ftw = Frame->State.AbridgedFTW;
|
||||
|
||||
// Reconstruct FSW
|
||||
fpstate->fsw = (Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) | (Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] << 10) | (Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] << 14);
|
||||
|
||||
// Copy over signal stack information
|
||||
guest_uctx->uc_stack.ss_flags = GuestStack->ss_flags;
|
||||
guest_uctx->uc_stack.ss_sp = GuestStack->ss_sp;
|
||||
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
|
||||
|
||||
// Apparently RAX is always set to zero in case of badly misbehaving C applications and variadics.
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RAX] = 0;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RDI] = Signal;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSI] = SigInfoLocation;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RDX] = UContextLocation;
|
||||
|
||||
// Set up the new SP for stack handling
|
||||
// The host is required to provide us a restorer.
|
||||
// If the guest didn't provide a restorer then the application should fail with a SIGSEGV.
|
||||
// TODO: Emulate SIGSEGV when the guest doesn't provide a restorer.
|
||||
NewGuestSP -= 8;
|
||||
if (GuestAction->restorer) {
|
||||
*(uint64_t*)NewGuestSP = (uint64_t)GuestAction->restorer;
|
||||
} else {
|
||||
// XXX: Emulate SIGSEGV here
|
||||
// *(uint64_t*)NewGuestSP = SignalReturn;
|
||||
}
|
||||
|
||||
return NewGuestSP;
|
||||
}
|
||||
|
||||
uint64_t SignalDelegator::SetupFrame_ia32(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* ContextBackup,
|
||||
FEXCore::Core::CpuStateFrame* Frame, int Signal, siginfo_t* HostSigInfo, void* ucontext,
|
||||
GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags) {
|
||||
|
||||
const bool IsAVXEnabled = Config.SupportsAVX;
|
||||
const uint64_t SignalReturn = reinterpret_cast<uint64_t>(VDSOPointers.VDSO_kernel_sigreturn);
|
||||
|
||||
NewGuestSP -= sizeof(uint64_t);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(uint64_t));
|
||||
|
||||
uint64_t HostStackLocation = NewGuestSP;
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(FEXCore::x86::xstate);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(FEXCore::x86::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(FEXCore::x86::_libc_fpstate);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(FEXCore::x86::_libc_fpstate));
|
||||
}
|
||||
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(SigFrame_i32);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(SigFrame_i32));
|
||||
uint64_t SigFrameLocation = NewGuestSP;
|
||||
|
||||
ContextBackup->FPStateLocation = FPStateLocation;
|
||||
ContextBackup->UContextLocation = SigFrameLocation;
|
||||
ContextBackup->SigInfoLocation = 0;
|
||||
|
||||
SigFrame_i32* guest_uctx = reinterpret_cast<SigFrame_i32*>(SigFrameLocation);
|
||||
// Store where the host context lives in the guest stack.
|
||||
*(uint64_t*)HostStackLocation = (uint64_t)ContextBackup;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->sc.fpstate = static_cast<uint32_t>(FPStateLocation);
|
||||
auto* xstate = reinterpret_cast<FEXCore::x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->sc.cs = Frame->State.cs_idx;
|
||||
guest_uctx->sc.ds = Frame->State.ds_idx;
|
||||
guest_uctx->sc.es = Frame->State.es_idx;
|
||||
guest_uctx->sc.fs = Frame->State.fs_idx;
|
||||
guest_uctx->sc.gs = Frame->State.gs_idx;
|
||||
guest_uctx->sc.ss = Frame->State.ss_idx;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->sc.trapno = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_uctx->sc.err = Frame->SynchronousFaultData.err_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
} else {
|
||||
guest_uctx->sc.trapno = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_uctx->sc.err = ConvertSignalToError(ucontext, Signal, HostSigInfo);
|
||||
}
|
||||
|
||||
guest_uctx->sc.ip = ContextBackup->OriginalRIP;
|
||||
guest_uctx->sc.flags = eflags;
|
||||
guest_uctx->sc.sp_at_signal = 0;
|
||||
|
||||
#define COPY_REG(x, y) guest_uctx->sc.x = Frame->State.gregs[FEXCore::X86State::REG_##y];
|
||||
COPY_REG(di, RDI);
|
||||
COPY_REG(si, RSI);
|
||||
COPY_REG(bp, RBP);
|
||||
COPY_REG(bx, RBX);
|
||||
COPY_REG(dx, RDX);
|
||||
COPY_REG(ax, RAX);
|
||||
COPY_REG(cx, RCX);
|
||||
COPY_REG(sp, RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
// Reconstruct FSW
|
||||
fpstate->fsw = (Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) | (Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] << 10) | (Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] << 14);
|
||||
fpstate->ftw = FEXCore::FPState::ConvertFromAbridgedFTW(fpstate->fsw, Frame->State.mm, Frame->State.AbridgedFTW);
|
||||
|
||||
// Curiously non-rt signals don't support altstack. So that state doesn't exist here.
|
||||
|
||||
// Copy over the signal information.
|
||||
guest_uctx->Signal = Signal;
|
||||
|
||||
// Retcode needs to be bit-exact for debuggers
|
||||
constexpr static uint8_t retcode[] = {
|
||||
0x58, // pop eax
|
||||
0xb8, // mov
|
||||
0x77, 0x00, 0x00, 0x00, // 32-bit sigreturn
|
||||
0xcd, 0x80, // int 0x80
|
||||
};
|
||||
|
||||
memcpy(guest_uctx->retcode, &retcode, sizeof(retcode));
|
||||
|
||||
// 32-bit Guest can provide its own restorer or we need to provide our own.
|
||||
// On a real host this restorer will live in VDSO.
|
||||
constexpr uint32_t SA_RESTORER = 0x04000000;
|
||||
const bool HasRestorer = (GuestAction->sa_flags & SA_RESTORER) == SA_RESTORER;
|
||||
if (HasRestorer) {
|
||||
guest_uctx->pretcode = (uint32_t)(uint64_t)GuestAction->restorer;
|
||||
} else {
|
||||
guest_uctx->pretcode = SignalReturn;
|
||||
LOGMAN_THROW_AA_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
}
|
||||
|
||||
// Support regparm=3
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RAX] = Signal;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RDX] = 0;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RCX] = 0;
|
||||
|
||||
return NewGuestSP;
|
||||
}
|
||||
|
||||
uint64_t SignalDelegator::SetupRTFrame_ia32(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* ContextBackup,
|
||||
FEXCore::Core::CpuStateFrame* Frame, int Signal, siginfo_t* HostSigInfo, void* ucontext,
|
||||
GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags) {
|
||||
|
||||
const bool IsAVXEnabled = Config.SupportsAVX;
|
||||
const uint64_t SignalReturn = reinterpret_cast<uint64_t>(VDSOPointers.VDSO_kernel_rt_sigreturn);
|
||||
|
||||
NewGuestSP -= sizeof(uint64_t);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(uint64_t));
|
||||
|
||||
uint64_t HostStackLocation = NewGuestSP;
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(FEXCore::x86::xstate);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(FEXCore::x86::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(FEXCore::x86::_libc_fpstate);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(FEXCore::x86::_libc_fpstate));
|
||||
}
|
||||
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(RTSigFrame_i32);
|
||||
NewGuestSP = FEXCore::AlignDown(NewGuestSP, alignof(RTSigFrame_i32));
|
||||
|
||||
uint64_t SigFrameLocation = NewGuestSP;
|
||||
RTSigFrame_i32* guest_uctx = reinterpret_cast<RTSigFrame_i32*>(SigFrameLocation);
|
||||
// Store where the host context lives in the guest stack.
|
||||
*(uint64_t*)HostStackLocation = (uint64_t)ContextBackup;
|
||||
|
||||
ContextBackup->FPStateLocation = FPStateLocation;
|
||||
ContextBackup->UContextLocation = SigFrameLocation;
|
||||
ContextBackup->SigInfoLocation = 0; // Part of frame.
|
||||
|
||||
// We have extended float information
|
||||
guest_uctx->uc.uc_flags = FEXCore::x86::UC_FP_XSTATE;
|
||||
guest_uctx->uc.uc_link = 0;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc.uc_mcontext.fpregs = static_cast<uint32_t>(FPStateLocation);
|
||||
auto* xstate = reinterpret_cast<FEXCore::x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs_idx;
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds_idx;
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es_idx;
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs_idx;
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs_idx;
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss_idx;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
} else {
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_uctx->info.si_code = HostSigInfo->si_code;
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(ucontext, Signal, HostSigInfo);
|
||||
}
|
||||
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = ContextBackup->OriginalRIP;
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = eflags;
|
||||
guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = Frame->State.gregs[FEXCore::X86State::REG_RSP];
|
||||
guest_uctx->uc.uc_mcontext.cr2 = 0;
|
||||
|
||||
#define COPY_REG(x) guest_uctx->uc.uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[FEXCore::X86State::REG_##x];
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
// Reconstruct FSW
|
||||
fpstate->fsw = (Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] << 11) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] << 8) | (Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] << 9) |
|
||||
(Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] << 10) | (Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] << 14);
|
||||
fpstate->ftw = FEXCore::FPState::ConvertFromAbridgedFTW(fpstate->fsw, Frame->State.mm, Frame->State.AbridgedFTW);
|
||||
|
||||
// Copy over signal stack information
|
||||
guest_uctx->uc.uc_stack.ss_flags = GuestStack->ss_flags;
|
||||
guest_uctx->uc.uc_stack.ss_sp = static_cast<uint32_t>(reinterpret_cast<uint64_t>(GuestStack->ss_sp));
|
||||
guest_uctx->uc.uc_stack.ss_size = GuestStack->ss_size;
|
||||
|
||||
// Setup siginfo
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->info.si_code = Frame->SynchronousFaultData.si_code;
|
||||
} else {
|
||||
guest_uctx->info.si_code = HostSigInfo->si_code;
|
||||
}
|
||||
|
||||
// These three elements are in every siginfo
|
||||
guest_uctx->info.si_signo = HostSigInfo->si_signo;
|
||||
guest_uctx->info.si_errno = HostSigInfo->si_errno;
|
||||
|
||||
const SigInfoLayout Layout = CalculateSigInfoLayout(Signal, guest_uctx->info.si_code);
|
||||
|
||||
switch (Layout) {
|
||||
case SigInfoLayout::LAYOUT_KILL:
|
||||
guest_uctx->info._sifields._kill.pid = HostSigInfo->si_pid;
|
||||
guest_uctx->info._sifields._kill.uid = HostSigInfo->si_uid;
|
||||
break;
|
||||
case SigInfoLayout::LAYOUT_TIMER:
|
||||
guest_uctx->info._sifields._timer.tid = HostSigInfo->si_timerid;
|
||||
guest_uctx->info._sifields._timer.overrun = HostSigInfo->si_overrun;
|
||||
guest_uctx->info._sifields._timer.sigval.sival_int = HostSigInfo->si_int;
|
||||
break;
|
||||
case SigInfoLayout::LAYOUT_POLL:
|
||||
guest_uctx->info._sifields._poll.band = HostSigInfo->si_band;
|
||||
guest_uctx->info._sifields._poll.fd = HostSigInfo->si_fd;
|
||||
break;
|
||||
case SigInfoLayout::LAYOUT_FAULT:
|
||||
// Macro expansion to get the si_addr
|
||||
// This is the address trying to be accessed, not the RIP
|
||||
guest_uctx->info._sifields._sigfault.addr = static_cast<uint32_t>(reinterpret_cast<uintptr_t>(HostSigInfo->si_addr));
|
||||
break;
|
||||
case SigInfoLayout::LAYOUT_FAULT_RIP:
|
||||
// Macro expansion to get the si_addr
|
||||
// Can't really give a real result here. Pull from the context for now
|
||||
guest_uctx->info._sifields._sigfault.addr = ContextBackup->OriginalRIP;
|
||||
break;
|
||||
case SigInfoLayout::LAYOUT_CHLD:
|
||||
guest_uctx->info._sifields._sigchld.pid = HostSigInfo->si_pid;
|
||||
guest_uctx->info._sifields._sigchld.uid = HostSigInfo->si_uid;
|
||||
guest_uctx->info._sifields._sigchld.status = HostSigInfo->si_status;
|
||||
guest_uctx->info._sifields._sigchld.utime = HostSigInfo->si_utime;
|
||||
guest_uctx->info._sifields._sigchld.stime = HostSigInfo->si_stime;
|
||||
break;
|
||||
case SigInfoLayout::LAYOUT_RT:
|
||||
guest_uctx->info._sifields._rt.pid = HostSigInfo->si_pid;
|
||||
guest_uctx->info._sifields._rt.uid = HostSigInfo->si_uid;
|
||||
guest_uctx->info._sifields._rt.sigval.sival_int = HostSigInfo->si_int;
|
||||
break;
|
||||
case SigInfoLayout::LAYOUT_SYS:
|
||||
guest_uctx->info._sifields._sigsys.call_addr = static_cast<uint32_t>(reinterpret_cast<uintptr_t>(HostSigInfo->si_call_addr));
|
||||
guest_uctx->info._sifields._sigsys.syscall = HostSigInfo->si_syscall;
|
||||
// We need to lie about the architecture here.
|
||||
// Otherwise we would expose incorrect information to the guest.
|
||||
constexpr uint32_t AUDIT_LE = 0x4000'0000U;
|
||||
constexpr uint32_t MACHINE_I386 = 3; // This matches the ELF definition.
|
||||
guest_uctx->info._sifields._sigsys.arch = AUDIT_LE | MACHINE_I386;
|
||||
break;
|
||||
}
|
||||
|
||||
// Setup the guest stack context.
|
||||
guest_uctx->Signal = Signal;
|
||||
guest_uctx->pinfo = (uint32_t)(uint64_t)&guest_uctx->info;
|
||||
guest_uctx->puc = (uint32_t)(uint64_t)&guest_uctx->uc;
|
||||
|
||||
// Retcode needs to be bit-exact for debuggers
|
||||
constexpr static uint8_t rt_retcode[] = {
|
||||
0xb8, // mov
|
||||
0xad, 0x00, 0x00, 0x00, // 32-bit rt_sigreturn
|
||||
0xcd, 0x80, // int 0x80
|
||||
0x0, // Pad
|
||||
};
|
||||
|
||||
memcpy(guest_uctx->retcode, &rt_retcode, sizeof(rt_retcode));
|
||||
|
||||
// 32-bit Guest can provide its own restorer or we need to provide our own.
|
||||
// On a real host this restorer will live in VDSO.
|
||||
constexpr uint32_t SA_RESTORER = 0x04000000;
|
||||
const bool HasRestorer = (GuestAction->sa_flags & SA_RESTORER) == SA_RESTORER;
|
||||
if (HasRestorer) {
|
||||
guest_uctx->pretcode = (uint32_t)(uint64_t)GuestAction->restorer;
|
||||
} else {
|
||||
guest_uctx->pretcode = SignalReturn;
|
||||
LOGMAN_THROW_AA_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
}
|
||||
|
||||
// Support regparm=3
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RAX] = Signal;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RDX] = guest_uctx->pinfo;
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RCX] = guest_uctx->puc;
|
||||
|
||||
return NewGuestSP;
|
||||
}
|
||||
|
||||
} // namespace FEX::HLE
|
||||
@@ -43,6 +43,8 @@ $end_info$
|
||||
#include <alloca.h>
|
||||
#include <charconv>
|
||||
#include <functional>
|
||||
#include <linux/audit.h>
|
||||
#include <linux/seccomp.h>
|
||||
#include <memory>
|
||||
#include <regex>
|
||||
#include <sched.h>
|
||||
@@ -51,6 +53,7 @@ $end_info$
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <string.h>
|
||||
#include <signal.h>
|
||||
#include <system_error>
|
||||
#include <syscall.h>
|
||||
#include <sys/mman.h>
|
||||
@@ -193,7 +196,7 @@ static bool IsShebangFilename(const fextl::string& Filename) {
|
||||
return IsShebang;
|
||||
}
|
||||
|
||||
uint64_t ExecveHandler(const char* pathname, char* const* argv, char* const* envp, ExecveAtArgs Args) {
|
||||
uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname, char* const* argv, char* const* envp, ExecveAtArgs Args) {
|
||||
auto SyscallHandler = FEX::HLE::_SyscallHandler;
|
||||
fextl::string Filename {};
|
||||
|
||||
@@ -204,6 +207,7 @@ uint64_t ExecveHandler(const char* pathname, char* const* argv, char* const* env
|
||||
const bool IsFDExec = (Args.flags & AT_EMPTY_PATH) && strlen(pathname) == 0;
|
||||
const bool SupportsProcFSInterpreter = SyscallHandler->FM.SupportsProcFSInterpreterPath();
|
||||
fextl::string FDExecEnv;
|
||||
fextl::string FDSeccompEnv;
|
||||
|
||||
bool IsShebang {};
|
||||
|
||||
@@ -260,6 +264,21 @@ uint64_t ExecveHandler(const char* pathname, char* const* argv, char* const* env
|
||||
char* const* EnvpPtr = envp;
|
||||
bool FDExecCopy {};
|
||||
|
||||
auto SeccompFD = SyscallHandler->SeccompEmulator.SerializeFilters(Frame);
|
||||
const auto HasSeccomp = SeccompFD.has_value() && *SeccompFD != -1;
|
||||
|
||||
auto CloseSeccompFD = [&HasSeccomp, &SeccompFD]() {
|
||||
if (HasSeccomp) {
|
||||
close(*SeccompFD);
|
||||
}
|
||||
};
|
||||
|
||||
auto CloseFDExecFD = [&FDExecCopy, &Args]() {
|
||||
if (FDExecCopy) {
|
||||
close(Args.dirfd);
|
||||
}
|
||||
};
|
||||
|
||||
// If we don't have the interpreter installed we need to be extra careful for ENOEXEC
|
||||
// Reasoning is that if we try executing a file from FEXLoader then this process loses the ENOEXEC flag
|
||||
// Kernel does its own checks for file format support for this
|
||||
@@ -282,7 +301,7 @@ uint64_t ExecveHandler(const char* pathname, char* const* argv, char* const* env
|
||||
// TODO: Additional future tasks that require envp copying in the future:
|
||||
// - seccomp inheritance
|
||||
// - FEXServer FD inheritance (unshare(CLONE_NEWNET))
|
||||
const bool NeedsEnvpCopy = IsFDExec && !(IsBinfmtCompatible || IsOtherELF);
|
||||
const bool NeedsEnvpCopy = (IsFDExec && !(IsBinfmtCompatible || IsOtherELF)) || HasSeccomp;
|
||||
|
||||
if (NeedsEnvpCopy) {
|
||||
if (envp) {
|
||||
@@ -294,7 +313,7 @@ uint64_t ExecveHandler(const char* pathname, char* const* argv, char* const* env
|
||||
}
|
||||
}
|
||||
|
||||
if (IsFDExec) {
|
||||
if (IsFDExec && !IsBinfmtCompatible) {
|
||||
int Flags = fcntl(Args.dirfd, F_GETFD);
|
||||
if (Flags & FD_CLOEXEC) {
|
||||
// FEX needs the FD to live past execve when binfmt_misc isn't used,
|
||||
@@ -316,6 +335,15 @@ uint64_t ExecveHandler(const char* pathname, char* const* argv, char* const* env
|
||||
EnvpArgs.emplace_back(FDExecEnv.data());
|
||||
}
|
||||
|
||||
if (HasSeccomp) {
|
||||
// Create the environment variable to pass the FD to our FEX.
|
||||
// Needs to stick around until execveat completes.
|
||||
FDSeccompEnv = fextl::fmt::format("FEX_SECCOMPFD={}", *SeccompFD);
|
||||
|
||||
// Insert the FD for FEX to track.
|
||||
EnvpArgs.emplace_back(FDSeccompEnv.data());
|
||||
}
|
||||
|
||||
// Emplace nullptr at the end to stop
|
||||
EnvpArgs.emplace_back(nullptr);
|
||||
|
||||
@@ -325,6 +353,8 @@ uint64_t ExecveHandler(const char* pathname, char* const* argv, char* const* env
|
||||
|
||||
if (IsBinfmtCompatible || IsOtherELF) {
|
||||
Result = ::syscall(SYS_execveat, Args.dirfd, Filename.c_str(), argv, EnvpPtr, Args.flags);
|
||||
CloseSeccompFD();
|
||||
CloseFDExecFD();
|
||||
SYSCALL_ERRNO();
|
||||
}
|
||||
|
||||
@@ -364,11 +394,8 @@ uint64_t ExecveHandler(const char* pathname, char* const* argv, char* const* env
|
||||
|
||||
const char* InterpreterPath = SupportsProcFSInterpreter ? "/proc/self/interpreter" : "/proc/self/exe";
|
||||
Result = ::syscall(SYS_execveat, Args.dirfd, InterpreterPath, const_cast<char* const*>(ExecveArgs.data()), EnvpPtr, Args.flags);
|
||||
|
||||
if (FDExecCopy) {
|
||||
///< Had to make a copy, close it now.
|
||||
close(Args.dirfd);
|
||||
}
|
||||
CloseSeccompFD();
|
||||
CloseFDExecFD();
|
||||
|
||||
SYSCALL_ERRNO();
|
||||
}
|
||||
@@ -719,6 +746,7 @@ void SyscallHandler::DefaultProgramBreak(uint64_t Base, uint64_t Size) {
|
||||
|
||||
SyscallHandler::SyscallHandler(FEXCore::Context::Context* _CTX, FEX::HLE::SignalDelegator* _SignalDelegation)
|
||||
: TM {_CTX, _SignalDelegation}
|
||||
, SeccompEmulator {this, _SignalDelegation}
|
||||
, FM {_CTX}
|
||||
, CTX {_CTX}
|
||||
, SignalDelegation {_SignalDelegation} {
|
||||
@@ -759,6 +787,15 @@ uint32_t SyscallHandler::CalculateGuestKernelVersion() {
|
||||
}
|
||||
|
||||
uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) {
|
||||
// Grab the return address which will be inside the JIT.
|
||||
const uint64_t JITPC = reinterpret_cast<uint64_t>(__builtin_extract_return_addr(__builtin_return_address(0)));
|
||||
|
||||
const auto SeccompResult = SeccompEmulator.ExecuteFilter(Frame, JITPC, Args);
|
||||
|
||||
if (SeccompResult.EarlyReturn) {
|
||||
return SeccompResult.Result;
|
||||
}
|
||||
|
||||
if (Args->Argument[0] >= Definitions.size()) {
|
||||
return -ENOSYS;
|
||||
}
|
||||
|
||||
Loaded 100 of 209 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user