mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 19:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8170307aef | ||
|
|
12cc979ec6 | ||
|
|
d6ecd6364c | ||
|
|
7d6dafe25d | ||
|
|
db6f4786d2 | ||
|
|
49fa69d4e2 | ||
|
|
448cbc8e3d | ||
|
|
81ce7b3f16 | ||
|
|
c9615032a9 | ||
|
|
52e2c4abc5 | ||
|
|
c4216755ab | ||
|
|
09071e4279 | ||
|
|
46a6e79233 | ||
|
|
09363f1b1b | ||
|
|
9d5f9d6783 | ||
|
|
7d0535e0d9 | ||
|
|
176b3292ed | ||
|
|
3d25f594a9 | ||
|
|
861d89e8a8 | ||
|
|
59c36ebc36 | ||
|
|
2d8304547e | ||
|
|
4880097f89 | ||
|
|
3a11b88d30 | ||
|
|
76a054e50a | ||
|
|
279d9219a6 | ||
|
|
ed4c5d9704 | ||
|
|
069e279a18 | ||
|
|
9ba1948375 | ||
|
|
835ca9cdf9 | ||
|
|
59a429ec19 | ||
|
|
36b5863c7c | ||
|
|
bdfbebe24d | ||
|
|
336ecb8cc7 | ||
|
|
ea9f08e481 | ||
|
|
da8c5dc460 | ||
|
|
fd8dcda67e | ||
|
|
b2fb48d711 | ||
|
|
8e6328e946 | ||
|
|
9262264525 | ||
|
|
4553564c9f | ||
|
|
2a86421172 | ||
|
|
bb85f90f78 | ||
|
|
4a1da85214 | ||
|
|
ff9268d6c7 | ||
|
|
c4f8e6c934 | ||
|
|
52c50735d5 | ||
|
|
aca8903b56 | ||
|
|
2e9b22e042 | ||
|
|
d4559ea0c6 | ||
|
|
d4dd4d0972 | ||
|
|
79db23b7d3 | ||
|
|
fbfc774446 | ||
|
|
35157c1251 | ||
|
|
17d836a178 | ||
|
|
28d754b466 | ||
|
|
01667bdfb2 | ||
|
|
2ddcaef525 | ||
|
|
c7cd2241c8 | ||
|
|
f67f3fe4de | ||
|
|
5d7822c987 | ||
|
|
738354a52d | ||
|
|
26531e9b96 | ||
|
|
282402a80c | ||
|
|
28254a3163 | ||
|
|
0dd568f7fb | ||
|
|
12adc5bc6e | ||
|
|
6ac942266f | ||
|
|
24a2a0c4eb | ||
|
|
910c624a8b | ||
|
|
767ea0fc9e | ||
|
|
087e5a9576 | ||
|
|
67e6ffbc93 | ||
|
|
6734d745c6 | ||
|
|
332123a38c | ||
|
|
24fdbe4e6d | ||
|
|
ea48f36511 | ||
|
|
d587485383 | ||
|
|
c9211ae77a | ||
|
|
7cbe9579ba | ||
|
|
ca482e3e96 | ||
|
|
eaa3df9cba | ||
|
|
c252dc99fc | ||
|
|
70dc5217ec | ||
|
|
1745bcceb6 | ||
|
|
cc8fa9d934 | ||
|
|
424b93b5f8 | ||
|
|
5fccf5f741 | ||
|
|
42f1dfeede | ||
|
|
21fa93f3b3 | ||
|
|
f088d97060 | ||
|
|
e369626929 | ||
|
|
cba4ca7d01 | ||
|
|
fb7964e6b1 | ||
|
|
ef7aff796f | ||
|
|
d2771669e6 | ||
|
|
e6e170805e | ||
|
|
1f89ca7218 | ||
|
|
adcee99625 | ||
|
|
a9623f0e2a | ||
|
|
c40ca14b5b | ||
|
|
100bc4c833 | ||
|
|
a8855a330a | ||
|
|
72a625da4a | ||
|
|
a67bdddbdc | ||
|
|
8274058895 | ||
|
|
52426ae9f3 | ||
|
|
4ff3705ff0 | ||
|
|
e5d3699b26 | ||
|
|
a5f07f1d01 | ||
|
|
d4789895a0 | ||
|
|
86cbf06778 | ||
|
|
d77a503510 | ||
|
|
9b4b136121 | ||
|
|
779aca7e95 | ||
|
|
3252f793df | ||
|
|
9635b34450 | ||
|
|
afd35be91f | ||
|
|
50b5ad5762 | ||
|
|
10ac1518e1 | ||
|
|
ff1c59b6af | ||
|
|
085fca01bc | ||
|
|
3d759a91ca | ||
|
|
b54385162e | ||
|
|
98714a4971 | ||
|
|
d08189c3ed | ||
|
|
2ad9ced80a | ||
|
|
12efcdc98d | ||
|
|
120ba3d171 | ||
|
|
6084bdf982 | ||
|
|
38d25d97f5 | ||
|
|
6fc117a67c | ||
|
|
1b72f73224 | ||
|
|
74f63c904e | ||
|
|
b6edc8dc53 | ||
|
|
af112f4c47 | ||
|
|
58c66b736d | ||
|
|
061f393186 | ||
|
|
f1fd197096 | ||
|
|
dc664b1e2e | ||
|
|
c971d3e56b | ||
|
|
0a03fbdfc6 | ||
|
|
10ec1d5d15 | ||
|
|
340919d0fb | ||
|
|
999443b898 | ||
|
|
adc5e4d6b9 | ||
|
|
903704ad48 | ||
|
|
1d2f6c3ba8 | ||
|
|
a07c57c52c | ||
|
|
9118285b60 | ||
|
|
64acc9ed3c | ||
|
|
d10fd4ddd4 | ||
|
|
a67b0c151e | ||
|
|
ba18b5dca9 | ||
|
|
0f1a41154d | ||
|
|
97dfe9b26e | ||
|
|
f94ce4c95b | ||
|
|
0e1892622c | ||
|
|
17977ab4ac | ||
|
|
78e207f837 | ||
|
|
e244142665 | ||
|
|
448c4ec797 | ||
|
|
08fad8fb0d | ||
|
|
29083a0b41 | ||
|
|
a1f4ca873c | ||
|
|
c3b0e4820b | ||
|
|
bb96955e05 | ||
|
|
16bb64aaab | ||
|
|
b9e53b6c46 | ||
|
|
616aa46c6b | ||
|
|
c98fb7ae45 | ||
|
|
6c5abc8819 | ||
|
|
f36726ecf6 | ||
|
|
6d20ae5dc6 | ||
|
|
e469031c40 | ||
|
|
b9a8c383fd | ||
|
|
57df4c6d61 | ||
|
|
c5cfb45f48 | ||
|
|
b6417153d7 | ||
|
|
599b579192 | ||
|
|
9ed0572fa4 | ||
|
|
00b0222129 | ||
|
|
32cc98856f | ||
|
|
dee7441795 | ||
|
|
f6a6f2eac8 | ||
|
|
c10e139964 | ||
|
|
379b200d05 | ||
|
|
b4cde28137 | ||
|
|
3f88150581 | ||
|
|
e95076f321 | ||
|
|
e2a16c53b6 | ||
|
|
ee89b7a794 | ||
|
|
5ac9b45528 | ||
|
|
6a94011b00 | ||
|
|
104fabc11f | ||
|
|
15b2eb48e5 | ||
|
|
62c446a5f9 | ||
|
|
d57a034220 | ||
|
|
18fe049976 | ||
|
|
c6e09ddec5 | ||
|
|
05ee4c363b | ||
|
|
6f2f4a9bc0 | ||
|
|
f699726409 | ||
|
|
a2b8fedba5 | ||
|
|
ed2336c4f2 | ||
|
|
748c3182a7 | ||
|
|
dfc454873c | ||
|
|
1b06a06a4e | ||
|
|
b2ad8d73ac | ||
|
|
fec078f924 | ||
|
|
38f1cded2c | ||
|
|
a313253510 | ||
|
|
7b2b2a3d07 | ||
|
|
910a25b0ed | ||
|
|
75bae81331 | ||
|
|
02abbd216a | ||
|
|
8b7a7e91c9 | ||
|
|
130f82a310 | ||
|
|
18e0f2636f | ||
|
|
2b9029623e | ||
|
|
6798afe91f | ||
|
|
4fa8522d9e | ||
|
|
58272ffc47 | ||
|
|
4c275ecc08 | ||
|
|
294ec80aaf | ||
|
|
e78d3d1a80 | ||
|
|
262389f4ea | ||
|
|
341b811642 | ||
|
|
32744d074e | ||
|
|
5f91bbe28d | ||
|
|
69035348f5 | ||
|
|
be46db10e4 | ||
|
|
aa2c18d8cc | ||
|
|
0a7f0a3441 | ||
|
|
ca2b04b309 | ||
|
|
376892793f | ||
|
|
61f73cf0bf | ||
|
|
9716a4f0cf | ||
|
|
35f3777797 | ||
|
|
d73b79470e | ||
|
|
676d9cb665 | ||
|
|
993b4513d9 | ||
|
|
2958744777 | ||
|
|
2e8aacffe6 | ||
|
|
2101914c9d | ||
|
|
97c4ba018b | ||
|
|
525d50f07e | ||
|
|
5cec8ddca0 | ||
|
|
39b2be15ab | ||
|
|
83bf79ab42 | ||
|
|
1f97a2baa9 | ||
|
|
0c2344166c | ||
|
|
0bf30e2024 | ||
|
|
933cdf76b4 | ||
|
|
36a77bb396 | ||
|
|
1936ebc59c | ||
|
|
9838309560 | ||
|
|
37e2210f64 | ||
|
|
71c32c0135 | ||
|
|
1f0bc49e54 | ||
|
|
718f7ef8e6 | ||
|
|
edd1dfdfe8 | ||
|
|
4243ed19a4 | ||
|
|
16467c0fb7 | ||
|
|
0b7587768d | ||
|
|
d115a57fe6 | ||
|
|
4c74478610 | ||
|
|
77a111cf19 | ||
|
|
7e551e5f88 | ||
|
|
9d44dfdca6 | ||
|
|
a7858f4d37 | ||
|
|
6b3b10f8ae | ||
|
|
9d5fa64f68 | ||
|
|
8b3bbd0ea1 | ||
|
|
cc477b6964 | ||
|
|
7a64bba8c4 | ||
|
|
2d342d4662 | ||
|
|
7b41808806 | ||
|
|
912d019ad8 | ||
|
|
1f9405b880 | ||
|
|
211e7bf0f0 | ||
|
|
822a08f271 | ||
|
|
6cba775d4e | ||
|
|
8b5873061a | ||
|
|
37a33bf127 | ||
|
|
d70c91ddc6 | ||
|
|
370f36c8f7 | ||
|
|
42bb27b1fb | ||
|
|
f522303837 | ||
|
|
3d09a55715 | ||
|
|
c103f54774 | ||
|
|
dc7437b8c6 | ||
|
|
b45503f93c | ||
|
|
8cf1b0263d | ||
|
|
7ecaf24e8b | ||
|
|
52f7ea5433 | ||
|
|
023a32ea26 | ||
|
|
36980131e0 | ||
|
|
b2eb13fa1c | ||
|
|
27b6497bee | ||
|
|
66ee89ec8a | ||
|
|
bee73199e1 | ||
|
|
0e6db843ef | ||
|
|
b45538f01e | ||
|
|
f8cf98d378 | ||
|
|
796c5ccbc9 | ||
|
|
b157a5a0fb | ||
|
|
783ceb2d56 | ||
|
|
d1fc65daaa | ||
|
|
94f464e1a4 | ||
|
|
dfb15aaeb9 | ||
|
|
565f0e6af3 | ||
|
|
b03554bb7d | ||
|
|
7b8cb5107a | ||
|
|
cbb20ef6c6 | ||
|
|
298481f9af | ||
|
|
0d108cd8fd | ||
|
|
490d7e57d7 | ||
|
|
fe77c3ef06 | ||
|
|
eb02afe952 | ||
|
|
b8dc63754b | ||
|
|
df8f1850ae | ||
|
|
e282fd2221 | ||
|
|
3147f0d84f | ||
|
|
4c02c9f037 | ||
|
|
83b4188569 | ||
|
|
82b04d9886 | ||
|
|
36fdd7f6e2 | ||
|
|
19e45544fb |
No files matched your search
@@ -33,3 +33,9 @@
|
||||
[submodule "External/jemalloc"]
|
||||
path = External/jemalloc
|
||||
url = https://github.com/FEX-Emu/jemalloc.git
|
||||
[submodule "External/fmt"]
|
||||
path = External/fmt
|
||||
url = https://github.com/fmtlib/fmt.git
|
||||
[submodule "External/drm-headers"]
|
||||
path = External/drm-headers
|
||||
url = https://github.com/FEX-Emu/drm-headers.git
|
||||
+66
-1
@@ -116,6 +116,8 @@ include_directories(External/jemalloc/pregen/include/)
|
||||
add_subdirectory(External/cpp-optparse/)
|
||||
include_directories(External/cpp-optparse/)
|
||||
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
include_directories(External/imgui/)
|
||||
|
||||
@@ -250,7 +252,7 @@ add_compile_options(-Wall)
|
||||
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/include/Config.h.in
|
||||
${CMAKE_BINARY_DIR}/generated/Config.h)
|
||||
${CMAKE_BINARY_DIR}/generated/ConfigDefines.h)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
include(CTest)
|
||||
@@ -259,6 +261,9 @@ if (BUILD_TESTS)
|
||||
endif()
|
||||
add_subdirectory(External/FEXCore)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
|
||||
add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
@@ -305,3 +310,63 @@ if (BUILD_THUNKS)
|
||||
DEPENDS guest-libs
|
||||
)
|
||||
endif()
|
||||
|
||||
set(FEX_VERSION_MAJOR "0")
|
||||
set(FEX_VERSION_MINOR "0")
|
||||
set(FEX_VERSION_PATCH "0")
|
||||
|
||||
find_package(Git)
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
RESULT_VARIABLE GIT_ERROR
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
|
||||
if (NOT ${GIT_ERROR} EQUAL 0)
|
||||
# Likely built in a way that doesn't have tags
|
||||
# Setup a version tag that is unknown
|
||||
set(GIT_DESCRIBE_STRING "FEX-0000")
|
||||
endif()
|
||||
|
||||
# Change something like `FEX-2106.1-76-<hash>` in to a list
|
||||
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
|
||||
|
||||
# Extract the `2106.1` element
|
||||
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
|
||||
|
||||
# Change `2106.1` in to a list
|
||||
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
|
||||
|
||||
# Calculate list size
|
||||
list(LENGTH DESCRIBE_LIST LIST_SIZE)
|
||||
|
||||
# Pull out the major version
|
||||
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
|
||||
|
||||
# Minor version only exists if there is a .1 at the end
|
||||
# eg: 2106 versus 2106.1
|
||||
if (LIST_SIZE GREATER 1)
|
||||
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Package creation
|
||||
set (CPACK_GENERATOR "DEB")
|
||||
set (CPACK_PACKAGE_CONTACT "team@fex-emu.org")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
|
||||
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
|
||||
|
||||
# Debian defines
|
||||
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libstdc++6")
|
||||
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA "${CMAKE_CURRENT_SOURCE_DIR}/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/CPack/prerm")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
# binfmt_misc conflicts with qemu-user-static
|
||||
# We also only install binfmt_misc on aarch64 hosts
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "qemu-user-static")
|
||||
endif()
|
||||
include (CPack)
|
||||
Executable
+18
@@ -0,0 +1,18 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Setup binfmt_misc
|
||||
update-binfmts --import FEX-x86
|
||||
update-binfmts --import FEX-x86_64
|
||||
}
|
||||
|
||||
# Install FEXInterpreter hardlink
|
||||
# Needs to be done before setting up binfmt_misc
|
||||
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
Executable
+17
@@ -0,0 +1,17 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Uninstall
|
||||
update-binfmts --unimport FEX-x86
|
||||
update-binfmts --unimport FEX-x86_64
|
||||
}
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
|
||||
# Remove FEXInterpreter hardlink
|
||||
unlink /usr/bin/FEXInterpreter
|
||||
@@ -0,0 +1,4 @@
|
||||
install(FILES FEX-x86
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
|
||||
install(FILES FEX-x86_64
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
|
||||
@@ -0,0 +1,9 @@
|
||||
package fex
|
||||
interpreter /usr/bin/FEXInterpreter
|
||||
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve no
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
package fex
|
||||
interpreter /usr/bin/FEXInterpreter
|
||||
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
credentials yes
|
||||
fix_binary yes
|
||||
preserve no
|
||||
+2
-2
@@ -3,7 +3,7 @@ FROM ubuntu:20.04 as builder
|
||||
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt install -y cmake \
|
||||
clang-10 llvm-10 nasm ninja-build libnuma-dev \
|
||||
clang-10 llvm-10 nasm ninja-build \
|
||||
libcap-dev libglfw3-dev libepoxy-dev python3-dev \
|
||||
python3 linux-headers-generic
|
||||
|
||||
@@ -23,7 +23,7 @@ FROM ubuntu:20.04
|
||||
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt-get update
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt install -y \
|
||||
libnuma-dev libcap-dev libglfw3-dev libepoxy-dev
|
||||
libcap-dev libglfw3-dev libepoxy-dev
|
||||
|
||||
COPY --from=builder /opt/FEX/build/Bin/* /usr/bin/
|
||||
|
||||
|
||||
+4
-4
@@ -124,8 +124,6 @@ set (SRCS
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/Allocator/64BitAllocator.cpp
|
||||
Utils/ELFContainer.cpp
|
||||
Utils/ELFSymbolDatabase.cpp
|
||||
Utils/LogManager.cpp
|
||||
Utils/Threads.cpp
|
||||
)
|
||||
@@ -267,7 +265,7 @@ function(AddObject Name Type)
|
||||
add_dependencies(${Name} IR_INC)
|
||||
add_dependencies(${Name} CONFIG_INC)
|
||||
|
||||
target_link_libraries(${Name} pthread vixl dl xxhash FEX_jemalloc)
|
||||
target_link_libraries(${Name} pthread vixl dl fmt::fmt xxhash FEX_jemalloc)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
|
||||
@@ -286,6 +284,8 @@ function(AddObject Name Type)
|
||||
target_compile_options(${Name}
|
||||
PRIVATE
|
||||
-Wall
|
||||
-Werror=cast-qual
|
||||
-Werror=ignored-qualifiers
|
||||
-Werror=implicit-fallthrough
|
||||
|
||||
-Wno-trigraphs
|
||||
@@ -306,7 +306,7 @@ endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} pthread vixl dl xxhash FEX_jemalloc)
|
||||
target_link_libraries(${Name} pthread vixl dl fmt::fmt xxhash FEX_jemalloc)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
|
||||
|
||||
+19
-11
@@ -6,10 +6,13 @@
|
||||
#include <sys/stat.h>
|
||||
|
||||
namespace FEXCore::Paths {
|
||||
std::string CachePath;
|
||||
std::string EntryCache;
|
||||
std::unique_ptr<std::string> CachePath;
|
||||
std::unique_ptr<std::string> EntryCache;
|
||||
|
||||
void InitializePaths() {
|
||||
CachePath = std::make_unique<std::string>();
|
||||
EntryCache = std::make_unique<std::string>();
|
||||
|
||||
char const *HomeDir = getenv("HOME");
|
||||
|
||||
if (!HomeDir) {
|
||||
@@ -22,29 +25,34 @@ namespace FEXCore::Paths {
|
||||
|
||||
char *XDGDataDir = getenv("XDG_DATA_DIR");
|
||||
if (XDGDataDir) {
|
||||
CachePath = XDGDataDir;
|
||||
*CachePath = XDGDataDir;
|
||||
}
|
||||
else {
|
||||
if (HomeDir) {
|
||||
CachePath = HomeDir;
|
||||
*CachePath = HomeDir;
|
||||
}
|
||||
}
|
||||
|
||||
CachePath += "/.fex-emu/";
|
||||
EntryCache = CachePath + "/EntryCache/";
|
||||
*CachePath += "/.fex-emu/";
|
||||
*EntryCache = *CachePath + "/EntryCache/";
|
||||
|
||||
// Ensure the folder structure is created for our Data
|
||||
if (!std::filesystem::exists(EntryCache) &&
|
||||
!std::filesystem::create_directories(EntryCache)) {
|
||||
LogMan::Msg::D("Couldn't create EntryCache directory: '%s'", EntryCache.c_str());
|
||||
if (!std::filesystem::exists(*EntryCache) &&
|
||||
!std::filesystem::create_directories(*EntryCache)) {
|
||||
LogMan::Msg::D("Couldn't create EntryCache directory: '%s'", EntryCache->c_str());
|
||||
}
|
||||
}
|
||||
|
||||
void ShutdownPaths() {
|
||||
CachePath.reset();
|
||||
EntryCache.reset();
|
||||
}
|
||||
|
||||
std::string GetCachePath() {
|
||||
return CachePath;
|
||||
return *CachePath;
|
||||
}
|
||||
|
||||
std::string GetEntryCachePath() {
|
||||
return EntryCache;
|
||||
return *EntryCache;
|
||||
}
|
||||
}
|
||||
+1
@@ -3,6 +3,7 @@
|
||||
|
||||
namespace FEXCore::Paths {
|
||||
void InitializePaths();
|
||||
void ShutdownPaths();
|
||||
std::string GetCachePath();
|
||||
std::string GetEntryCachePath();
|
||||
}
|
||||
+13
-11
@@ -1,4 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cmath>
|
||||
@@ -158,18 +160,18 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
operator float() const {
|
||||
float32_t Result = extF80_to_f32(*this);
|
||||
return *(float*)&Result;
|
||||
const float32_t Result = extF80_to_f32(*this);
|
||||
return FEXCore::BitCast<float>(Result);
|
||||
}
|
||||
|
||||
operator double() const {
|
||||
float64_t Result = extF80_to_f64(*this);
|
||||
return *(double*)&Result;
|
||||
const float64_t Result = extF80_to_f64(*this);
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
}
|
||||
|
||||
operator BIGFLOAT() const {
|
||||
float128_t Result = extF80_to_f128(*this);
|
||||
return *(BIGFLOAT*)&Result;
|
||||
const float128_t Result = extF80_to_f128(*this);
|
||||
return FEXCore::BitCast<BIGFLOAT>(Result);
|
||||
}
|
||||
|
||||
operator int16_t() const {
|
||||
@@ -196,11 +198,11 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
void operator=(const float rhs) {
|
||||
*this = f32_to_extF80(*(float32_t*)&rhs);
|
||||
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
|
||||
}
|
||||
|
||||
void operator=(const double rhs) {
|
||||
*this = f64_to_extF80(*(float64_t*)&rhs);
|
||||
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
void operator=(const int16_t rhs) {
|
||||
@@ -226,15 +228,15 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
X80SoftFloat(const float rhs) {
|
||||
*this = f32_to_extF80(*(float32_t*)&rhs);
|
||||
*this = f32_to_extF80(FEXCore::BitCast<float32_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(const double rhs) {
|
||||
*this = f64_to_extF80(*(float64_t*)&rhs);
|
||||
*this = f64_to_extF80(FEXCore::BitCast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(BIGFLOAT rhs) {
|
||||
*this = f128_to_extF80(*(float128_t*)&rhs);
|
||||
*this = f128_to_extF80(FEXCore::BitCast<float128_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(const int16_t rhs) {
|
||||
|
||||
+14
-3
@@ -83,17 +83,28 @@ namespace FEXCore::Config {
|
||||
return ConfigFile;
|
||||
}
|
||||
|
||||
std::string GetApplicationConfig(std::string &Filename, bool Global) {
|
||||
std::string GetApplicationConfig(const std::string &Filename, bool Global) {
|
||||
std::string ConfigFile = GetConfigDirectory(Global);
|
||||
if (!Global &&
|
||||
!std::filesystem::exists(ConfigFile) &&
|
||||
!std::filesystem::create_directories(ConfigFile)) {
|
||||
LogMan::Msg::D("Couldn't create config directory: '%s'", ConfigFile.c_str());
|
||||
// Let's go local in this case
|
||||
return "./";
|
||||
return "./" + Filename + ".json";
|
||||
}
|
||||
|
||||
ConfigFile += "AppConfig/" + Filename + ".json";
|
||||
ConfigFile += "AppConfig/";
|
||||
|
||||
// Attempt to create the local folder if it doesn't exist
|
||||
if (!Global &&
|
||||
!std::filesystem::exists(ConfigFile) &&
|
||||
!std::filesystem::create_directories(ConfigFile)) {
|
||||
LogMan::Msg::D("Couldn't create AppConfig directory: '%s'", ConfigFile.c_str());
|
||||
// Let's go local in this case
|
||||
return "./" + Filename + ".json";
|
||||
}
|
||||
|
||||
ConfigFile += Filename + ".json";
|
||||
return ConfigFile;
|
||||
}
|
||||
|
||||
|
||||
+19
-12
@@ -14,6 +14,10 @@ namespace FEXCore::Context {
|
||||
IR::InstallOpcodeHandlers(Mode);
|
||||
}
|
||||
|
||||
void ShutdownStaticTables() {
|
||||
FEXCore::Paths::ShutdownPaths();
|
||||
}
|
||||
|
||||
FEXCore::Context::Context *CreateNewContext() {
|
||||
return new FEXCore::Context::Context{};
|
||||
}
|
||||
@@ -33,12 +37,11 @@ namespace FEXCore::Context {
|
||||
return CTX->InitCore(Loader);
|
||||
}
|
||||
|
||||
void SetExitHandler(FEXCore::Context::Context *CTX,
|
||||
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> handler) {
|
||||
CTX->CustomExitHandler = handler;
|
||||
void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler) {
|
||||
CTX->CustomExitHandler = std::move(handler);
|
||||
}
|
||||
|
||||
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> GetExitHandler(FEXCore::Context::Context *CTX) {
|
||||
ExitHandler GetExitHandler(FEXCore::Context::Context *CTX) {
|
||||
return CTX->CustomExitHandler;
|
||||
}
|
||||
|
||||
@@ -50,8 +53,8 @@ namespace FEXCore::Context {
|
||||
CTX->Step();
|
||||
}
|
||||
|
||||
void CompileRIP(FEXCore::Context::Context *CTX, uint64_t GuestRIP) {
|
||||
CTX->CompileBlock(CTX->ParentThread->CurrentFrame, GuestRIP);
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
Thread->CTX->CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason RunUntilExit(FEXCore::Context::Context *CTX) {
|
||||
@@ -101,12 +104,12 @@ namespace FEXCore::Context {
|
||||
CTX->HandleCallback(RIP);
|
||||
}
|
||||
|
||||
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
|
||||
CTX->RegisterHostSignalHandler(Signal, Func);
|
||||
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func) {
|
||||
CTX->RegisterFrontendHostSignalHandler(Signal, Func);
|
||||
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterFrontendHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
@@ -149,8 +152,12 @@ namespace FEXCore::Context {
|
||||
CTX->AOTIRLoader = CacheReader;
|
||||
}
|
||||
|
||||
bool WriteAOTIR(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
|
||||
return CTX->WriteAOTIRCache(CacheWriter);
|
||||
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
|
||||
CTX->AOTIRWriter = CacheWriter;
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache(FEXCore::Context::Context *CTX) {
|
||||
CTX->FinalizeAOTIRCache();
|
||||
}
|
||||
|
||||
void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
|
||||
+54
-22
@@ -1,4 +1,5 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
@@ -9,6 +10,7 @@
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <stdint.h>
|
||||
|
||||
@@ -20,7 +22,9 @@
|
||||
#include <optional>
|
||||
#include <ostream>
|
||||
#include <set>
|
||||
#include <shared_mutex>
|
||||
#include <unordered_map>
|
||||
#include <queue>
|
||||
|
||||
namespace FEXCore {
|
||||
class ThunkHandler;
|
||||
@@ -52,14 +56,6 @@ namespace FEXCore::Context {
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
|
||||
struct AOTIRCaptureCacheEntry {
|
||||
uint64_t start;
|
||||
uint64_t len;
|
||||
uint64_t crc;
|
||||
IR::IRListView *IR;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
};
|
||||
|
||||
struct AOTIRInlineEntry {
|
||||
uint64_t GuestHash;
|
||||
uint64_t GuestLength;
|
||||
@@ -67,8 +63,8 @@ namespace FEXCore::Context {
|
||||
/* RAData followed by IRData */
|
||||
uint8_t InlineData[0];
|
||||
|
||||
IR::RegisterAllocationData *GetRAData();
|
||||
IR::IRListView *GetIRData();
|
||||
IR::RegisterAllocationData *GetRAData();
|
||||
IR::IRListView *GetIRData();
|
||||
};
|
||||
|
||||
struct AOTIRInlineIndexEntry {
|
||||
@@ -85,6 +81,13 @@ namespace FEXCore::Context {
|
||||
AOTIRInlineEntry *GetInlineEntry(uint64_t DataOffset);
|
||||
};
|
||||
|
||||
struct AOTIRCaptureCacheEntry {
|
||||
std::unique_ptr<std::ostream> Stream;
|
||||
std::map<uint64_t, uint64_t> Index;
|
||||
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData);
|
||||
};
|
||||
|
||||
struct Context {
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
@@ -121,7 +124,7 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(DumpIR, DUMPIR);
|
||||
} Config;
|
||||
|
||||
using IntCallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
using IntCallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
IntCallbackReturn InterpreterCallbackReturn;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
@@ -144,7 +147,7 @@ namespace FEXCore::Context {
|
||||
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> CustomExitHandler;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
AOTIRInlineIndex *Array;
|
||||
@@ -154,7 +157,8 @@ namespace FEXCore::Context {
|
||||
|
||||
std::unordered_map<std::string, AOTIRCacheEntry> AOTIRCache;
|
||||
std::function<int(const std::string&)> AOTIRLoader;
|
||||
std::unordered_map<std::string, std::map<uint64_t, AOTIRCaptureCacheEntry>> AOTIRCaptureCache;
|
||||
std::function<std::unique_ptr<std::ostream>(const std::string&)> AOTIRWriter;
|
||||
std::unordered_map<std::string, AOTIRCaptureCacheEntry> AOTIRCaptureCache;
|
||||
|
||||
struct AddrToFileEntry {
|
||||
uint64_t Start;
|
||||
@@ -181,7 +185,7 @@ namespace FEXCore::Context {
|
||||
|
||||
bool InitCore(FEXCore::CodeLoader *Loader);
|
||||
FEXCore::Context::ExitReason RunUntilExit();
|
||||
int GetProgramStatus();
|
||||
int GetProgramStatus() const;
|
||||
bool IsPaused() const { return !Running; }
|
||||
void Pause();
|
||||
void Run();
|
||||
@@ -192,12 +196,12 @@ namespace FEXCore::Context {
|
||||
void StopThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
|
||||
|
||||
bool GetGdbServerStatus() { return (bool)DebugServer; }
|
||||
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
|
||||
void StartGdbServer();
|
||||
void StopGdbServer();
|
||||
void HandleCallback(uint64_t RIP);
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func);
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
static void RemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
@@ -213,16 +217,35 @@ namespace FEXCore::Context {
|
||||
bool GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data);
|
||||
bool FindHostCodeForRIP(uint64_t RIP, uint8_t **Code);
|
||||
|
||||
std::tuple<FEXCore::IR::IRListView *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t, uint64_t, uint64_t> GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
// User's responsibility to deallocate this.
|
||||
FEXCore::IR::RegisterAllocationData* RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
std::tuple<void *, FEXCore::IR::IRListView *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool, uint64_t, uint64_t> CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
FEXCore::IR::IRListView* IRData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
// User's responsibility to deallocate this.
|
||||
FEXCore::IR::RegisterAllocationData* RAData;
|
||||
bool GeneratedIR;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
bool LoadAOTIRCache(int streamfd);
|
||||
bool WriteAOTIRCache(std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
|
||||
void FinalizeAOTIRCache();
|
||||
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
@@ -236,7 +259,9 @@ namespace FEXCore::Context {
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
void CleanupAfterFork(FEXCore::Core::InternalThreadState *ExceptForThread);
|
||||
|
||||
std::vector<FEXCore::Core::InternalThreadState*> *const GetThreads() { return &Threads; }
|
||||
std::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
|
||||
@@ -257,7 +282,6 @@ namespace FEXCore::Context {
|
||||
void NotifyPause();
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length);
|
||||
|
||||
FEXCore::CodeLoader *LocalLoader{};
|
||||
|
||||
// Entry Cache
|
||||
@@ -265,6 +289,14 @@ namespace FEXCore::Context {
|
||||
std::mutex ExitMutex;
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
std::shared_mutex AOTIRCacheLock;
|
||||
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
|
||||
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
|
||||
|
||||
std::queue<std::function<void()>> AOTIRCaptureCacheWriteoutQueue;
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn);
|
||||
|
||||
bool StartPaused = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
};
|
||||
|
||||
+238
-180
@@ -157,6 +157,13 @@ bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -200,20 +207,18 @@ bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
mcontext->regs[ExpectedReg1] = FailedResult & ~0U;
|
||||
mcontext->regs[ExpectedReg2] = FailedResult >> 32;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -221,13 +226,19 @@ bool HandleCASPAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return false;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
using CASExpectedFn = T (*)(T Src, T Expected);
|
||||
template <typename T>
|
||||
using CASDesiredFn = T (*)(T Src, T Desired);
|
||||
|
||||
template<bool Retry>
|
||||
static
|
||||
std::tuple<uint16_t, bool> DoCAS16(
|
||||
uint16_t DoCAS16(
|
||||
uint16_t DesiredSrc,
|
||||
uint16_t ExpectedSrc,
|
||||
uint64_t Addr,
|
||||
std::function<uint16_t(uint16_t SrcVal, uint16_t Expected)> ExpectedFunction,
|
||||
std::function<uint16_t(uint16_t SrcVal, uint16_t Desired)> DesiredFunction) {
|
||||
CASExpectedFn<uint16_t> ExpectedFunction,
|
||||
CASDesiredFn<uint16_t> DesiredFunction) {
|
||||
// 16 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) == 15) {
|
||||
@@ -235,49 +246,66 @@ std::tuple<uint16_t, bool> DoCAS16(
|
||||
// Need a dual 8bit CAS loop
|
||||
uint64_t AddrUpper = Addr + 1;
|
||||
|
||||
uint8_t ActualUpper{};
|
||||
uint8_t ActualLower{};
|
||||
// Careful ordering here
|
||||
ActualUpper = LoadAcquire8(AddrUpper);
|
||||
ActualLower = LoadAcquire8(Addr);
|
||||
while (1) {
|
||||
uint8_t ActualUpper{};
|
||||
uint8_t ActualLower{};
|
||||
// Careful ordering here
|
||||
ActualUpper = LoadAcquire8(AddrUpper);
|
||||
ActualLower = LoadAcquire8(Addr);
|
||||
|
||||
uint16_t Actual = ActualUpper;
|
||||
Actual <<= 8;
|
||||
Actual |= ActualLower;
|
||||
uint16_t Actual = ActualUpper;
|
||||
Actual <<= 8;
|
||||
Actual |= ActualLower;
|
||||
|
||||
uint16_t Desired = DesiredFunction(Actual, DesiredSrc);
|
||||
uint8_t DesiredLower = Desired;
|
||||
uint8_t DesiredUpper = Desired >> 8;
|
||||
uint16_t Desired = DesiredFunction(Actual, DesiredSrc);
|
||||
uint8_t DesiredLower = Desired;
|
||||
uint8_t DesiredUpper = Desired >> 8;
|
||||
|
||||
uint16_t Expected = ExpectedFunction(Actual, ExpectedSrc);
|
||||
uint8_t ExpectedLower = Expected;
|
||||
uint8_t ExpectedUpper = Expected >> 8;
|
||||
uint16_t Expected = ExpectedFunction(Actual, ExpectedSrc);
|
||||
uint8_t ExpectedLower = Expected;
|
||||
uint8_t ExpectedUpper = Expected >> 8;
|
||||
|
||||
if (ActualUpper == ExpectedUpper &&
|
||||
ActualLower == ExpectedLower) {
|
||||
if (StoreCAS8(ExpectedUpper, DesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS8(ExpectedLower, DesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return std::make_tuple(Expected, true);
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
bool Tear = false;
|
||||
if (ActualUpper == ExpectedUpper &&
|
||||
ActualLower == ExpectedLower) {
|
||||
if (StoreCAS8(ExpectedUpper, DesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS8(ExpectedLower, DesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return Expected;
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
}
|
||||
}
|
||||
|
||||
ActualLower = ExpectedLower;
|
||||
ActualUpper = ExpectedUpper;
|
||||
}
|
||||
|
||||
ActualLower = ExpectedLower;
|
||||
ActualUpper = ExpectedUpper;
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint16_t FailedResult = ActualUpper;
|
||||
FailedResult <<= 8;
|
||||
FailedResult |= ActualLower;
|
||||
|
||||
if constexpr (Retry) {
|
||||
if (Tear) {
|
||||
// If we are retrying and tearing then we can't do anything here
|
||||
// XXX: Resolve with TME
|
||||
return FailedResult;
|
||||
}
|
||||
else {
|
||||
// We can retry safely
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Without Retry (CAS) then we have failed regardless of tear
|
||||
// CAS failed but handled successfully
|
||||
return FailedResult;
|
||||
}
|
||||
}
|
||||
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint16_t FailedResult = ActualUpper;
|
||||
FailedResult <<= 8;
|
||||
FailedResult |= ActualLower;
|
||||
|
||||
// CAS failed but handled successfully
|
||||
return std::make_tuple(FailedResult, false);
|
||||
}
|
||||
else {
|
||||
AlignmentMask = 0b111;
|
||||
@@ -316,28 +344,30 @@ std::tuple<uint16_t, bool> DoCAS16(
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return std::make_tuple(Expected >> (Alignment * 8), true);
|
||||
return Expected >> (Alignment * 8);
|
||||
}
|
||||
else {
|
||||
if constexpr (Retry) {
|
||||
// If we failed but we have enabled retry then just retry without checking results
|
||||
// CAS can't retry but atomic memory ops need to retry until passing
|
||||
continue;
|
||||
}
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
LogMan::Msg::D("Expected 0x%04x, Desired 0x%04x, Result 0x%04x", (uint16_t)(Expected >> (Alignment * 8)), DesiredSrc, FailedResult);
|
||||
// CAS failed but handled successfully
|
||||
return std::make_tuple(FailedResult, false);
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return FailedResult;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -379,28 +409,31 @@ std::tuple<uint16_t, bool> DoCAS16(
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return std::make_tuple(Expected >> (Alignment * 8), true);
|
||||
return Expected >> (Alignment * 8);
|
||||
}
|
||||
else {
|
||||
if constexpr (Retry) {
|
||||
// If we failed but we have enabled retry then just retry without checking results
|
||||
// CAS can't retry but atomic memory ops need to retry until passing
|
||||
continue;
|
||||
}
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we can try again
|
||||
uint64_t FailedResultOurBits = TmpExpected & Mask;
|
||||
uint64_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
uint64_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
uint64_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return std::make_tuple(FailedResult, false);
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return FailedResult;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -442,28 +475,31 @@ std::tuple<uint16_t, bool> DoCAS16(
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return std::make_tuple(Expected >> (Alignment * 8), true);
|
||||
return Expected >> (Alignment * 8);
|
||||
}
|
||||
else {
|
||||
if constexpr (Retry) {
|
||||
// If we failed but we have enabled retry then just retry without checking results
|
||||
// CAS can't retry but atomic memory ops need to retry until passing
|
||||
continue;
|
||||
}
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we can try again
|
||||
uint32_t FailedResultOurBits = TmpExpected & Mask;
|
||||
uint32_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
uint32_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
uint32_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return std::make_tuple(FailedResult, false);
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return FailedResult;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -471,13 +507,14 @@ std::tuple<uint16_t, bool> DoCAS16(
|
||||
}
|
||||
}
|
||||
|
||||
template<bool Retry>
|
||||
static
|
||||
std::tuple<uint32_t, bool> DoCAS32(
|
||||
uint32_t DoCAS32(
|
||||
uint32_t DesiredSrc,
|
||||
uint32_t ExpectedSrc,
|
||||
uint64_t Addr,
|
||||
std::function<uint32_t(uint32_t SrcVal, uint32_t Expected)> ExpectedFunction,
|
||||
std::function<uint32_t(uint32_t SrcVal, uint32_t Desired)> DesiredFunction) {
|
||||
CASExpectedFn<uint32_t> ExpectedFunction,
|
||||
CASDesiredFn<uint32_t> DesiredFunction) {
|
||||
// 32 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 12) {
|
||||
@@ -509,6 +546,7 @@ std::tuple<uint32_t, bool> DoCAS32(
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired << (Alignment * 8);
|
||||
|
||||
bool Tear = false;
|
||||
if (TmpExpected == TmpActual) {
|
||||
uint32_t TmpExpectedLower = TmpExpected;
|
||||
uint32_t TmpExpectedUpper = TmpExpected >> 32;
|
||||
@@ -519,11 +557,12 @@ std::tuple<uint32_t, bool> DoCAS32(
|
||||
if (StoreCAS32(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS32(TmpExpectedLower, TmpDesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return std::make_tuple(Expected, true);
|
||||
return Expected;
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -541,18 +580,30 @@ std::tuple<uint32_t, bool> DoCAS32(
|
||||
uint64_t FailedResultOurBits = TmpExpected & Mask;
|
||||
uint64_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
uint64_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
uint64_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
|
||||
if constexpr (Retry) {
|
||||
if (Tear) {
|
||||
// If we are retrying and tearing then we can't do anything here
|
||||
// XXX: Resolve with TME
|
||||
return FailedResult;
|
||||
}
|
||||
else {
|
||||
// We can retry safely
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Without Retry (CAS) then we have failed regardless of tear
|
||||
// CAS failed but handled successfully
|
||||
return std::make_tuple(FailedResult, false);
|
||||
return FailedResult;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -591,27 +642,31 @@ std::tuple<uint32_t, bool> DoCAS32(
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Stored successfully
|
||||
return std::make_tuple(Expected, true);
|
||||
return Expected;
|
||||
}
|
||||
else {
|
||||
if constexpr (Retry) {
|
||||
// If we failed but we have enabled retry then just retry without checking results
|
||||
// CAS can't retry but atomic memory ops need to retry until passing
|
||||
continue;
|
||||
}
|
||||
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return std::make_tuple(FailedResult, false);
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return FailedResult;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -650,41 +705,46 @@ std::tuple<uint32_t, bool> DoCAS32(
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Stored successfully
|
||||
return std::make_tuple(Expected, true);
|
||||
return Expected;
|
||||
}
|
||||
else {
|
||||
if constexpr (Retry) {
|
||||
// If we failed but we have enabled retry then just retry without checking results
|
||||
// CAS can't retry but atomic memory ops need to retry until passing
|
||||
continue;
|
||||
}
|
||||
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we can try again
|
||||
uint64_t FailedResultOurBits = TmpExpected & Mask;
|
||||
uint64_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
uint64_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
uint64_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return std::make_tuple(FailedResult, false);
|
||||
}
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return FailedResult;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<bool Retry>
|
||||
static
|
||||
std::tuple<uint64_t, bool> DoCAS64(
|
||||
uint64_t DoCAS64(
|
||||
uint64_t DesiredSrc,
|
||||
uint64_t ExpectedSrc,
|
||||
uint64_t Addr,
|
||||
std::function<uint64_t(uint64_t SrcVal, uint64_t Expected)> ExpectedFunction,
|
||||
std::function<uint64_t(uint64_t SrcVal, uint64_t Desired)> DesiredFunction) {
|
||||
CASExpectedFn<uint64_t> ExpectedFunction,
|
||||
CASDesiredFn<uint64_t> DesiredFunction) {
|
||||
// 64bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
@@ -724,15 +784,17 @@ std::tuple<uint64_t, bool> DoCAS64(
|
||||
uint64_t TmpDesiredLower = TmpDesired;
|
||||
uint64_t TmpDesiredUpper = TmpDesired >> 64;
|
||||
|
||||
bool Tear = false;
|
||||
if (TmpExpected == TmpActual) {
|
||||
if (StoreCAS64(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) {
|
||||
if (StoreCAS64(TmpExpectedLower, TmpDesiredLower, Addr)) {
|
||||
// Stored successfully
|
||||
return std::make_tuple(Expected, true);
|
||||
return Expected;
|
||||
}
|
||||
else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -750,18 +812,30 @@ std::tuple<uint64_t, bool> DoCAS64(
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
|
||||
if constexpr (Retry) {
|
||||
if (Tear) {
|
||||
// If we are retrying and tearing then we can't do anything here
|
||||
// XXX: Resolve with TME
|
||||
return FailedResult;
|
||||
}
|
||||
else {
|
||||
// We can retry safely
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Without Retry (CAS) then we have failed regardless of tear
|
||||
// CAS failed but handled successfully
|
||||
return std::make_tuple(FailedResult, false);
|
||||
return FailedResult;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -796,35 +870,34 @@ std::tuple<uint64_t, bool> DoCAS64(
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Stored successfully
|
||||
return std::make_tuple(Expected, true);
|
||||
return Expected;
|
||||
}
|
||||
else {
|
||||
if constexpr (Retry) {
|
||||
// If we failed but we have enabled retry then just retry without checking results
|
||||
// CAS can't retry but atomic memory ops need to retry until passing
|
||||
continue;
|
||||
}
|
||||
|
||||
// Not successful
|
||||
// Now we need to check the results to see if we need to try again
|
||||
__uint128_t FailedResultOurBits = TmpExpected & Mask;
|
||||
__uint128_t FailedResultNotOurBits = TmpExpected & NegMask;
|
||||
|
||||
__uint128_t FailedDesiredOurBits = TmpDesired & Mask;
|
||||
__uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask;
|
||||
if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) {
|
||||
// If the bits changed that weren't part of our regular CAS then we need to try again
|
||||
continue;
|
||||
}
|
||||
if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) {
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
// We need to extract the bits and return them in EXPECTED
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return std::make_tuple(FailedResult, false);
|
||||
}
|
||||
|
||||
// If we got here, that means the CAS failed
|
||||
// NotOurBits didn't change and bits we cared about didn't change
|
||||
ERROR_AND_DIE("Impossible");
|
||||
// This happens in the case that between Load and CAS that something has store our desired in to the memory location
|
||||
// This means our CAS fails because what we wanted to store was already stored
|
||||
uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8);
|
||||
// CAS failed but handled successfully
|
||||
return FailedResult;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
@@ -855,7 +928,7 @@ bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
// 8bit can't be unaligned
|
||||
// Only need to handle 16, 32, 64
|
||||
if (Size == 2) {
|
||||
auto Res = DoCAS16(
|
||||
auto Res = DoCAS16<false>(
|
||||
mcontext->regs[DesiredReg],
|
||||
mcontext->regs[ExpectedReg],
|
||||
Addr,
|
||||
@@ -871,12 +944,12 @@ bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
mcontext->regs[ExpectedReg] = std::get<0>(Res);
|
||||
mcontext->regs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
else if (Size == 4) {
|
||||
auto Res = DoCAS32(
|
||||
auto Res = DoCAS32<false>(
|
||||
mcontext->regs[DesiredReg],
|
||||
mcontext->regs[ExpectedReg],
|
||||
Addr,
|
||||
@@ -892,12 +965,12 @@ bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
mcontext->regs[ExpectedReg] = std::get<0>(Res);
|
||||
mcontext->regs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
else if (Size == 8) {
|
||||
auto Res = DoCAS64(
|
||||
auto Res = DoCAS64<false>(
|
||||
mcontext->regs[DesiredReg],
|
||||
mcontext->regs[ExpectedReg],
|
||||
Addr,
|
||||
@@ -913,7 +986,7 @@ bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
mcontext->regs[ExpectedReg] = std::get<0>(Res);
|
||||
mcontext->regs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -964,7 +1037,7 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
std::function<uint16_t(uint16_t SrcVal, uint16_t Desired)> DesiredFunction;
|
||||
CASDesiredFn<uint16_t> DesiredFunction{};
|
||||
|
||||
switch (Op) {
|
||||
case ATOMIC_ADD_OP:
|
||||
@@ -988,21 +1061,16 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
break;
|
||||
}
|
||||
|
||||
bool Passed = false;
|
||||
while (!Passed) {
|
||||
auto Res = DoCAS16(
|
||||
mcontext->regs[SourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
DesiredFunction);
|
||||
Passed = std::get<1>(Res);
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (Passed &&
|
||||
ResultReg != 31) {
|
||||
mcontext->regs[ResultReg] = std::get<0>(Res);
|
||||
}
|
||||
auto Res = DoCAS16<true>(
|
||||
mcontext->regs[SourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
DesiredFunction);
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (ResultReg != 31) {
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1031,7 +1099,7 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
std::function<uint32_t(uint32_t SrcVal, uint32_t Desired)> DesiredFunction;
|
||||
CASDesiredFn<uint32_t> DesiredFunction{};
|
||||
|
||||
switch (Op) {
|
||||
case ATOMIC_ADD_OP:
|
||||
@@ -1055,21 +1123,16 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
break;
|
||||
}
|
||||
|
||||
bool Passed = false;
|
||||
while (!Passed) {
|
||||
auto Res = DoCAS32(
|
||||
mcontext->regs[SourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
DesiredFunction);
|
||||
Passed = std::get<1>(Res);
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (Passed &&
|
||||
ResultReg != 31) {
|
||||
mcontext->regs[ResultReg] = std::get<0>(Res);
|
||||
}
|
||||
auto Res = DoCAS32<true>(
|
||||
mcontext->regs[SourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
DesiredFunction);
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (ResultReg != 31) {
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -1098,7 +1161,7 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
return Desired;
|
||||
};
|
||||
|
||||
std::function<uint64_t(uint64_t SrcVal, uint64_t Desired)> DesiredFunction;
|
||||
CASDesiredFn<uint64_t> DesiredFunction{};
|
||||
|
||||
switch (Op) {
|
||||
case ATOMIC_ADD_OP:
|
||||
@@ -1122,21 +1185,16 @@ bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
break;
|
||||
}
|
||||
|
||||
bool Passed = false;
|
||||
while (!Passed) {
|
||||
auto Res = DoCAS64(
|
||||
mcontext->regs[SourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
DesiredFunction);
|
||||
Passed = std::get<1>(Res);
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (Passed &&
|
||||
ResultReg != 31) {
|
||||
mcontext->regs[ResultReg] = std::get<0>(Res);
|
||||
}
|
||||
auto Res = DoCAS64<true>(
|
||||
mcontext->regs[SourceReg],
|
||||
0, // Unused
|
||||
Addr,
|
||||
NOPExpected,
|
||||
DesiredFunction);
|
||||
// If we passed and our destination register is not zero
|
||||
// Then we need to update the result register with what was in memory
|
||||
if (ResultReg != 31) {
|
||||
mcontext->regs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
+444
-165
@@ -15,7 +15,26 @@ $end_info$
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
//#define CPUID_AMD
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
// #define CPUID_AMD
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
0 | // Stepping
|
||||
(0xA << 4) | // Model
|
||||
(0xF << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(0 << 16) | // Extended model ID
|
||||
(1 << 20); // Extended family ID
|
||||
#else
|
||||
constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
0 | // Stepping
|
||||
(0x7 << 4) | // Model
|
||||
(0x6 << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(1 << 16) | // Extended model ID
|
||||
(0x0 << 20); // Extended family ID
|
||||
#endif
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
static uint32_t GetCycleCounterFrequency() {
|
||||
uint64_t Result{};
|
||||
@@ -38,7 +57,7 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
}
|
||||
#endif
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
// EBX, EDX, ECX become the manufacturer id string
|
||||
@@ -57,25 +76,23 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h() {
|
||||
}
|
||||
|
||||
// Processor Info and Features bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
Res.eax = 0 | // Stepping
|
||||
(0 << 4) | // Model
|
||||
(0xF << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(0 << 16) | // Extended model ID
|
||||
(0 << 20); // Extended family ID
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(8 << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(CoreCount << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(0 << 24); // Local APIC ID
|
||||
|
||||
Res.ecx =
|
||||
(1 << 0) | // SSE3
|
||||
(0 << 1) | // PCLMULQDQ
|
||||
(1 << 2) | // DS area supports 64bit layout
|
||||
(1 << 3) | // MWait
|
||||
(1 << 4) | // DS-CPL
|
||||
(0 << 4) | // DS-CPL
|
||||
(0 << 5) | // VMX
|
||||
(0 << 6) | // SMX
|
||||
(0 << 7) | // Intel SpeedStep
|
||||
@@ -89,8 +106,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h() {
|
||||
(0 << 15) | // Perfmon and debug capability
|
||||
(0 << 16) | // Reserved
|
||||
(0 << 17) | // Process-context identifiers
|
||||
(1 << 18) | // Prefetching from memory mapped device
|
||||
(0 << 19) | // SSE4.1
|
||||
(0 << 18) | // Prefetching from memory mapped device
|
||||
(1 << 19) | // SSE4.1
|
||||
(0 << 20) | // SSE4.2
|
||||
(0 << 21) | // X2APIC
|
||||
(1 << 22) | // MOVBE
|
||||
@@ -99,7 +116,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h() {
|
||||
(CTX->HostFeatures.SupportsAES << 25) | // AES
|
||||
(0 << 26) | // XSAVE
|
||||
(0 << 27) | // OSXSAVE
|
||||
(0 << 28) | // AVX
|
||||
(SUPPORTS_AVX << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(0 << 30) | // RDRAND
|
||||
(0 << 31); // Hypervisor always returns zero
|
||||
@@ -132,16 +149,16 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h() {
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // SSE
|
||||
(1 << 26) | // SSE2
|
||||
(1 << 27) | // Self Snoop
|
||||
(0 << 27) | // Self Snoop
|
||||
(1 << 28) | // Max APIC IDs reserved field is valid
|
||||
(1 << 29) | // Thermal monitor
|
||||
(0 << 29) | // Thermal monitor
|
||||
(0 << 30) | // Reserved
|
||||
(1 << 31); // Pending break enable
|
||||
(0 << 31); // Pending break enable
|
||||
return Res;
|
||||
}
|
||||
|
||||
// 2: Cache and TLB information
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
// returns default values from i7 model 1Ah
|
||||
@@ -165,124 +182,286 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h() {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h() {
|
||||
// 4: Deterministic cache parameters for each level
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
constexpr uint32_t CacheType_Data = 1;
|
||||
constexpr uint32_t CacheType_Instruction = 2;
|
||||
constexpr uint32_t CacheType_Unified = 3;
|
||||
|
||||
if (Leaf == 0) {
|
||||
// Report L1D
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
|
||||
Res.eax = CacheType_Data | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
(1 << 8) | // Self initializing cache level
|
||||
(0 << 9) | // Fully associative
|
||||
(0 << 14) | // Maximum number of addressable IDs for logical processors sharing this cache (With SMT this would be 1)
|
||||
(CoreCount << 26); // Maximum number of addressable IDs for processor cores in the physical package
|
||||
|
||||
Res.ebx =
|
||||
(63 << 0) | // Line Size - 1 : Claiming 64 byte
|
||||
(0 << 12) | // Physical Line partitions
|
||||
(7 << 22); // Associativity - 1 : Claiming 8 way
|
||||
|
||||
// 32KB
|
||||
Res.ecx = 63; // Number of sets - 1 : Claiming 64 sets
|
||||
|
||||
Res.edx =
|
||||
(0 << 0) | // Write-back invalidate
|
||||
(0 << 1) | // Cache inclusiveness - Includes lower caches
|
||||
(0 << 2); // Complex cache indexing - 0: Direct, 1: Complex
|
||||
}
|
||||
else if (Leaf == 1) {
|
||||
// Report L1I
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
|
||||
Res.eax = CacheType_Instruction | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
(1 << 8) | // Self initializing cache level
|
||||
(0 << 9) | // Fully associative
|
||||
(0 << 14) | // Maximum number of addressable IDs for logical processors sharing this cache (With SMT this would be 1)
|
||||
(CoreCount << 26); // Maximum number of addressable IDs for processor cores in the physical package
|
||||
|
||||
Res.ebx =
|
||||
(63 << 0) | // Line Size - 1 : Claiming 64 byte
|
||||
(0 << 12) | // Physical Line partitions
|
||||
(7 << 22); // Associativity - 1 : Claiming 8 way
|
||||
|
||||
// 32KB
|
||||
Res.ecx = 63; // Number of sets - 1 : Claiming 64 sets
|
||||
|
||||
Res.edx =
|
||||
(0 << 0) | // Write-back invalidate
|
||||
(0 << 1) | // Cache inclusiveness - Includes lower caches
|
||||
(0 << 2); // Complex cache indexing - 0: Direct, 1: Complex
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
// Report L2
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b010 << 5) | // Cache level
|
||||
(1 << 8) | // Self initializing cache level
|
||||
(0 << 9) | // Fully associative
|
||||
(0 << 14) | // Maximum number of addressable IDs for logical processors sharing this cache
|
||||
(CoreCount << 26); // Maximum number of addressable IDs for processor cores in the physical package
|
||||
|
||||
Res.ebx =
|
||||
(63 << 0) | // Line Size - 1 : Claiming 64 byte
|
||||
(0 << 12) | // Physical Line partitions
|
||||
(7 << 22); // Associativity - 1 : Claiming 8 way
|
||||
|
||||
// 512KB
|
||||
Res.ecx = 0x3FF; // Number of sets - 1 : Claiming 1024 sets
|
||||
|
||||
Res.edx =
|
||||
(0 << 0) | // Write-back invalidate
|
||||
(0 << 1) | // Cache inclusiveness - Includes lower caches
|
||||
(0 << 2); // Complex cache indexing - 0: Direct, 1: Complex
|
||||
}
|
||||
else if (Leaf == 3) {
|
||||
// Report L3
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b011 << 5) | // Cache level
|
||||
(1 << 8) | // Self initializing cache level
|
||||
(0 << 9) | // Fully associative
|
||||
(CoreCount << 14) | // Maximum number of addressable IDs for logical processors sharing this cache
|
||||
(CoreCount << 26); // Maximum number of addressable IDs for processor cores in the physical package
|
||||
|
||||
Res.ebx =
|
||||
(63 << 0) | // Line Size - 1 : Claiming 64 byte
|
||||
(0 << 12) | // Physical Line partitions
|
||||
(7 << 22); // Associativity - 1 : Claiming 8 way
|
||||
|
||||
// 8MB
|
||||
Res.ecx = 0x4000; // Number of sets - 1 : Claiming 16384 sets
|
||||
|
||||
Res.edx =
|
||||
(0 << 0) | // Write-back invalidate
|
||||
(0 << 1) | // Cache inclusiveness - Includes lower caches
|
||||
(1 << 2); // Complex cache indexing - 0: Direct, 1: Complex
|
||||
}
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax = (1 << 2); // Always running APIC
|
||||
Res.ecx = (0 << 3); // Intel performance energy bias preference (EPB)
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
if (Leaf == 0) {
|
||||
// Number of subfunctions
|
||||
Res.eax = 0x0;
|
||||
Res.ebx =
|
||||
(1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(0 << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(0 << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(0 << 8) | // BMI2
|
||||
(0 << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
(0 << 12) | // Intel resource directory technology Monitoring
|
||||
(1 << 13) | // Deprecates FPU CS and DS
|
||||
(0 << 14) | // Intel MPX
|
||||
(0 << 15) | // Intel Resource Directory Technology Allocation
|
||||
(0 << 16) | // Reserved
|
||||
(0 << 17) | // Reserved
|
||||
(0 << 18) | // RDSEED
|
||||
(0 << 19) | // ADCX and ADOX instructions
|
||||
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
|
||||
(0 << 21) | // Reserved
|
||||
(0 << 22) | // Reserved
|
||||
(0 << 23) | // CLFLUSHOPT instruction
|
||||
(0 << 24) | // CLWB instruction
|
||||
(0 << 25) | // Intel processor trace
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
(0 << 28) | // Reserved
|
||||
(0 << 29) | // SHA instructions
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
// Number of subfunctions
|
||||
Res.eax = 0x0;
|
||||
Res.ebx =
|
||||
(1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(0 << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(0 << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(0 << 8) | // BMI2
|
||||
(0 << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
(0 << 12) | // Intel resource directory technology Monitoring
|
||||
(1 << 13) | // Deprecates FPU CS and DS
|
||||
(0 << 14) | // Intel MPX
|
||||
(0 << 15) | // Intel Resource Directory Technology Allocation
|
||||
(0 << 16) | // Reserved
|
||||
(0 << 17) | // Reserved
|
||||
(0 << 18) | // RDSEED
|
||||
(0 << 19) | // ADCX and ADOX instructions
|
||||
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
|
||||
(0 << 21) | // Reserved
|
||||
(0 << 22) | // Reserved
|
||||
(0 << 23) | // CLFLUSHOPT instruction
|
||||
(0 << 24) | // CLWB instruction
|
||||
(0 << 25) | // Intel processor trace
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
(0 << 28) | // Reserved
|
||||
(0 << 29) | // SHA instructions
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
Res.ecx =
|
||||
(1 << 0) | // PREFETCHWT1
|
||||
(0 << 1) | // AVX512VBMI
|
||||
(0 << 2) | // Usermode instruction prevention
|
||||
(0 << 3) | // Protection keys for user mode pages
|
||||
(0 << 4) | // OS protection keys
|
||||
(0 << 5) | // waitpkg
|
||||
(0 << 6) | // AVX512_VBMI2
|
||||
(0 << 7) | // CET shadow stack
|
||||
(0 << 8) | // GFNI
|
||||
(0 << 9) | // VAES
|
||||
(0 << 10) | // VPCLMULQDQ
|
||||
(0 << 11) | // AVX512_VNNI
|
||||
(0 << 12) | // AVX512_BITALG
|
||||
(0 << 13) | // Intel Total Memory Encryption
|
||||
(0 << 14) | // AVX512_VPOPCNTDQ
|
||||
(0 << 15) | // Reserved
|
||||
(0 << 16) | // 5 Level page tables
|
||||
(0 << 17) | // MPX MAWAU
|
||||
(0 << 18) | // MPX MAWAU
|
||||
(0 << 19) | // MPX MAWAU
|
||||
(0 << 20) | // MPX MAWAU
|
||||
(0 << 21) | // MPX MAWAU
|
||||
(0 << 22) | // RDPID Read Processor ID
|
||||
(0 << 23) | // Reserved
|
||||
(0 << 24) | // Reserved
|
||||
(0 << 25) | // CLDEMOTE
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // MOVDIRI
|
||||
(0 << 28) | // MOVDIR64B
|
||||
(0 << 29) | // Reserved
|
||||
(0 << 30) | // SGX Launch configuration
|
||||
(0 << 31); // Reserved
|
||||
|
||||
Res.ecx =
|
||||
(1 << 0) | // PREFETCHWT1
|
||||
(0 << 1) | // AVX512VBMI
|
||||
(0 << 2) | // Usermode instruction prevention
|
||||
(0 << 3) | // Protection keys for user mode pages
|
||||
(1 << 4) | // OS protection keys
|
||||
(0 << 5) | // waitpkg
|
||||
(0 << 6) | // AVX512_VBMI2
|
||||
(0 << 7) | // CET shadow stack
|
||||
(0 << 8) | // GFNI
|
||||
(0 << 9) | // VAES
|
||||
(0 << 10) | // VPCLMULQDQ
|
||||
(0 << 11) | // AVX512_VNNI
|
||||
(0 << 12) | // AVX512_BITALG
|
||||
(0 << 13) | // Intel Total Memory Encryption
|
||||
(0 << 14) | // AVX512_VPOPCNTDQ
|
||||
(0 << 15) | // Reserved
|
||||
(0 << 16) | // 5 Level page tables
|
||||
(0 << 17) | // MPX MAWAU
|
||||
(0 << 18) | // MPX MAWAU
|
||||
(0 << 19) | // MPX MAWAU
|
||||
(0 << 20) | // MPX MAWAU
|
||||
(0 << 21) | // MPX MAWAU
|
||||
(0 << 22) | // RDPID Read Processor ID
|
||||
(0 << 23) | // Reserved
|
||||
(0 << 24) | // Reserved
|
||||
(0 << 25) | // CLDEMOTE
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // MOVDIRI
|
||||
(0 << 28) | // MOVDIR64B
|
||||
(0 << 29) | // Reserved
|
||||
(0 << 30) | // SGX Launch configuration
|
||||
(0 << 31); // Reserved
|
||||
|
||||
Res.edx =
|
||||
(0 << 0) | // Reserved
|
||||
(0 << 1) | // Reserved
|
||||
(0 << 2) | // AVX512_4VNNIW
|
||||
(0 << 3) | // AVX512_4FMAPS
|
||||
(0 << 4) | // Fast Short Rep Mov
|
||||
(0 << 5) | // Reserved
|
||||
(0 << 6) | // Reserved
|
||||
(0 << 7) | // Reserved
|
||||
(0 << 8) | // AVX512_VP2INTERSECT
|
||||
(0 << 9) | // Reserved
|
||||
(0 << 10) | // VERW clears CPU buffers
|
||||
(0 << 11) | // Reserved
|
||||
(0 << 12) | // Reserved
|
||||
(0 << 13) | // Reserved
|
||||
(0 << 14) | // SERIALIZE instruction
|
||||
(0 << 15) | // Reserved
|
||||
(0 << 16) | // Reserved
|
||||
(0 << 17) | // Reserved
|
||||
(0 << 18) | // Intel PCONFIG
|
||||
(0 << 19) | // Intel Architectural LBR
|
||||
(0 << 20) | // Intel CET
|
||||
(0 << 21) | // Reserved
|
||||
(0 << 22) | // Reserved
|
||||
(0 << 23) | // Reserved
|
||||
(0 << 24) | // Reserved
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
(0 << 28) | // L1D Flush
|
||||
(0 << 29) | // Arch capabilities
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
Res.edx =
|
||||
(0 << 0) | // Reserved
|
||||
(0 << 1) | // Reserved
|
||||
(0 << 2) | // AVX512_4VNNIW
|
||||
(0 << 3) | // AVX512_4FMAPS
|
||||
(0 << 4) | // Fast Short Rep Mov
|
||||
(0 << 5) | // Reserved
|
||||
(0 << 6) | // Reserved
|
||||
(0 << 7) | // Reserved
|
||||
(0 << 8) | // AVX512_VP2INTERSECT
|
||||
(0 << 9) | // Reserved
|
||||
(0 << 10) | // VERW clears CPU buffers
|
||||
(0 << 11) | // Reserved
|
||||
(0 << 12) | // Reserved
|
||||
(0 << 13) | // Reserved
|
||||
(0 << 14) | // SERIALIZE instruction
|
||||
(0 << 15) | // Reserved
|
||||
(0 << 16) | // Reserved
|
||||
(0 << 17) | // Reserved
|
||||
(0 << 18) | // Intel PCONFIG
|
||||
(0 << 19) | // Intel Architectural LBR
|
||||
(0 << 20) | // Intel CET
|
||||
(0 << 21) | // Reserved
|
||||
(0 << 22) | // Reserved
|
||||
(0 << 23) | // Reserved
|
||||
(0 << 24) | // Reserved
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // Reserved
|
||||
(0 << 28) | // L1D Flush
|
||||
(0 << 29) | // Arch capabilities
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
}
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
// Leaf 0
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
uint32_t XFeatureSupportedSizeMax = SUPPORTS_AVX ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
|
||||
if (Leaf == 0) {
|
||||
// XFeatureSupportedMask[31:0]
|
||||
Res.eax =
|
||||
(1 << 0) | // X87 support
|
||||
(1 << 1) | // 128-bit SSE support
|
||||
(SUPPORTS_AVX << 2) | // 256-bit AVX support
|
||||
(0b00 << 3) | // MPX State
|
||||
(0b000 << 5) | // AVX-512 state
|
||||
(0 << 8) | // "Used for IA32_XSS" ... Used for what?
|
||||
(0 << 9); // PKRU state
|
||||
|
||||
// EBX and ECX doesn't need to match if a feature is supported but not enabled
|
||||
Res.ebx = XFeatureSupportedSizeMax;
|
||||
Res.ecx = XFeatureSupportedSizeMax; // XFeatureSupportedSizeMax: Size in bytes of XSAVE/XRSTOR area
|
||||
|
||||
// XFeatureSupportedMask[63:32]
|
||||
Res.edx = 0; // Upper 32-bits of XFeatureSupportedMask
|
||||
}
|
||||
else if (Leaf == 1) {
|
||||
Res.eax =
|
||||
(0 << 0) | // XSAVEOPT
|
||||
(0 << 1) | // XSAVEC (and XRSTOR)
|
||||
(0 << 2) | // XGETBV - XGETBV with ECX=1 supported
|
||||
(0 << 3); // XSAVES - XSAVES, XRSTORS, and IA32_XSS supported
|
||||
|
||||
// Same information as Leaf 0 for ebx
|
||||
Res.ebx = XFeatureSupportedSizeMax;
|
||||
|
||||
// Lower supported 32bits of IA32_XSS MSR. IA32_XSS[n] can only be set to 1 if ECX[n] is 1
|
||||
Res.ecx =
|
||||
(0b0000'0000 << 0) | // Used for XCR0
|
||||
(0 << 8) | // PT state
|
||||
(0 << 9); // Used for XCR0
|
||||
|
||||
// Upper supported 32bits of IA32_XSS MSR. IA32_XSS[n+32] can only be set to 1 if EDX[n] is 1
|
||||
// Entirely reserved atm
|
||||
Res.edx = 0;
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
Res.eax = SUPPORTS_AVX ? 0x0000'0100 : 0; // YmmSaveStateSize
|
||||
Res.ebx = SUPPORTS_AVX ? 0x0000'0240 : 0; // YmmSaveStateOffset
|
||||
|
||||
// Reserved
|
||||
Res.ecx = 0;
|
||||
Res.edx = 0;
|
||||
}
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
// TSC frequency = ECX * EBX / EAX
|
||||
uint32_t FrequencyHz = GetCycleCounterFrequency();
|
||||
@@ -295,7 +474,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h() {
|
||||
}
|
||||
|
||||
// Highest extended function implemented
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax = 0x8000001F;
|
||||
|
||||
@@ -314,15 +493,10 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h() {
|
||||
}
|
||||
|
||||
// Extended processor and feature bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
Res.eax = 0 | // Stepping
|
||||
(0 << 4) | // Model
|
||||
(0 << 8) | // Family ID
|
||||
(0 << 12) | // Processor type
|
||||
(0 << 16) | // Extended model ID
|
||||
(0 << 20); // Extended family ID
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
Res.ecx =
|
||||
(1 << 0) | // LAHF/SAHF
|
||||
@@ -347,13 +521,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h() {
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // Reserved
|
||||
(1 << 22) | // Topology extensions support
|
||||
(1 << 23) | // Core performance counter extensions
|
||||
(1 << 24) | // NB performance counter extensions
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(1 << 27) | // Performance TSC
|
||||
(1 << 28) | // L2 perf counter extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // Reserved
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
@@ -385,7 +559,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h() {
|
||||
(1 << 23) | // MMX
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(1 << 26) | // 1 gigabit pages
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(0 << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
@@ -400,26 +574,26 @@ constexpr char ProcessorBrand[48] = {
|
||||
};
|
||||
|
||||
//Processor brand string
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memcpy(&Res, &ProcessorBrand[0], sizeof(FEXCore::CPUID::FunctionResults));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memcpy(&Res, &ProcessorBrand[16], sizeof(FEXCore::CPUID::FunctionResults));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memcpy(&Res, &ProcessorBrand[32], sizeof(FEXCore::CPUID::FunctionResults));
|
||||
return Res;
|
||||
}
|
||||
|
||||
// L1 Cache and TLB identifiers
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0005h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0005h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
// L1 TLB Information for 2MB and 4MB pages
|
||||
@@ -454,7 +628,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0005h() {
|
||||
}
|
||||
|
||||
// L2 Cache identifiers
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0006h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0006h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
// L2 TLB Information for 2MB and 4MB pages
|
||||
@@ -488,7 +662,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0006h() {
|
||||
}
|
||||
|
||||
// Advanced power management
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0007h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0007h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax = (1 << 2); // APIC timer not affected by p-state
|
||||
Res.edx =
|
||||
@@ -497,7 +671,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0007h() {
|
||||
}
|
||||
|
||||
// Virtual and physical address sizes
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax =
|
||||
(48 << 0) | // PhysAddrSize = 48-bit
|
||||
@@ -512,14 +686,14 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h() {
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
Res.ecx =
|
||||
(0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
(0 << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
(CoreCount << 0); // Count count subtract one
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
// TLB 1GB page identifiers
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0019h() {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0019h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax =
|
||||
(0xF << 28) | // L1 DTLB associativity for 1GB pages
|
||||
@@ -535,27 +709,128 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0019h() {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved() {
|
||||
// Deterministic cache parameters for each level
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_001Dh(uint32_t Leaf) {
|
||||
// This is nearly a copy of CPUID function 4h
|
||||
// There are some minor changes though
|
||||
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
constexpr uint32_t CacheType_Data = 1;
|
||||
constexpr uint32_t CacheType_Instruction = 2;
|
||||
constexpr uint32_t CacheType_Unified = 3;
|
||||
|
||||
if (Leaf == 0) {
|
||||
// Report L1D
|
||||
Res.eax = CacheType_Data | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
(1 << 8) | // Self initializing cache level
|
||||
(0 << 9) | // Fully associative
|
||||
(0 << 14); // Maximum number of addressable IDs for logical processors sharing this cache (With SMT this would be 1)
|
||||
|
||||
Res.ebx =
|
||||
(63 << 0) | // Line Size - 1 : Claiming 64 byte
|
||||
(0 << 12) | // Physical Line partitions
|
||||
(7 << 22); // Associativity - 1 : Claiming 8 way
|
||||
|
||||
// 32KB
|
||||
Res.ecx = 63; // Number of sets - 1 : Claiming 64 sets
|
||||
|
||||
Res.edx =
|
||||
(0 << 0) | // Write-back invalidate
|
||||
(0 << 1); // Cache inclusiveness - Includes lower caches
|
||||
}
|
||||
else if (Leaf == 1) {
|
||||
// Report L1I
|
||||
Res.eax = CacheType_Instruction | // Cache type
|
||||
(0b001 << 5) | // Cache level
|
||||
(1 << 8) | // Self initializing cache level
|
||||
(0 << 9) | // Fully associative
|
||||
(0 << 14); // Maximum number of addressable IDs for logical processors sharing this cache (With SMT this would be 1)
|
||||
|
||||
Res.ebx =
|
||||
(63 << 0) | // Line Size - 1 : Claiming 64 byte
|
||||
(0 << 12) | // Physical Line partitions
|
||||
(7 << 22); // Associativity - 1 : Claiming 8 way
|
||||
|
||||
// 32KB
|
||||
Res.ecx = 63; // Number of sets - 1 : Claiming 64 sets
|
||||
|
||||
Res.edx =
|
||||
(0 << 0) | // Write-back invalidate
|
||||
(0 << 1); // Cache inclusiveness - Includes lower caches
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
// Report L2
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b010 << 5) | // Cache level
|
||||
(1 << 8) | // Self initializing cache level
|
||||
(0 << 9) | // Fully associative
|
||||
(0 << 14); // Maximum number of addressable IDs for logical processors sharing this cache
|
||||
|
||||
Res.ebx =
|
||||
(63 << 0) | // Line Size - 1 : Claiming 64 byte
|
||||
(0 << 12) | // Physical Line partitions
|
||||
(7 << 22); // Associativity - 1 : Claiming 8 way
|
||||
|
||||
// 512KB
|
||||
Res.ecx = 0x3FF; // Number of sets - 1 : Claiming 1024 sets
|
||||
|
||||
Res.edx =
|
||||
(0 << 0) | // Write-back invalidate
|
||||
(0 << 1); // Cache inclusiveness - Includes lower caches
|
||||
}
|
||||
else if (Leaf == 3) {
|
||||
// Report L3
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
|
||||
Res.eax = CacheType_Unified | // Cache type
|
||||
(0b011 << 5) | // Cache level
|
||||
(1 << 8) | // Self initializing cache level
|
||||
(0 << 9) | // Fully associative
|
||||
(CoreCount << 14); // Maximum number of addressable IDs for logical processors sharing this cache
|
||||
|
||||
Res.ebx =
|
||||
(63 << 0) | // Line Size - 1 : Claiming 64 byte
|
||||
(0 << 12) | // Physical Line partitions
|
||||
(7 << 22); // Associativity - 1 : Claiming 8 way
|
||||
|
||||
// 8MB
|
||||
Res.ecx = 0x4000; // Number of sets - 1 : Claiming 16384 sets
|
||||
|
||||
Res.edx =
|
||||
(0 << 0) | // Write-back invalidate
|
||||
(0 << 1); // Cache inclusiveness - Includes lower caches
|
||||
}
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
return Res;
|
||||
}
|
||||
|
||||
void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
CTX = ctx;
|
||||
RegisterFunction(0, std::bind(&CPUIDEmu::Function_0h, this));
|
||||
RegisterFunction(1, std::bind(&CPUIDEmu::Function_01h, this));
|
||||
RegisterFunction(2, std::bind(&CPUIDEmu::Function_02h, this));
|
||||
using namespace std::placeholders;
|
||||
RegisterFunction(0, std::bind(&CPUIDEmu::Function_0h, this, _1));
|
||||
RegisterFunction(1, std::bind(&CPUIDEmu::Function_01h, this, _1));
|
||||
RegisterFunction(2, std::bind(&CPUIDEmu::Function_02h, this, _1));
|
||||
// 3: Serial Number(previously), now reserved
|
||||
// 4: Deterministic cache parameters for each level
|
||||
#ifndef CPUID_AMD
|
||||
// Deterministic cache parameters for each level
|
||||
RegisterFunction(0x4, std::bind(&CPUIDEmu::Function_04h, this, _1));
|
||||
#endif
|
||||
// 5: Monitor/mwait
|
||||
// Thermal and power management
|
||||
RegisterFunction(6, std::bind(&CPUIDEmu::Function_06h, this));
|
||||
RegisterFunction(6, std::bind(&CPUIDEmu::Function_06h, this, _1));
|
||||
// Extended feature flags
|
||||
RegisterFunction(7, std::bind(&CPUIDEmu::Function_07h, this));
|
||||
RegisterFunction(7, std::bind(&CPUIDEmu::Function_07h, this, _1));
|
||||
// 9: Direct Cache Access information
|
||||
// 0x0A: Architectural performance monitoring
|
||||
// 0x0B: Extended topology enumeration
|
||||
// 0x0D: Processor extended state enumeration
|
||||
RegisterFunction(0x0D, std::bind(&CPUIDEmu::Function_0Dh, this, _1));
|
||||
// 0x0F: Intel RDT monitoring
|
||||
// 0x10: Intel RDT allocation enumeration
|
||||
// 0x12: Intel SGX capability enumeration
|
||||
@@ -564,43 +839,47 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
#ifndef CPUID_AMD
|
||||
// Timestamp counter information
|
||||
// Doesn't exist on AMD hardware
|
||||
RegisterFunction(0x15, std::bind(&CPUIDEmu::Function_15h, this));
|
||||
RegisterFunction(0x15, std::bind(&CPUIDEmu::Function_15h, this, _1));
|
||||
#endif
|
||||
// 0x16: Processor frequency information
|
||||
// 0x17: SoC vendor attribute enumeration
|
||||
|
||||
// Largest extended function number
|
||||
RegisterFunction(0x8000'0000, std::bind(&CPUIDEmu::Function_8000_0000h, this));
|
||||
RegisterFunction(0x8000'0000, std::bind(&CPUIDEmu::Function_8000_0000h, this, _1));
|
||||
// Processor vendor
|
||||
RegisterFunction(0x8000'0001, std::bind(&CPUIDEmu::Function_8000_0001h, this));
|
||||
RegisterFunction(0x8000'0001, std::bind(&CPUIDEmu::Function_8000_0001h, this, _1));
|
||||
// Processor brand string
|
||||
RegisterFunction(0x8000'0002, std::bind(&CPUIDEmu::Function_8000_0002h, this));
|
||||
RegisterFunction(0x8000'0002, std::bind(&CPUIDEmu::Function_8000_0002h, this, _1));
|
||||
// Processor brand string continued
|
||||
RegisterFunction(0x8000'0003, std::bind(&CPUIDEmu::Function_8000_0003h, this));
|
||||
RegisterFunction(0x8000'0003, std::bind(&CPUIDEmu::Function_8000_0003h, this, _1));
|
||||
// Processor brand string continued
|
||||
RegisterFunction(0x8000'0004, std::bind(&CPUIDEmu::Function_8000_0004h, this));
|
||||
RegisterFunction(0x8000'0004, std::bind(&CPUIDEmu::Function_8000_0004h, this, _1));
|
||||
// 0x8000'0005: L1 Cache and TLB identifiers
|
||||
#ifdef CPUID_AMD
|
||||
RegisterFunction(0x8000'0005, std::bind(&CPUIDEmu::Function_8000_0005h, this));
|
||||
RegisterFunction(0x8000'0005, std::bind(&CPUIDEmu::Function_8000_0005h, this, _1));
|
||||
#else
|
||||
// This is full reserved on Intel platforms
|
||||
RegisterFunction(0x8000'0005, std::bind(&CPUIDEmu::Function_Reserved, this));
|
||||
RegisterFunction(0x8000'0005, std::bind(&CPUIDEmu::Function_Reserved, this, _1));
|
||||
#endif
|
||||
// 0x8000'0006: L2 Cache identifiers
|
||||
RegisterFunction(0x8000'0006, std::bind(&CPUIDEmu::Function_8000_0006h, this));
|
||||
RegisterFunction(0x8000'0006, std::bind(&CPUIDEmu::Function_8000_0006h, this, _1));
|
||||
// Advanced power management information
|
||||
RegisterFunction(0x8000'0007, std::bind(&CPUIDEmu::Function_8000_0007h, this));
|
||||
RegisterFunction(0x8000'0007, std::bind(&CPUIDEmu::Function_8000_0007h, this, _1));
|
||||
// Virtual and physical address sizes
|
||||
RegisterFunction(0x8000'0008, std::bind(&CPUIDEmu::Function_8000_0008h, this));
|
||||
RegisterFunction(0x8000'0008, std::bind(&CPUIDEmu::Function_8000_0008h, this, _1));
|
||||
|
||||
// 0x8000'000A: SVM Revision
|
||||
// TLB 1GB page identifiers
|
||||
RegisterFunction(0x8000'0019, std::bind(&CPUIDEmu::Function_8000_0019h, this));
|
||||
RegisterFunction(0x8000'0019, std::bind(&CPUIDEmu::Function_8000_0019h, this, _1));
|
||||
|
||||
// 0x8000'001A: Performance optimization identifiers
|
||||
// 0x8000'001B: Instruction based sampling identifiers
|
||||
// 0x8000'001C: Lightweight profiling capabilities
|
||||
// 0x8000'001D: Cache properties
|
||||
#ifdef CPUID_AMD
|
||||
// Deterministic cache parameters for each level
|
||||
RegisterFunction(0x8000'001D, std::bind(&CPUIDEmu::Function_8000_001Dh, this, _1));
|
||||
#endif
|
||||
// 0x8000'001E: Extended APIC ID
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
}
|
||||
|
||||
+26
-23
@@ -24,23 +24,23 @@ private:
|
||||
public:
|
||||
void Init(FEXCore::Context::Context *ctx);
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, [[maybe_unused]] uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) {
|
||||
auto Handler = FunctionHandlers.find(Function);
|
||||
|
||||
if (Handler == FunctionHandlers.end()) {
|
||||
#ifndef NDEBUG
|
||||
LogMan::Msg::E("Unhandled CPU ID function, 0x%x", Function);
|
||||
LogMan::Msg::E("Unhandled CPU ID function, 0x%x-0x%x", Function, Leaf);
|
||||
#endif
|
||||
return Function_Reserved();
|
||||
return Function_Reserved(Leaf);
|
||||
}
|
||||
|
||||
return Handler->second();
|
||||
return Handler->second(Leaf);
|
||||
}
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
|
||||
using FunctionHandler = std::function<FEXCore::CPUID::FunctionResults()>;
|
||||
using FunctionHandler = std::function<FEXCore::CPUID::FunctionResults(uint32_t Leaf)>;
|
||||
void RegisterFunction(uint32_t Function, FunctionHandler Handler) {
|
||||
FunctionHandlers[Function] = Handler;
|
||||
}
|
||||
@@ -48,23 +48,26 @@ private:
|
||||
std::unordered_map<uint32_t, FunctionHandler> FunctionHandlers;
|
||||
|
||||
// Functions
|
||||
FEXCore::CPUID::FunctionResults Function_0h();
|
||||
FEXCore::CPUID::FunctionResults Function_01h();
|
||||
FEXCore::CPUID::FunctionResults Function_02h();
|
||||
FEXCore::CPUID::FunctionResults Function_06h();
|
||||
FEXCore::CPUID::FunctionResults Function_07h();
|
||||
FEXCore::CPUID::FunctionResults Function_15h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0005h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0006h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0007h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0008h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0009h();
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0019h();
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved();
|
||||
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_01h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_02h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_04h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_06h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_07h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0005h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0008h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0009h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
|
||||
};
|
||||
}
|
||||
+192
-147
@@ -52,8 +52,6 @@ namespace FEXCore::CPU {
|
||||
}
|
||||
}
|
||||
|
||||
static std::mutex AOTIRCacheLock;
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct ThreadLocalData {
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
@@ -112,41 +110,56 @@ constexpr std::array<std::string_view const, 16> RegNames = {
|
||||
std::string_view const& GetGRegName(unsigned Reg) {
|
||||
return RegNames[Reg];
|
||||
}
|
||||
|
||||
namespace DefaultFallbackCore {
|
||||
class DefaultFallbackCore final : public FEXCore::CPU::CPUBackend {
|
||||
public:
|
||||
explicit DefaultFallbackCore(FEXCore::Core::ThreadState *Thread)
|
||||
: ThreadState {reinterpret_cast<FEXCore::Core::InternalThreadState*>(Thread)} {
|
||||
}
|
||||
~DefaultFallbackCore() override = default;
|
||||
|
||||
std::string GetName() override { return "Default Fallback"; }
|
||||
|
||||
void *MapRegion(void *HostPtr, uint64_t VirtualGuestPtr, uint64_t Size) override {
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
void Initialize() override {}
|
||||
bool NeedsOpDispatch() override { return false; }
|
||||
|
||||
void *CompileCode(uint64_t Entry, FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override {
|
||||
LogMan::Msg::E("Fell back to default code handler at RIP: 0x%lx", ThreadState->CurrentFrame->State.rip);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
};
|
||||
|
||||
FEXCore::CPU::CPUBackend *CPUCreationFactory(FEXCore::Context::Context* CTX, FEXCore::Core::ThreadState *Thread) {
|
||||
return new DefaultFallbackCore(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void Context::AOTIRCaptureCacheWriteoutQueue_Flush() {
|
||||
{
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
AOTIRCaptureCacheWriteoutLock.lock();
|
||||
std::function<void()> fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
bool MaybeEmpty = false;
|
||||
AOTIRCaptureCacheWriteoutQueue.pop();
|
||||
MaybeEmpty = AOTIRCaptureCacheWriteoutQueue.size() == 0;
|
||||
AOTIRCaptureCacheWriteoutLock.unlock();
|
||||
|
||||
fn();
|
||||
if (MaybeEmpty) {
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A("Must never get here");
|
||||
}
|
||||
|
||||
void Context::AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn) {
|
||||
bool Flush = false;
|
||||
|
||||
{
|
||||
std::unique_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
AOTIRCaptureCacheWriteoutQueue.push(fn);
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() > 10000) {
|
||||
Flush = true;
|
||||
}
|
||||
}
|
||||
|
||||
bool test_val = false;
|
||||
if (Flush && AOTIRCaptureCacheWriteoutFlusing.compare_exchange_strong(test_val, true)) {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
}
|
||||
}
|
||||
|
||||
Context::Context() {
|
||||
#ifdef BLOCKSTATS
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
@@ -177,14 +190,6 @@ namespace FEXCore::Context {
|
||||
Threads.clear();
|
||||
}
|
||||
|
||||
// AOTIRCaptureCache needs manual clear
|
||||
for (auto &Mod: AOTIRCaptureCache) {
|
||||
for (auto &Entry: Mod.second) {
|
||||
delete Entry.second.IR;
|
||||
FEXCore::Allocator::free(Entry.second.RAData);
|
||||
}
|
||||
}
|
||||
|
||||
for (auto &Mod: AOTIRCache) {
|
||||
FEXCore::Allocator::munmap(Mod.second.mapping, Mod.second.size);
|
||||
}
|
||||
@@ -242,12 +247,12 @@ namespace FEXCore::Context {
|
||||
Thread->CPUBackend->CallbackPtr(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
|
||||
void Context::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func) {
|
||||
SignalDelegation->RegisterHostSignalHandler(Signal, Func);
|
||||
void Context::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SignalDelegation->RegisterHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void Context::RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func) {
|
||||
SignalDelegation->RegisterFrontendHostSignalHandler(Signal, Func);
|
||||
void Context::RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SignalDelegation->RegisterFrontendHostSignalHandler(Signal, Func, Required);
|
||||
}
|
||||
|
||||
void Context::WaitForIdle() {
|
||||
@@ -282,7 +287,7 @@ namespace FEXCore::Context {
|
||||
// Tell all the threads that they should pause
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::SIGNALEVENT_PAUSE);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Pause);
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
// Only attempt to stop this thread if it is running
|
||||
tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
@@ -303,7 +308,7 @@ namespace FEXCore::Context {
|
||||
// Spin up all the threads
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::SIGNALEVENT_RETURN);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Return);
|
||||
Thread->RunningEvents.WaitingToStart.store(true);
|
||||
}
|
||||
|
||||
@@ -369,8 +374,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
StopThread(Thread);
|
||||
} else {
|
||||
LogMan::Msg::D("Skipping thread %p: Already stopped", Thread);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -383,7 +386,7 @@ namespace FEXCore::Context {
|
||||
|
||||
void Context::StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Thread->RunningEvents.Running.exchange(false)) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::SIGNALEVENT_STOP);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Stop);
|
||||
tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
@@ -411,7 +414,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
int Context::GetProgramStatus() {
|
||||
int Context::GetProgramStatus() const {
|
||||
return ParentThread->StatusCode;
|
||||
}
|
||||
|
||||
@@ -442,8 +445,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::InitializeThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
// This will create the execution thread but it won't actually start executing
|
||||
ExecutionThreadHandler *Arg = reinterpret_cast<ExecutionThreadHandler*>(FEXCore::Allocator::malloc(sizeof(ExecutionThreadHandler)));
|
||||
Arg->This = this;
|
||||
@@ -485,21 +486,23 @@ namespace FEXCore::Context {
|
||||
// Create CPU backend
|
||||
switch (Config.Core) {
|
||||
case FEXCore::Config::CONFIG_INTERPRETER:
|
||||
State->CPUBackend.reset(FEXCore::CPU::CreateInterpreterCore(this, State, CompileThread));
|
||||
State->CPUBackend = FEXCore::CPU::CreateInterpreterCore(this, State, CompileThread);
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
State->PassManager->InsertRegisterAllocationPass(DoSRA);
|
||||
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
State->CPUBackend.reset(FEXCore::CPU::CreateX86JITCore(this, State, CompileThread));
|
||||
State->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, State, CompileThread);
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
State->CPUBackend.reset(FEXCore::CPU::CreateArm64JITCore(this, State, CompileThread));
|
||||
State->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, State, CompileThread);
|
||||
#else
|
||||
ERROR_AND_DIE("FEXCore has been compiled without a viable JIT core");
|
||||
#endif
|
||||
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM: State->CPUBackend.reset(CustomCPUFactory(this, State)); break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM:
|
||||
State->CPUBackend = CustomCPUFactory(this, State);
|
||||
break;
|
||||
default: ERROR_AND_DIE("Unknown core configuration");
|
||||
}
|
||||
}
|
||||
@@ -522,6 +525,7 @@ namespace FEXCore::Context {
|
||||
Thread->ThreadManager.parent_tid = ParentTID;
|
||||
|
||||
InitializeCompiler(Thread, false);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
return Thread;
|
||||
}
|
||||
@@ -579,6 +583,9 @@ namespace FEXCore::Context {
|
||||
|
||||
// We now only have one thread
|
||||
IdleWaitRefCount = 1;
|
||||
|
||||
// Clean up dead stacks
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
}
|
||||
|
||||
void Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length) {
|
||||
@@ -597,7 +604,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
std::tuple<FEXCore::IR::IRListView *, FEXCore::IR::RegisterAllocationData *, uint64_t, uint64_t, uint64_t, uint64_t> Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
uint8_t const *GuestCode{};
|
||||
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
|
||||
|
||||
@@ -607,14 +614,14 @@ namespace FEXCore::Context {
|
||||
uint64_t TotalInstructionsLength {0};
|
||||
|
||||
if (!Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP)) {
|
||||
return { nullptr, nullptr, 0, 0, 0, 0 };
|
||||
return {};
|
||||
}
|
||||
|
||||
auto CodeBlocks = Thread->FrontendDecoder->GetDecodedBlocks();
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks);
|
||||
|
||||
uint8_t GPRSize = Config.Is64BitMode ? 8 : 4;
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
FEXCore::Frontend::Decoder::DecodedBlocks const &Block = CodeBlocks->at(j);
|
||||
@@ -694,7 +701,7 @@ namespace FEXCore::Context {
|
||||
return { nullptr, nullptr, 0, 0, 0, 0 };
|
||||
}
|
||||
else {
|
||||
uint8_t GPRSize = Config.Is64BitMode ? 8 : 4;
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry + BlockInstructionsLength - GuestRIP, GPRSize));
|
||||
@@ -764,7 +771,6 @@ namespace FEXCore::Context {
|
||||
LogMan::Msg::I("two:\n %s", out2.str().c_str());
|
||||
LOGMAN_MSG_A("Parsed ir doesn't match\n");
|
||||
}
|
||||
delete reparsed;
|
||||
}
|
||||
}
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
@@ -786,7 +792,14 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
return {IRList, RAData.release(), TotalInstructions, TotalInstructionsLength, Thread->FrontendDecoder->DecodedMinAddress, Thread->FrontendDecoder->DecodedMaxAddress - Thread->FrontendDecoder->DecodedMinAddress };
|
||||
return {
|
||||
.IRList = IRList,
|
||||
.RAData = RAData.release(),
|
||||
.TotalInstructions = TotalInstructions,
|
||||
.TotalInstructionsLength = TotalInstructionsLength,
|
||||
.StartAddr = Thread->FrontendDecoder->DecodedMinAddress,
|
||||
.Length = Thread->FrontendDecoder->DecodedMaxAddress - Thread->FrontendDecoder->DecodedMinAddress,
|
||||
};
|
||||
}
|
||||
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::GetInlineEntry(uint64_t DataOffset) {
|
||||
@@ -824,7 +837,29 @@ namespace FEXCore::Context {
|
||||
return (IR::IRListView *)&InlineData[Offset];
|
||||
}
|
||||
|
||||
std::tuple<void *, FEXCore::IR::IRListView *, FEXCore::Core::DebugData *, FEXCore::IR::RegisterAllocationData *, bool, uint64_t, uint64_t> Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
auto Inserted = Index.emplace(GuestRIP, Stream->tellp());
|
||||
|
||||
if (Inserted.second) {
|
||||
//GuestHash
|
||||
Stream->write((const char*)&Hash, sizeof(Hash));
|
||||
|
||||
//GuestLength
|
||||
Stream->write((const char*)&Length, sizeof(Length));
|
||||
|
||||
// RAData (inline)
|
||||
// In file, IsShared is always set
|
||||
auto Shared = RAData->IsShared;
|
||||
RAData->IsShared = true;
|
||||
Stream->write((const char*)RAData, RAData->Size(RAData->MapCount));
|
||||
RAData->IsShared = Shared;
|
||||
|
||||
// IRData (inline)
|
||||
IRList->Serialize(*Stream);
|
||||
}
|
||||
}
|
||||
|
||||
Context::CompileCodeResult Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
@@ -848,7 +883,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
@@ -860,7 +895,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
if (IRList == nullptr && Config.AOTIRLoad) {
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
@@ -922,10 +957,18 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
if (IRList == nullptr) {
|
||||
return { nullptr, nullptr, nullptr, nullptr, false, 0, 0 };
|
||||
return {};
|
||||
}
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
return { Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData), IRList, DebugData, RAData, GeneratedIR, StartAddr, Length};
|
||||
return {
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData),
|
||||
.IRData = IRList,
|
||||
.DebugData = DebugData,
|
||||
.RAData = RAData,
|
||||
.GeneratedIR = GeneratedIR,
|
||||
.StartAddr = StartAddr,
|
||||
.Length = Length,
|
||||
};
|
||||
}
|
||||
|
||||
static bool readAll(int fd, void *data, size_t size) {
|
||||
@@ -939,31 +982,44 @@ namespace FEXCore::Context {
|
||||
|
||||
bool Context::LoadAOTIRCache(int streamfd) {
|
||||
uint64_t tag;
|
||||
|
||||
if (!readAll(streamfd, (char*)&tag, sizeof(tag)) || tag != 0xDEADBEEFC0D30003)
|
||||
|
||||
if (!readAll(streamfd, (char*)&tag, sizeof(tag)) || tag != 0xDEADBEEFC0D30004)
|
||||
return false;
|
||||
|
||||
|
||||
std::string Module;
|
||||
uint64_t ModSize;
|
||||
|
||||
uint64_t IndexSize;
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&ModSize, sizeof(ModSize)))
|
||||
return false;
|
||||
|
||||
Module.resize(ModSize);
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize, SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&Module[0], Module.size()))
|
||||
return false;
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize - sizeof(IndexSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&IndexSize, sizeof(IndexSize)))
|
||||
return false;
|
||||
|
||||
struct stat fileinfo;
|
||||
if (fstat(streamfd, &fileinfo) < 0)
|
||||
return false;
|
||||
size_t Size = (fileinfo.st_size + 4095) & ~4095;
|
||||
|
||||
size_t IndexOffset = fileinfo.st_size - IndexSize -sizeof(ModSize) - ModSize - sizeof(IndexSize);
|
||||
|
||||
void *FilePtr = FEXCore::Allocator::mmap(nullptr, Size, PROT_READ, MAP_SHARED, streamfd, 0);
|
||||
|
||||
if (FilePtr == MAP_FAILED)
|
||||
return false;
|
||||
|
||||
auto Array = (AOTIRInlineIndex *)((char*)FilePtr + sizeof(tag) + sizeof(ModSize) + ((ModSize+31) & ~31));
|
||||
auto Array = (AOTIRInlineIndex *)((char*)FilePtr + IndexOffset);
|
||||
|
||||
AOTIRCache.insert({Module, {Array, FilePtr, Size}});
|
||||
|
||||
@@ -974,83 +1030,53 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for( const auto &File: FilesWithCode) {
|
||||
Writer(File.first, File.second);
|
||||
}
|
||||
}
|
||||
|
||||
bool Context::WriteAOTIRCache(std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
void Context::FinalizeAOTIRCache() {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
|
||||
bool rv = true;
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
for (auto AOTModule: AOTIRCaptureCache) {
|
||||
if (AOTModule.second.size() == 0) {
|
||||
for (auto& [String, Entry] : AOTIRCaptureCache) {
|
||||
if (!Entry.Stream) {
|
||||
continue;
|
||||
}
|
||||
|
||||
auto stream = CacheWriter(AOTModule.first);
|
||||
if (!*stream) {
|
||||
rv = false;
|
||||
}
|
||||
uint64_t tag = 0xDEADBEEFC0D30003;
|
||||
stream->write((char*)&tag, sizeof(tag));
|
||||
const auto ModSize = String.size();
|
||||
auto &stream = Entry.Stream;
|
||||
|
||||
auto ModSize = AOTModule.first.size();
|
||||
stream->write((char*)&ModSize, sizeof(ModSize));
|
||||
stream->write((char*)&AOTModule.first[0], ModSize);
|
||||
|
||||
auto Skip = ((ModSize + 31) & ~31) - ModSize;
|
||||
char Zero = 0;
|
||||
for (int i = 0; i < Skip; i++)
|
||||
// pad to 32 bytes
|
||||
constexpr char Zero = 0;
|
||||
while(stream->tellp() & 31)
|
||||
stream->write(&Zero, 1);
|
||||
|
||||
// AOTIRInlineIndex
|
||||
|
||||
const auto FnCount = Entry.Index.size();
|
||||
const size_t DataBase = -stream->tellp();
|
||||
|
||||
auto FnCount = AOTModule.second.size();
|
||||
stream->write((char*)&FnCount, sizeof(FnCount));
|
||||
stream->write((const char*)&FnCount, sizeof(FnCount));
|
||||
stream->write((const char*)&DataBase, sizeof(DataBase));
|
||||
|
||||
size_t DataBase = sizeof(FnCount) + sizeof(DataBase) + FnCount * sizeof(AOTIRInlineIndexEntry);
|
||||
stream->write((char*)&DataBase, sizeof(DataBase));
|
||||
|
||||
size_t DataOffset = 0;
|
||||
for (auto entry: AOTModule.second) {
|
||||
for (const auto& [GuestStart, DataOffset] : Entry.Index) {
|
||||
//AOTIRInlineIndexEntry
|
||||
|
||||
// GuestStart
|
||||
stream->write((char*)&entry.first, sizeof(entry.first));
|
||||
stream->write((const char*)&GuestStart, sizeof(GuestStart));
|
||||
|
||||
// DataOffset
|
||||
stream->write((char*)&DataOffset, sizeof(DataOffset));
|
||||
|
||||
|
||||
DataOffset += sizeof(entry.second.crc);
|
||||
DataOffset += sizeof(entry.second.len);
|
||||
|
||||
DataOffset += entry.second.RAData->Size(entry.second.RAData->MapCount);
|
||||
|
||||
DataOffset += entry.second.IR->GetInlineSize();
|
||||
stream->write((const char*)&DataOffset, sizeof(DataOffset));
|
||||
}
|
||||
|
||||
// AOTIRInlineEntry
|
||||
for (auto entry: AOTModule.second) {
|
||||
//GuestHash
|
||||
stream->write((char*)&entry.second.crc, sizeof(entry.second.crc));
|
||||
|
||||
//GuestLength
|
||||
stream->write((char*)&entry.second.len, sizeof(entry.second.len));
|
||||
|
||||
// RAData (inline)
|
||||
stream->write((char*)entry.second.RAData, entry.second.RAData->Size(entry.second.RAData->MapCount));
|
||||
|
||||
// IRData (inline)
|
||||
entry.second.IR->Serialize(*stream);
|
||||
}
|
||||
// End of file header
|
||||
const auto IndexSize = FnCount * sizeof(AOTIRInlineIndexEntry) + sizeof(DataBase) + sizeof(FnCount);
|
||||
stream->write((const char*)&IndexSize, sizeof(IndexSize));
|
||||
stream->write(String.c_str(), ModSize);
|
||||
stream->write((const char*)&ModSize, sizeof(ModSize));
|
||||
}
|
||||
|
||||
return rv;
|
||||
}
|
||||
|
||||
void Context::CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
@@ -1061,7 +1087,7 @@ namespace FEXCore::Context {
|
||||
abort();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
@@ -1135,26 +1161,49 @@ namespace FEXCore::Context {
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), decltype(Entry.RAData)(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
Thread->LocalIRCache.insert({GuestRIP, std::move(Entry)});
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if ((Config.AOTIRCapture() || Config.AOTIRGenerate()) && RAData) {
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
|
||||
RAData->IsShared = true;
|
||||
IRList->SetShared(true);
|
||||
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
|
||||
auto file = AddrToFile.lower_bound(StartAddr);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (file->second.Start <= StartAddr && (file->second.Start + file->second.Len) >= (StartAddr + Length)) {
|
||||
AOTIRCaptureCache[file->second.fileid].insert({GuestRIP - file->second.Start + file->second.Offset, {StartAddr - file->second.Start + file->second.Offset, Length, hash, IRList, RAData}});
|
||||
auto LocalRIP = GuestRIP - file->second.Start + file->second.Offset;
|
||||
auto LocalStartAddr = StartAddr - file->second.Start + file->second.Offset;
|
||||
auto fileid = file->second.fileid;
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRList, RAData, fileid]() {
|
||||
auto *AotFile = &AOTIRCaptureCache[fileid];
|
||||
|
||||
if (!AotFile->Stream) {
|
||||
AotFile->Stream = AOTIRWriter(fileid);
|
||||
uint64_t tag = 0xDEADBEEFC0D30004;
|
||||
AotFile->Stream->write((char*)&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRList, RAData);
|
||||
delete IRList;
|
||||
FEXCore::Allocator::free(RAData);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (Config.AOTIRGenerate()) {
|
||||
// cleanup memory and early exit here -- we're not running the application
|
||||
|
||||
if (DecrementRefCount)
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
|
||||
Thread->CPUBackend->ClearCache();
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
}
|
||||
|
||||
// Add to thread local ir cache
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), decltype(Entry.RAData)(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
Thread->LocalIRCache.insert({GuestRIP, std::move(Entry)});
|
||||
}
|
||||
|
||||
if (DecrementRefCount)
|
||||
@@ -1178,8 +1227,6 @@ namespace FEXCore::Context {
|
||||
|
||||
++IdleWaitRefCount;
|
||||
|
||||
LogMan::Msg::D("[%d] Waiting to run", Thread->ThreadManager.TID.load());
|
||||
|
||||
// Now notify the thread that we are initialized
|
||||
Thread->ThreadWaiting.NotifyAll();
|
||||
|
||||
@@ -1188,8 +1235,6 @@ namespace FEXCore::Context {
|
||||
Thread->StartRunning.Wait();
|
||||
}
|
||||
|
||||
LogMan::Msg::D("[%d] Running", Thread->ThreadManager.TID.load());
|
||||
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_NONE;
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
@@ -1303,7 +1348,7 @@ namespace FEXCore::Context {
|
||||
fileid += Config.ABILocalFlags ? "L" : "l";
|
||||
fileid += Config.ABINoPF ? "p" : "P";
|
||||
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
AddrToFile.insert({ Base, { Base, Size, Offset, fileid, filename, nullptr, false} });
|
||||
|
||||
@@ -1318,13 +1363,13 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::RemoveNamedRegion(uintptr_t Base, uintptr_t Size) {
|
||||
std::lock_guard<std::mutex> lk(AOTIRCacheLock);
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
// TODO: Support partial removing
|
||||
AddrToFile.erase(Base);
|
||||
}
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Context::Context *CTX, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
CTX->ParentThread->FrontendDecoder->SetExternalBranches(ExternalBranches);
|
||||
CTX->ParentThread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
|
||||
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
|
||||
}
|
||||
}
|
||||
@@ -5,6 +5,7 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
@@ -103,7 +104,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
ldr(x0, &l_PagePtr);
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
if (__builtin_popcountl(VirtualMemorySize) == 1) {
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(x3, RipReg, Thread->LookupCache->GetVirtualMemorySize() - 1);
|
||||
}
|
||||
else {
|
||||
|
||||
@@ -175,17 +175,10 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
|
||||
// siginfo_t
|
||||
siginfo_t *HostSigInfo = reinterpret_cast<siginfo_t*>(info);
|
||||
guest_siginfo->si_signo = Signal;
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
case SIGBUS:
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
guest_siginfo->si_errno = HostSigInfo->si_errno;
|
||||
// Macro expansion to get the si_addr
|
||||
guest_siginfo->si_addr = HostSigInfo->si_addr;
|
||||
break;
|
||||
default: LogMan::Msg::D("Unhandled siginfo_t signal: %d", Signal); break;
|
||||
}
|
||||
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
|
||||
// SI_USER could also potentially have random data in it, needs to be bit perfect
|
||||
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
Frame->State.gregs[X86State::REG_RSI] = SigInfoLocation;
|
||||
Frame->State.gregs[X86State::REG_RDX] = UContextLocation;
|
||||
@@ -195,7 +188,36 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
NewGuestSP -= sizeof(FEXCore::x86::ucontext_t);
|
||||
uint64_t UContextLocation = 0; // NewGuestSP;
|
||||
NewGuestSP -= sizeof(FEXCore::x86::siginfo_t);
|
||||
uint64_t SigInfoLocation = 0; // NewGuestSP;
|
||||
uint64_t SigInfoLocation = NewGuestSP;
|
||||
|
||||
FEXCore::x86::siginfo_t *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(SigInfoLocation);
|
||||
siginfo_t *HostSigInfo = reinterpret_cast<siginfo_t*>(info);
|
||||
|
||||
// These three elements are in every siginfo
|
||||
guest_siginfo->si_signo = HostSigInfo->si_signo;
|
||||
guest_siginfo->si_errno = HostSigInfo->si_errno;
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
case SIGBUS:
|
||||
// Macro expansion to get the si_addr
|
||||
// Can't really give a real result here. Pull from the context for now
|
||||
guest_siginfo->_sifields._sigfault.addr = Frame->State.rip;
|
||||
break;
|
||||
case SIGCHLD:
|
||||
guest_siginfo->_sifields._sigchld.pid = HostSigInfo->si_pid;
|
||||
guest_siginfo->_sifields._sigchld.uid = HostSigInfo->si_uid;
|
||||
guest_siginfo->_sifields._sigchld.status = HostSigInfo->si_status;
|
||||
guest_siginfo->_sifields._sigchld.utime = HostSigInfo->si_utime;
|
||||
guest_siginfo->_sifields._sigchld.stime = HostSigInfo->si_stime;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::D("Unhandled siginfo_t signal: %d", Signal);
|
||||
// Hope for the best, most things just copy over
|
||||
memcpy(guest_siginfo, info, sizeof(siginfo_t));
|
||||
break;
|
||||
}
|
||||
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = UContextLocation;
|
||||
@@ -254,7 +276,7 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = ThreadState->SignalReason.load();
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_PAUSE) {
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
|
||||
// Store our thread state so we can come back to this
|
||||
StoreThreadState(Signal, ucontext);
|
||||
|
||||
@@ -279,11 +301,11 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
// We use this to track if it is safe to clear cache
|
||||
++SignalHandlerRefCounter;
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_STOP) {
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Stop) {
|
||||
// Our thread is stopping
|
||||
// We don't care about anything at this point
|
||||
// Set the stack to our starting location when we entered the core and get out safely
|
||||
@@ -304,18 +326,18 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
|
||||
}
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::SIGNALEVENT_RETURN) {
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Return) {
|
||||
RestoreThreadState(ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SIGNALEVENT_NONE);
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -344,7 +366,7 @@ void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher) {
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher) const {
|
||||
for (auto [start, end] : CodeBuffers) {
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
|
||||
@@ -54,8 +54,8 @@ public:
|
||||
|
||||
void RemoveCodeBuffer(uint8_t* start);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true);
|
||||
bool IsAddressInDispatcher(uint64_t Address) {
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const;
|
||||
bool IsAddressInDispatcher(uint64_t Address) const {
|
||||
return Address >= Start && Address < End;
|
||||
}
|
||||
|
||||
|
||||
@@ -266,7 +266,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
{
|
||||
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
|
||||
// using CallbackReturn = __attribute__((naked)) void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
|
||||
// rdi = thread
|
||||
// rsi = rsp
|
||||
@@ -293,7 +293,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
#if ENABLE_JITSYMBOLS
|
||||
std::string Name = "Dispatch_" + std::to_string(::gettid());
|
||||
CTX->Symbols.Register(Start, End-Start, Name);
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(Start), End-Start, Name);
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
+58
-57
@@ -21,7 +21,9 @@ namespace FEXCore::Frontend {
|
||||
using namespace FEXCore::X86Tables;
|
||||
|
||||
static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool HasREX, bool HasXMM, bool HasMM, uint8_t InvalidOffset = 16) {
|
||||
constexpr std::array<uint64_t, 16> GPRIndexes = {
|
||||
using GPRArray = std::array<uint32_t, 16>;
|
||||
|
||||
static constexpr GPRArray GPRIndexes = {
|
||||
// Classical ordering?
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RCX,
|
||||
@@ -41,7 +43,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
FEXCore::X86State::REG_R15,
|
||||
};
|
||||
|
||||
constexpr std::array<uint64_t, 16> GPR8BitHighIndexes = {
|
||||
static constexpr GPRArray GPR8BitHighIndexes = {
|
||||
// Classical ordering?
|
||||
FEXCore::X86State::REG_RAX,
|
||||
FEXCore::X86State::REG_RCX,
|
||||
@@ -61,7 +63,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
FEXCore::X86State::REG_R15,
|
||||
};
|
||||
|
||||
constexpr std::array<uint64_t, 16> XMMIndexes = {
|
||||
static constexpr GPRArray XMMIndexes = {
|
||||
FEXCore::X86State::REG_XMM_0,
|
||||
FEXCore::X86State::REG_XMM_1,
|
||||
FEXCore::X86State::REG_XMM_2,
|
||||
@@ -80,7 +82,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
FEXCore::X86State::REG_XMM_15,
|
||||
};
|
||||
|
||||
constexpr std::array<uint64_t, 16> MMIndexes = {
|
||||
static constexpr GPRArray MMIndexes = {
|
||||
FEXCore::X86State::REG_MM_0,
|
||||
FEXCore::X86State::REG_MM_1,
|
||||
FEXCore::X86State::REG_MM_2,
|
||||
@@ -99,7 +101,7 @@ static uint32_t MapModRMToReg(uint8_t REX, uint8_t bits, bool HighBits, bool Has
|
||||
FEXCore::X86State::REG_INVALID
|
||||
};
|
||||
|
||||
const std::array<uint64_t, 16> *GPRs = &GPRIndexes;
|
||||
const GPRArray *GPRs = &GPRIndexes;
|
||||
if (HasXMM) {
|
||||
GPRs = &XMMIndexes;
|
||||
}
|
||||
@@ -131,7 +133,7 @@ uint8_t Decoder::ReadByte() {
|
||||
return Byte;
|
||||
}
|
||||
|
||||
uint8_t Decoder::PeekByte(uint8_t Offset) {
|
||||
uint8_t Decoder::PeekByte(uint8_t Offset) const {
|
||||
uint8_t Byte = InstStream[InstructionSize + Offset];
|
||||
return Byte;
|
||||
}
|
||||
@@ -197,9 +199,9 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
}
|
||||
}
|
||||
|
||||
Operand->TypeSIB.Type = DecodedOperand::TYPE_SIB;
|
||||
Operand->TypeSIB.Scale = 1;
|
||||
Operand->TypeSIB.Offset = Literal;
|
||||
Operand->Type = DecodedOperand::OpType::SIB;
|
||||
Operand->Data.SIB.Scale = 1;
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
|
||||
// Only called when ModRM.mod != 0b11
|
||||
struct Encodings {
|
||||
@@ -238,8 +240,8 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
|
||||
uint8_t LookupIndex = ModRM.mod << 3 | ModRM.rm;
|
||||
auto it = Lookup[LookupIndex];
|
||||
Operand->TypeSIB.Base = it.Base;
|
||||
Operand->TypeSIB.Index = it.Index;
|
||||
Operand->Data.SIB.Base = it.Base;
|
||||
Operand->Data.SIB.Index = it.Index;
|
||||
}
|
||||
|
||||
void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModRMDecoded ModRM) {
|
||||
@@ -277,12 +279,12 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
}
|
||||
|
||||
// SIB
|
||||
Operand->TypeSIB.Type = DecodedOperand::TYPE_SIB;
|
||||
Operand->TypeSIB.Scale = 1 << SIB.scale;
|
||||
Operand->Type = DecodedOperand::OpType::SIB;
|
||||
Operand->Data.SIB.Scale = 1 << SIB.scale;
|
||||
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
Operand->TypeSIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->TypeSIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
|
||||
uint64_t Literal {0};
|
||||
LOGMAN_THROW_A(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
@@ -291,7 +293,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
Operand->TypeSIB.Offset = Literal;
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
}
|
||||
else if (ModRM.mod == 0) {
|
||||
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
|
||||
@@ -300,13 +302,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
uint32_t Literal;
|
||||
Literal = ReadData(4);
|
||||
|
||||
Operand->TypeRIPLiteral.Type = DecodedOperand::TYPE_RIP_RELATIVE;
|
||||
Operand->TypeRIPLiteral.Literal.u = Literal;
|
||||
Operand->Type = DecodedOperand::OpType::RIPRelative;
|
||||
Operand->Data.RIPLiteral.Value.u = Literal;
|
||||
}
|
||||
else {
|
||||
// Register-direct addressing
|
||||
Operand->TypeGPR.Type = DecodedOperand::TYPE_GPR_DIRECT;
|
||||
Operand->TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
Operand->Type = DecodedOperand::OpType::GPRDirect;
|
||||
Operand->Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -318,9 +320,9 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
}
|
||||
Displacement = DisplacementSize;
|
||||
|
||||
Operand->TypeGPRIndirect.Type = DecodedOperand::TYPE_GPR_INDIRECT;
|
||||
Operand->TypeGPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
Operand->TypeGPRIndirect.Displacement = Literal;
|
||||
Operand->Type = DecodedOperand::OpType::GPRIndirect;
|
||||
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
Operand->Data.GPRIndirect.Displacement = Literal;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -460,9 +462,9 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ||
|
||||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RDX)) {
|
||||
// Some instructions hardcode their destination as RAX
|
||||
CurrentDest->TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
CurrentDest->TypeGPR.HighBits = false;
|
||||
CurrentDest->TypeGPR.GPR = HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR = HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
|
||||
CurrentDest = &DecodeInst->Src[0];
|
||||
}
|
||||
|
||||
@@ -473,11 +475,11 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
// This also means that the destination is always a GPR on these ones
|
||||
// ADDITIONALLY:
|
||||
// If there is a REX prefix then that allows extended GPR usage
|
||||
CurrentDest->TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
DecodeInst->Dest.TypeGPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100) || HasHighXMM;
|
||||
CurrentDest->TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Dest.Data.GPR.HighBits = (Is8BitDest && !HasREX && (Op & 0b111) >= 0b100) || HasHighXMM;
|
||||
CurrentDest->Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
|
||||
|
||||
if (CurrentDest->TypeGPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
if (CurrentDest->Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -501,20 +503,20 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
|
||||
// Decode the GPR source first
|
||||
GPR.TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
GPR.TypeGPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX) || HasHighXMM;
|
||||
GPR.TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_R ? 1 : 0, ModRM.reg, GPR8Bit, HasREX, HasXMMGPR, HasMMGPR);
|
||||
GPR.Type = DecodedOperand::OpType::GPR;
|
||||
GPR.Data.GPR.HighBits = (GPR8Bit && ModRM.reg >= 0b100 && !HasREX) || HasHighXMM;
|
||||
GPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_R ? 1 : 0, ModRM.reg, GPR8Bit, HasREX, HasXMMGPR, HasMMGPR);
|
||||
|
||||
if (GPR.TypeGPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
if (GPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
return false;
|
||||
|
||||
// ModRM.mod == 0b11 == Register
|
||||
// ModRM.Mod != 0b11 == Register-direct addressing
|
||||
if (ModRM.mod == 0b11) {
|
||||
NonGPR.TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
NonGPR.TypeGPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX) || HasHighXMM;
|
||||
NonGPR.TypeGPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, NonGPR8Bit, HasREX, HasXMMNonGPR, HasMMNonGPR);
|
||||
if (NonGPR.TypeGPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
NonGPR.Type = DecodedOperand::OpType::GPR;
|
||||
NonGPR.Data.GPR.HighBits = (NonGPR8Bit && ModRM.rm >= 0b100 && !HasREX) || HasHighXMM;
|
||||
NonGPR.Data.GPR.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, NonGPR8Bit, HasREX, HasXMMNonGPR, HasMMNonGPR);
|
||||
if (NonGPR.Data.GPR.GPR == FEXCore::X86State::REG_INVALID)
|
||||
return false;
|
||||
}
|
||||
else {
|
||||
@@ -540,25 +542,24 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RAX)) {
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.GPR = FEXCore::X86State::REG_RAX;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RAX;
|
||||
++CurrentSrc;
|
||||
}
|
||||
else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RCX)) {
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.Type = DecodedOperand::TYPE_GPR;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].TypeGPR.GPR = FEXCore::X86State::REG_RCX;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RCX;
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_A(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
DecodeInst->Src[CurrentSrc].TypeLiteral.Size = Bytes;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
|
||||
uint64_t Literal {0};
|
||||
Literal = ReadData(Bytes);
|
||||
uint64_t Literal = ReadData(Bytes);
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT) ||
|
||||
(DecodeFlags::GetSizeDstFlags(DecodeInst->Flags) == DecodeFlags::SIZE_64BIT && Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SRC_SEXT64BIT)) {
|
||||
@@ -571,12 +572,12 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op)
|
||||
else {
|
||||
Literal = static_cast<int32_t>(Literal);
|
||||
}
|
||||
DecodeInst->Src[CurrentSrc].TypeLiteral.Size = DestSize;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
|
||||
}
|
||||
|
||||
Bytes = 0;
|
||||
DecodeInst->Src[CurrentSrc].TypeLiteral.Type = DecodedOperand::TYPE_LITERAL;
|
||||
DecodeInst->Src[CurrentSrc].TypeLiteral.Literal = Literal;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A(Bytes == 0, "Inst at 0x%lx: 0x%04x '%s' Had an instruction of size %d with %d remaining", DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name, InstructionSize, Bytes);
|
||||
@@ -927,8 +928,8 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.TypeNone.Type == FEXCore::X86Tables::DecodedOperand::TYPE_GPR) {
|
||||
assert(DecodeInst->Dest.TypeGPR.GPR != 255);
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
assert(DecodeInst->Dest.Data.GPR.GPR != 255);
|
||||
}
|
||||
|
||||
return true;
|
||||
@@ -940,7 +941,7 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
|
||||
// If the RIP setting is conditional AND within our symbol range then it can be considered for multiblock
|
||||
uint64_t TargetRIP = 0;
|
||||
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
bool Conditional = true;
|
||||
|
||||
switch (DecodeInst->OP) {
|
||||
@@ -950,14 +951,14 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// auto RIPOffset = LoadSource(Op, Op->Src[0], Op->Flags);
|
||||
// auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
|
||||
// Target offset is PC + InstSize + Literal
|
||||
LOGMAN_THROW_A(DecodeInst->Src[0].TypeNone.Type == DecodedOperand::TYPE_LITERAL, "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].TypeLiteral.Literal;
|
||||
LOGMAN_THROW_A(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
break;
|
||||
}
|
||||
case 0xE9:
|
||||
case 0xEB: // Both are unconditional JMP instructions
|
||||
LOGMAN_THROW_A(DecodeInst->Src[0].TypeNone.Type == DecodedOperand::TYPE_LITERAL, "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].TypeLiteral.Literal;
|
||||
LOGMAN_THROW_A(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
Conditional = false;
|
||||
break;
|
||||
case 0xE8: // Call - Immediate target, We don't want to inline calls
|
||||
|
||||
+2
-2
@@ -26,7 +26,7 @@ public:
|
||||
Decoder(FEXCore::Context::Context *ctx);
|
||||
bool DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
|
||||
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() {
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
return &Blocks;
|
||||
}
|
||||
|
||||
@@ -43,7 +43,7 @@ private:
|
||||
void BranchTargetInMultiblockRange();
|
||||
|
||||
uint8_t ReadByte();
|
||||
uint8_t PeekByte(uint8_t Offset);
|
||||
uint8_t PeekByte(uint8_t Offset) const;
|
||||
uint64_t ReadData(uint8_t Size);
|
||||
void SkipBytes(uint8_t Size) { InstructionSize += Size; }
|
||||
bool NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op);
|
||||
|
||||
+77
-93
@@ -15,15 +15,19 @@ $end_info$
|
||||
#include <optional>
|
||||
#include "Common/NetStream.h"
|
||||
#include "Common/SoftFloat.h"
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <sys/types.h>
|
||||
#include <sys/socket.h>
|
||||
#include <netdb.h>
|
||||
#include <string.h>
|
||||
#include <cstring>
|
||||
#include <fcntl.h>
|
||||
#include <unistd.h>
|
||||
#include <fmt/format.h>
|
||||
#include <fstream>
|
||||
#include <netdb.h>
|
||||
#include <sys/socket.h>
|
||||
#include <sys/types.h>
|
||||
#include <unistd.h>
|
||||
|
||||
|
||||
#include "GdbServer.h"
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
@@ -34,20 +38,20 @@ namespace FEXCore
|
||||
|
||||
void GdbServer::Break(int signal) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
if (!CommsStream) {
|
||||
return;
|
||||
}
|
||||
|
||||
std::ostringstream ss;
|
||||
ss << "S" << std::setfill('0') << std::setw(2) << std::hex << signal;
|
||||
|
||||
if (CommsStream)
|
||||
SendPacket(*CommsStream, ss.str());
|
||||
const auto str = fmt::format("S{:02x}", signal);
|
||||
SendPacket(*CommsStream, str);
|
||||
}
|
||||
|
||||
GdbServer::GdbServer(FEXCore::Context::Context *ctx) : CTX(ctx) {
|
||||
ctx->CustomExitHandler = [this](uint64_t ThreadId, FEXCore::Context::ExitReason ExitReason) {
|
||||
Context::SetExitHandler(ctx, [this](uint64_t ThreadId, FEXCore::Context::ExitReason ExitReason) {
|
||||
if (ExitReason == FEXCore::Context::ExitReason::EXIT_DEBUG) {
|
||||
this->Break(SIGTRAP);
|
||||
}
|
||||
};
|
||||
});
|
||||
|
||||
// This is a total hack as there is currently no way to resume once hitting a segfault
|
||||
// But it's semi-useful for debugging.
|
||||
@@ -60,12 +64,12 @@ GdbServer::GdbServer(FEXCore::Context::Context *ctx) : CTX(ctx) {
|
||||
usleep(100000);
|
||||
|
||||
return true;
|
||||
});
|
||||
}, true);
|
||||
|
||||
StartThread();
|
||||
}
|
||||
|
||||
static int calculateChecksum(std::string &packet) {
|
||||
static int calculateChecksum(const std::string &packet) {
|
||||
unsigned char checksum = 0;
|
||||
for (const char &c : packet) {
|
||||
checksum += c;
|
||||
@@ -99,11 +103,9 @@ static std::string encodeHex(unsigned char *data, size_t length) {
|
||||
}
|
||||
|
||||
static std::string getThreadName(uint32_t ThreadID) {
|
||||
std::fstream fs;
|
||||
std::ostringstream ThreadFile;
|
||||
ThreadFile << "/proc/" << getpid() << "/task/" << ThreadID << "/comm";
|
||||
const auto ThreadFile = fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
|
||||
std::fstream fs(ThreadFile, std::fstream::in | std::fstream::binary);
|
||||
|
||||
fs.open(ThreadFile.str(), std::fstream::in | std::fstream::binary);
|
||||
if (fs.is_open()) {
|
||||
std::string ThreadName;
|
||||
fs >> ThreadName;
|
||||
@@ -135,7 +137,7 @@ std::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
switch(c) {
|
||||
case '$': // start of packet
|
||||
if (packet.size() != 0)
|
||||
LogMan::Msg::E("Dropping unexpected data: \"%s\"", packet.c_str());
|
||||
LogMan::Msg::EFmt("Dropping unexpected data: \"{}\"", packet);
|
||||
|
||||
// clear any existing data, must have been a mistake.
|
||||
packet = std::string();
|
||||
@@ -156,7 +158,7 @@ std::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
if (calculateChecksum(packet) == expected_checksum) {
|
||||
return packet;
|
||||
} else {
|
||||
LogMan::Msg::E("Received Invalid Packet: $%s#%02x %c%c", packet.c_str(), expected_checksum);
|
||||
LogMan::Msg::EFmt("Received Invalid Packet: ${}#{:02x}", packet, expected_checksum);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -169,10 +171,10 @@ std::string GdbServer::ReadPacket(std::iostream &stream) {
|
||||
return "";
|
||||
}
|
||||
|
||||
static std::string escapePacket(std::string packet) {
|
||||
static std::string escapePacket(const std::string& packet) {
|
||||
std::ostringstream ss;
|
||||
|
||||
for(auto &c : packet) {
|
||||
for(const auto &c : packet) {
|
||||
switch (c) {
|
||||
case '$':
|
||||
case '#':
|
||||
@@ -191,13 +193,11 @@ static std::string escapePacket(std::string packet) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
void GdbServer::SendPacket(std::ostream &stream, std::string packet) {
|
||||
auto escaped = escapePacket(packet);
|
||||
std::ostringstream ss;
|
||||
void GdbServer::SendPacket(std::ostream &stream, const std::string& packet) {
|
||||
const auto escaped = escapePacket(packet);
|
||||
const auto str = fmt::format("${}#{:02x}", escaped, calculateChecksum(escaped));
|
||||
|
||||
ss << '$' << escaped << '#';
|
||||
ss << std::setfill('0') << std::setw(2) << std::hex << (int)calculateChecksum(escaped);
|
||||
stream << ss.str() << std::flush;
|
||||
stream << str << std::flush;
|
||||
}
|
||||
|
||||
void GdbServer::SendACK(std::ostream &stream, bool NACK) {
|
||||
@@ -218,7 +218,7 @@ void GdbServer::SendACK(std::ostream &stream, bool NACK) {
|
||||
}
|
||||
}
|
||||
|
||||
struct __attribute__((packed)) GDBContextDefinition {
|
||||
struct FEX_PACKED GDBContextDefinition {
|
||||
uint64_t gregs[16];
|
||||
uint64_t rip;
|
||||
uint32_t eflags;
|
||||
@@ -279,7 +279,7 @@ std::string GdbServer::readRegs() {
|
||||
return encodeHex((unsigned char *)&GDB, sizeof(GDBContextDefinition));
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::readReg(std::string& packet) {
|
||||
GdbServer::HandledPacketType GdbServer::readReg(const std::string& packet) {
|
||||
size_t addr;
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.get(); // Drop first letter
|
||||
@@ -357,7 +357,7 @@ GdbServer::HandledPacketType GdbServer::readReg(std::string& packet) {
|
||||
return {encodeHex((unsigned char *)(&Empty), sizeof(uint32_t)), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
LogMan::Msg::E("Unknown GDB register 0x%lx", addr);
|
||||
LogMan::Msg::EFmt("Unknown GDB register 0x{:x}", addr);
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
@@ -462,7 +462,7 @@ std::string buildTargetXML() {
|
||||
return xml.str();
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleXfer(std::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
std::string object;
|
||||
std::string rw;
|
||||
std::string annex;
|
||||
@@ -548,10 +548,9 @@ GdbServer::HandledPacketType GdbServer::handleXfer(std::string &packet) {
|
||||
|
||||
static size_t CheckMemMapping(uint64_t Address, size_t Size) {
|
||||
uint64_t AddressEnd = Address + Size;
|
||||
|
||||
std::fstream fs;
|
||||
fs.open("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::string Line;
|
||||
|
||||
while (std::getline(fs, Line)) {
|
||||
if (fs.eof()) break;
|
||||
uint64_t Begin, End;
|
||||
@@ -568,32 +567,29 @@ static size_t CheckMemMapping(uint64_t Address, size_t Size) {
|
||||
}
|
||||
}
|
||||
|
||||
fs.close();
|
||||
return 0;
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleProgramOffsets() {
|
||||
std::fstream fs;
|
||||
fs.open("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::string Line;
|
||||
std::string const &RuntimeExecutable = Filename();
|
||||
|
||||
while (std::getline(fs, Line)) {
|
||||
uint64_t Begin, End;
|
||||
char Filename[255];
|
||||
if (sscanf(Line.c_str(), "%lx-%lx %*c%*c%*c%*c %*x %*x:%*x %*d%s", &Begin, &End, Filename) == 3) {
|
||||
if (RuntimeExecutable == Filename) {
|
||||
std::ostringstream ss;
|
||||
ss << "Text=" << std::hex << Begin << ";Data=" << std::hex << Begin << ";Bss=" << std::hex << Begin;
|
||||
ss << std::flush;
|
||||
return {ss.str(), HandledPacketType::TYPE_ACK};
|
||||
auto str = fmt::format("Text={:x};Data={:x};Bss={:x}", Begin, Begin, Begin);
|
||||
return {std::move(str), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
}
|
||||
}
|
||||
fs.close();
|
||||
|
||||
return {"Text=0;Data=0;Bss=0", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleMemory(std::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::handleMemory(const std::string &packet) {
|
||||
bool write;
|
||||
size_t addr;
|
||||
size_t length;
|
||||
@@ -634,8 +630,8 @@ GdbServer::HandledPacketType GdbServer::handleMemory(std::string &packet) {
|
||||
}
|
||||
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleQuery(std::string &packet) {
|
||||
auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
GdbServer::HandledPacketType GdbServer::handleQuery(const std::string &packet) {
|
||||
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
|
||||
if (match("qSupported")) {
|
||||
return {"PacketSize=5000;xmlRegisters=i386;qXfer:exec-file:read+;qXfer:features:read+;", HandledPacketType::TYPE_ACK};
|
||||
@@ -693,8 +689,8 @@ GdbServer::HandledPacketType GdbServer::handleQuery(std::string &packet) {
|
||||
return {"", HandledPacketType::TYPE_UNKNOWN};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleV(std::string& packet) {
|
||||
auto match = [&](std::string str) -> std::optional<std::istringstream> {
|
||||
GdbServer::HandledPacketType GdbServer::handleV(const std::string& packet) {
|
||||
const auto match = [&](const std::string& str) -> std::optional<std::istringstream> {
|
||||
if (packet.rfind(str, 0) == 0) {
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(str.size());
|
||||
@@ -703,18 +699,11 @@ GdbServer::HandledPacketType GdbServer::handleV(std::string& packet) {
|
||||
return std::nullopt;
|
||||
};
|
||||
|
||||
auto F = [](int result) {
|
||||
std::ostringstream ss;
|
||||
ss << "F" << std::hex << result;
|
||||
return ss.str(); };
|
||||
auto F_error = [&]() {
|
||||
std::ostringstream ss;
|
||||
ss << "F-1," << std::hex << errno;
|
||||
return ss.str(); };
|
||||
auto F_data = [&](int result, std::string data) {
|
||||
std::ostringstream ss;
|
||||
ss << "F" << std::hex << result << ";" << data;
|
||||
return ss.str(); };
|
||||
const auto F = [](int result) { return fmt::format("F{:x}", result); };
|
||||
const auto F_error = [] { return fmt::format("F-1,{:x}", errno); };
|
||||
const auto F_data = [](int result, const std::string& data) {
|
||||
return fmt::format("F{:x};{}", result, data);
|
||||
};
|
||||
|
||||
std::optional<std::istringstream> ss;
|
||||
if((ss = match("vFile:open:"))) {
|
||||
@@ -736,11 +725,11 @@ GdbServer::HandledPacketType GdbServer::handleV(std::string& packet) {
|
||||
return {F(pid == 0 ? 0 : -1), HandledPacketType::TYPE_ACK}; // Only support the common filesystem
|
||||
}
|
||||
if((ss = match("vFile:close:"))) {
|
||||
int fd;
|
||||
*ss >> std::hex >> fd;
|
||||
close(fd);
|
||||
return {F(0), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
int fd;
|
||||
*ss >> std::hex >> fd;
|
||||
close(fd);
|
||||
return {F(0), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if((ss = match("vFile:pread:"))) {
|
||||
int fd, count, offset;
|
||||
|
||||
@@ -777,7 +766,7 @@ GdbServer::HandledPacketType GdbServer::handleV(std::string& packet) {
|
||||
}
|
||||
|
||||
if (ss->fail()) {
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
switch (action) {
|
||||
@@ -787,27 +776,25 @@ GdbServer::HandledPacketType GdbServer::handleV(std::string& packet) {
|
||||
}
|
||||
case 's': {
|
||||
CTX->Step();
|
||||
SendPacketPair({"OK", HandledPacketType::TYPE_ACK});
|
||||
std::ostringstream ss;
|
||||
ss << "T05thread:" << std::setfill('0') << std::setw(2) << std::hex << getpid() << ";core:2c;";
|
||||
|
||||
SendPacketPair({ss.str(), HandledPacketType::TYPE_ACK});
|
||||
SendPacketPair({"OK", HandledPacketType::TYPE_ACK});
|
||||
auto str = fmt::format("T05thread:{:02x};core:2c;", getpid());
|
||||
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
case 't':
|
||||
// This thread isn't part of the thread pool
|
||||
CTX->Stop(false /* Ignore current thread */);
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
default:
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
}
|
||||
return {"", HandledPacketType::TYPE_ACK};
|
||||
return {"", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleThreadOp(std::string &packet) {
|
||||
auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet) {
|
||||
const auto match = [&](const char *str) -> bool { return packet.rfind(str, 0) == 0; };
|
||||
|
||||
if (match("Hc")) {
|
||||
// Sets thread to this ID for stepping
|
||||
@@ -823,7 +810,7 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(std::string &packet) {
|
||||
if (match("Hg")) {
|
||||
// Sets thread for "other" operations
|
||||
auto ss = std::istringstream(packet);
|
||||
ss.seekg(std::string("Hg").size());
|
||||
ss.seekg(std::string_view("Hg").size());
|
||||
ss >> std::hex >> CurrentDebuggingThread;
|
||||
|
||||
// This must return quick otherwise IDA complains
|
||||
@@ -834,7 +821,7 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(std::string &packet) {
|
||||
return {"", HandledPacketType::TYPE_UNKNOWN};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleBreakpoint(std::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::handleBreakpoint(const std::string &packet) {
|
||||
auto ss = std::istringstream(packet);
|
||||
|
||||
bool Set{};
|
||||
@@ -850,17 +837,15 @@ GdbServer::HandledPacketType GdbServer::handleBreakpoint(std::string &packet) {
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::ProcessPacket(std::string &packet) {
|
||||
GdbServer::HandledPacketType GdbServer::ProcessPacket(const std::string &packet) {
|
||||
switch (packet[0]) {
|
||||
case '?': {
|
||||
// Indicates the reason that the thread has stopped
|
||||
// Behaviour changes if the target is in non-stop mode
|
||||
// Binja doesn't support S response here
|
||||
//return {"S00", HandledPacketType::TYPE_ACK};
|
||||
std::ostringstream ss;
|
||||
ss << "T00thread:" << std::setfill('0') << std::setw(2) << std::hex << getpid() << ";core:2c;";
|
||||
|
||||
return {ss.str(), HandledPacketType::TYPE_ACK};
|
||||
auto str = fmt::format("T00thread:{:02x};core:2c;", getpid());
|
||||
return {std::move(str), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
case 'g':
|
||||
return {readRegs(), HandledPacketType::TYPE_ACK};
|
||||
@@ -890,14 +875,14 @@ GdbServer::HandledPacketType GdbServer::ProcessPacket(std::string &packet) {
|
||||
}
|
||||
}
|
||||
|
||||
void GdbServer::SendPacketPair(HandledPacketType response) {
|
||||
void GdbServer::SendPacketPair(const HandledPacketType& response) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_ACK ||
|
||||
response.TypeResponse == HandledPacketType::TYPE_ONLYACK) {
|
||||
SendACK(*CommsStream, false);
|
||||
}
|
||||
else if (response.TypeResponse == HandledPacketType::TYPE_NACK ||
|
||||
response.TypeResponse == HandledPacketType::TYPE_ONLYNACK) {
|
||||
response.TypeResponse == HandledPacketType::TYPE_ONLYNACK) {
|
||||
SendACK(*CommsStream, true);
|
||||
}
|
||||
|
||||
@@ -905,8 +890,8 @@ void GdbServer::SendPacketPair(HandledPacketType response) {
|
||||
SendPacket(*CommsStream, "");
|
||||
}
|
||||
else if (response.TypeResponse != HandledPacketType::TYPE_ONLYNACK &&
|
||||
response.TypeResponse != HandledPacketType::TYPE_ONLYACK &&
|
||||
response.TypeResponse != HandledPacketType::TYPE_NONE) {
|
||||
response.TypeResponse != HandledPacketType::TYPE_ONLYACK &&
|
||||
response.TypeResponse != HandledPacketType::TYPE_NONE) {
|
||||
SendPacket(*CommsStream, response.Response);
|
||||
}
|
||||
}
|
||||
@@ -927,7 +912,7 @@ void GdbServer::GdbServerLoop() {
|
||||
response = ProcessPacket(packet);
|
||||
SendPacketPair(response);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::D("Unknown packet %s", packet.c_str());
|
||||
LogMan::Msg::DFmt("Unknown packet {}", packet);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -943,13 +928,12 @@ void GdbServer::GdbServerLoop() {
|
||||
break;
|
||||
case '\x03': { // ASCII EOT
|
||||
CTX->Pause();
|
||||
std::ostringstream ss;
|
||||
ss << "T02thread:" << std::setfill('0') << std::setw(2) << std::hex << getpid() << ";core:2c;";
|
||||
SendPacketPair({ss.str(), HandledPacketType::TYPE_ACK});
|
||||
auto str = fmt::format("T02thread:{:02x};core:2c;", getpid());
|
||||
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LogMan::Msg::D("GdbServer: Unexpected byte %c (%02x)", c, c);
|
||||
LogMan::Msg::DFmt("GdbServer: Unexpected byte {} ({:02x})", static_cast<char>(c), c);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1003,7 +987,7 @@ std::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
|
||||
// Block until a connection arrives
|
||||
|
||||
LogMan::Msg::I("GdbServer, waiting for connection on localhost:8086");
|
||||
LogMan::Msg::IFmt("GdbServer, waiting for connection on localhost:8086");
|
||||
listen(sockfd, 1);
|
||||
|
||||
new_fd = accept(sockfd, (struct sockaddr *)&their_addr, &addr_size);
|
||||
|
||||
+10
-10
@@ -30,7 +30,7 @@ private:
|
||||
std::unique_ptr<std::iostream> OpenSocket();
|
||||
void StartThread();
|
||||
std::string ReadPacket(std::iostream &stream);
|
||||
void SendPacket(std::ostream &stream, std::string packet);
|
||||
void SendPacket(std::ostream &stream, const std::string& packet);
|
||||
|
||||
void SendACK(std::ostream &stream, bool NACK);
|
||||
|
||||
@@ -47,18 +47,18 @@ private:
|
||||
ResponseType TypeResponse{};
|
||||
};
|
||||
|
||||
void SendPacketPair(HandledPacketType packetPair);
|
||||
HandledPacketType ProcessPacket(std::string &packet);
|
||||
HandledPacketType handleQuery(std::string &packet);
|
||||
HandledPacketType handleXfer(std::string &packet);
|
||||
HandledPacketType handleMemory(std::string &packet);
|
||||
HandledPacketType handleV(std::string& packet);
|
||||
HandledPacketType handleThreadOp(std::string &packet);
|
||||
HandledPacketType handleBreakpoint(std::string &packet);
|
||||
void SendPacketPair(const HandledPacketType& packetPair);
|
||||
HandledPacketType ProcessPacket(const std::string &packet);
|
||||
HandledPacketType handleQuery(const std::string &packet);
|
||||
HandledPacketType handleXfer(const std::string &packet);
|
||||
HandledPacketType handleMemory(const std::string &packet);
|
||||
HandledPacketType handleV(const std::string& packet);
|
||||
HandledPacketType handleThreadOp(const std::string &packet);
|
||||
HandledPacketType handleBreakpoint(const std::string &packet);
|
||||
HandledPacketType handleProgramOffsets();
|
||||
|
||||
std::string readRegs();
|
||||
HandledPacketType readReg(std::string& packet);
|
||||
HandledPacketType readReg(const std::string& packet);
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
std::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
|
||||
@@ -93,12 +93,12 @@ InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
});
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->HandleSIGBUS(Signal, info, ucontext);
|
||||
});
|
||||
}, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
@@ -115,8 +115,8 @@ void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR:
|
||||
return reinterpret_cast<void*>(InterpreterExecution);
|
||||
}
|
||||
|
||||
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return new InterpreterCore(ctx, Thread, CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<InterpreterCore>(ctx, Thread, CompileThread);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,5 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
@@ -11,6 +13,6 @@ namespace FEXCore::Core {
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
FEXCore::CPU::CPUBackend *CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
|
||||
}
|
||||
+285
-166
@@ -10,15 +10,18 @@
|
||||
#include "Interface/Core/DebugData.h"
|
||||
#include "Interface/Core/InternalThreadState.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/BitUtils.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <atomic>
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
@@ -411,15 +414,15 @@ static void StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->StopThread(Thread);
|
||||
|
||||
LOGMAN_MSG_A("unreachable");
|
||||
__builtin_unreachable();
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[noreturn]]
|
||||
static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->SignalThread(Thread, FEXCore::Core::SIGNALEVENT_RETURN);
|
||||
Thread->CTX->SignalThread(Thread, FEXCore::Core::SignalEvent::Return);
|
||||
|
||||
LOGMAN_MSG_A("unreachable");
|
||||
__builtin_unreachable();
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
template<IR::IROps Op>
|
||||
@@ -1215,7 +1218,41 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_VDUPELEMENT: {
|
||||
auto Op = IROp->C<IR::IROp_VDupElement>();
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_A(OpSize <= 16, "OpSize is too large for VDupElement: %d", OpSize);
|
||||
if (OpSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(SSAData, Op->Header.Args[0]);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
for (size_t i = 0; i < Elements; ++i) {
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * i)),
|
||||
&Src, Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(SSAData, Op->Header.Args[0]);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
for (size_t i = 0; i < Elements; ++i) {
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * i)),
|
||||
&Src, Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_ENTRYPOINTOFFSET: {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
GD = Entry + Op->Offset;
|
||||
@@ -1573,10 +1610,10 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<int32_t>(Src1) << (Src2 & Mask);
|
||||
GD = static_cast<uint32_t>(Src1) << (Src2 & Mask);
|
||||
break;
|
||||
case 8:
|
||||
GD = static_cast<int64_t>(Src1) << (Src2 & Mask);
|
||||
GD = static_cast<uint64_t>(Src1) << (Src2 & Mask);
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unknown LSHL Size: %d\n", OpSize); break;
|
||||
};
|
||||
@@ -1866,23 +1903,23 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
case IR::OP_POPCOUNT: {
|
||||
auto Op = IROp->C<IR::IROp_Popcount>();
|
||||
uint64_t Src = *GetSrc<uint64_t*>(SSAData, Op->Header.Args[0]);
|
||||
GD = __builtin_popcountl(Src);
|
||||
GD = std::popcount(Src);
|
||||
break;
|
||||
}
|
||||
case IR::OP_FINDLSB: {
|
||||
auto Op = IROp->C<IR::IROp_FindLSB>();
|
||||
uint64_t Src = *GetSrc<uint64_t*>(SSAData, Op->Header.Args[0]);
|
||||
uint64_t Result = __builtin_ffsll(Src);
|
||||
uint64_t Result = FindFirstSetBit(Src);
|
||||
GD = Result - 1;
|
||||
break;
|
||||
}
|
||||
case IR::OP_FINDMSB: {
|
||||
auto Op = IROp->C<IR::IROp_FindMSB>();
|
||||
switch (OpSize) {
|
||||
case 1: GD = ((24 + OpSize * 8) - __builtin_clz(*GetSrc<uint8_t*>(SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 2: GD = ((16 + OpSize * 8) - __builtin_clz(*GetSrc<uint16_t*>(SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 4: GD = (OpSize * 8 - __builtin_clz(*GetSrc<uint32_t*>(SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 8: GD = (OpSize * 8 - __builtin_clzll(*GetSrc<uint64_t*>(SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 1: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint8_t*>(SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 2: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint16_t*>(SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 4: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint32_t*>(SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 8: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint64_t*>(SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
default: LOGMAN_MSG_A("Unknown REV size: %d", OpSize); break;
|
||||
}
|
||||
break;
|
||||
@@ -1890,9 +1927,9 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
case IR::OP_REV: {
|
||||
auto Op = IROp->C<IR::IROp_Rev>();
|
||||
switch (OpSize) {
|
||||
case 2: GD = __builtin_bswap16(*GetSrc<uint16_t*>(SSAData, Op->Header.Args[0])); break;
|
||||
case 4: GD = __builtin_bswap32(*GetSrc<uint32_t*>(SSAData, Op->Header.Args[0])); break;
|
||||
case 8: GD = __builtin_bswap64(*GetSrc<uint64_t*>(SSAData, Op->Header.Args[0])); break;
|
||||
case 2: GD = BSwap16(*GetSrc<uint16_t*>(SSAData, Op->Header.Args[0])); break;
|
||||
case 4: GD = BSwap32(*GetSrc<uint32_t*>(SSAData, Op->Header.Args[0])); break;
|
||||
case 8: GD = BSwap64(*GetSrc<uint64_t*>(SSAData, Op->Header.Args[0])); break;
|
||||
default: LOGMAN_MSG_A("Unknown REV size: %d", OpSize); break;
|
||||
}
|
||||
break;
|
||||
@@ -1902,34 +1939,22 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto Src = *GetSrc<uint8_t*>(SSAData, Op->Header.Args[0]);
|
||||
if (Src)
|
||||
GD = __builtin_ctz(Src);
|
||||
else
|
||||
GD = sizeof(Src) * 8;
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto Src = *GetSrc<uint16_t*>(SSAData, Op->Header.Args[0]);
|
||||
if (Src)
|
||||
GD = __builtin_ctz(Src);
|
||||
else
|
||||
GD = sizeof(Src) * 8;
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto Src = *GetSrc<uint32_t*>(SSAData, Op->Header.Args[0]);
|
||||
if (Src)
|
||||
GD = __builtin_ctz(Src);
|
||||
else
|
||||
GD = sizeof(Src) * 8;
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = *GetSrc<uint64_t*>(SSAData, Op->Header.Args[0]);
|
||||
if (Src)
|
||||
GD = __builtin_ctzll(Src);
|
||||
else
|
||||
GD = sizeof(Src) * 8;
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown size: %d", OpSize); break;
|
||||
@@ -1940,37 +1965,23 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
auto Op = IROp->C<IR::IROp_CountLeadingZeroes>();
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
uint32_t Src = *GetSrc<uint8_t*>(SSAData, Op->Header.Args[0]);
|
||||
Src <<= 24;
|
||||
if (Src)
|
||||
GD = __builtin_clz(Src);
|
||||
else
|
||||
GD = 8;
|
||||
auto Src = *GetSrc<uint8_t*>(SSAData, Op->Header.Args[0]);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
uint32_t Src = *GetSrc<uint16_t*>(SSAData, Op->Header.Args[0]);
|
||||
Src <<= 16;
|
||||
if (Src)
|
||||
GD = __builtin_clz(Src);
|
||||
else
|
||||
GD = 16;
|
||||
auto Src = *GetSrc<uint16_t*>(SSAData, Op->Header.Args[0]);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto Src = *GetSrc<uint32_t*>(SSAData, Op->Header.Args[0]);
|
||||
if (Src)
|
||||
GD = __builtin_clz(Src);
|
||||
else
|
||||
GD = sizeof(Src) * 8;
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = *GetSrc<uint64_t*>(SSAData, Op->Header.Args[0]);
|
||||
if (Src)
|
||||
GD = __builtin_clzll(Src);
|
||||
else
|
||||
GD = sizeof(Src) * 8;
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown size: %d", OpSize); break;
|
||||
@@ -2543,6 +2554,16 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
memcpy(GDP, &Dst, 16);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VBIC: {
|
||||
auto Op = IROp->C<IR::IROp_VBic>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(SSAData, Op->Header.Args[1]);
|
||||
|
||||
__uint128_t Dst = Src1 & ~Src2;
|
||||
memcpy(GDP, &Dst, 16);
|
||||
break;
|
||||
}
|
||||
|
||||
case IR::OP_VXOR: {
|
||||
auto Op = IROp->C<IR::IROp_VXor>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(SSAData, Op->Header.Args[0]);
|
||||
@@ -2988,6 +3009,24 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
memcpy(GDP, Tmp, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VUMINV: {
|
||||
auto Op = IROp->C<IR::IROp_VUMinV>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16];
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto current, auto a) { return std::min(current, a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_REDUCE_1SRC_OP(1, uint8_t, Func, ~0)
|
||||
DO_VECTOR_REDUCE_1SRC_OP(2, uint16_t, Func, ~0)
|
||||
DO_VECTOR_REDUCE_1SRC_OP(4, uint32_t, Func, ~0U)
|
||||
DO_VECTOR_REDUCE_1SRC_OP(8, uint64_t, Func, ~0ULL)
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VURAVG: {
|
||||
auto Op = IROp->C<IR::IROp_VURAvg>();
|
||||
void *Src1 = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
@@ -3023,6 +3062,24 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VPOPCOUNT: {
|
||||
auto Op = IROp->C<IR::IROp_VPopcount>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16];
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a) { return std::popcount(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(1, uint8_t, Func)
|
||||
DO_VECTOR_1SRC_OP(2, uint16_t, Func)
|
||||
DO_VECTOR_1SRC_OP(4, uint32_t, Func)
|
||||
DO_VECTOR_1SRC_OP(8, uint64_t, Func)
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VFMUL: {
|
||||
auto Op = IROp->C<IR::IROp_VFMul>();
|
||||
void *Src1 = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
@@ -3292,22 +3349,6 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VECTOR_UTOF: {
|
||||
auto Op = IROp->C<IR::IROp_Vector_UToF>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16];
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, uint32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, uint64_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VECTOR_STOF: {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
@@ -3324,22 +3365,6 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VECTOR_FTOZU: {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZU>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16];
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, uint32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, uint64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VECTOR_FTOZS: {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
@@ -3347,7 +3372,7 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
@@ -3356,22 +3381,6 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VECTOR_FTOU: {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToU>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16];
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, uint32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, uint64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VECTOR_FTOS: {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
@@ -3379,7 +3388,7 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
@@ -3502,6 +3511,28 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
memcpy(GDP, Tmp, Op->Header.Size);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VUABDL: {
|
||||
auto Op = IROp->C<IR::IROp_VUABDL>();
|
||||
void *Src1 = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(SSAData, Op->Header.Args[1]);
|
||||
|
||||
uint8_t Tmp[16];
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func8 = [](auto a, auto b) { return std::abs((int16_t)a - (int16_t)b); };
|
||||
auto Func16 = [](auto a, auto b) { return std::abs((int32_t)a - (int32_t)b); };
|
||||
auto Func32 = [](auto a, auto b) { return std::abs((int64_t)a - (int64_t)b); };
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_2SRC_2TYPE_OP(2, uint16_t, uint8_t, Func8)
|
||||
DO_VECTOR_2SRC_2TYPE_OP(4, uint32_t, uint16_t, Func16)
|
||||
DO_VECTOR_2SRC_2TYPE_OP(8, uint64_t, uint32_t, Func32)
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VSXTL: {
|
||||
auto Op = IROp->C<IR::IROp_VSXTL>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
@@ -3822,6 +3853,64 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VUNZIP2:
|
||||
case IR::OP_VUNZIP: {
|
||||
auto Op = IROp->C<IR::IROp_VUnZip>();
|
||||
void *Src1 = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(SSAData, Op->Header.Args[1]);
|
||||
uint8_t Tmp[16];
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
unsigned Start = IROp->Op == IR::OP_VUNZIP ? 0 : 1;
|
||||
Elements >>= 1;
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
auto *Dst_d = reinterpret_cast<uint8_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint8_t*>(Src1);
|
||||
auto *Src2_d = reinterpret_cast<uint8_t*>(Src2);
|
||||
for (unsigned i = 0; i < Elements; ++i) {
|
||||
Dst_d[i] = Src1_d[Start + (i * 2)];
|
||||
Dst_d[Elements+i] = Src2_d[Start + (i * 2)];
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto *Dst_d = reinterpret_cast<uint16_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint16_t*>(Src1);
|
||||
auto *Src2_d = reinterpret_cast<uint16_t*>(Src2);
|
||||
for (unsigned i = 0; i < Elements; ++i) {
|
||||
Dst_d[i] = Src1_d[Start + (i * 2)];
|
||||
Dst_d[Elements+i] = Src2_d[Start + (i * 2)];
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto *Dst_d = reinterpret_cast<uint32_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint32_t*>(Src1);
|
||||
auto *Src2_d = reinterpret_cast<uint32_t*>(Src2);
|
||||
for (unsigned i = 0; i < Elements; ++i) {
|
||||
Dst_d[i] = Src1_d[Start + (i * 2)];
|
||||
Dst_d[Elements+i] = Src2_d[Start + (i * 2)];
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto *Dst_d = reinterpret_cast<uint64_t*>(Tmp);
|
||||
auto *Src1_d = reinterpret_cast<uint64_t*>(Src1);
|
||||
auto *Src2_d = reinterpret_cast<uint64_t*>(Src2);
|
||||
for (unsigned i = 0; i < Elements; ++i) {
|
||||
Dst_d[i] = Src1_d[Start + (i * 2)];
|
||||
Dst_d[Elements+i] = Src2_d[Start + (i * 2)];
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
|
||||
case IR::OP_VINSELEMENT: {
|
||||
auto Op = IROp->C<IR::IROp_VInsElement>();
|
||||
void *Src1 = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
@@ -4204,6 +4293,10 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
|
||||
uint64_t Offset = Op->Index * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
Mask = ~0ULL;
|
||||
}
|
||||
Src2 = Src2 & Mask;
|
||||
Mask <<= Offset;
|
||||
Mask = ~Mask;
|
||||
__uint128_t Dst = Src1 & Mask;
|
||||
@@ -4240,78 +4333,57 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_FLOAT_FROMGPR_U: {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_U>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
float Dst = (float)*GetSrc<uint32_t*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
float Dst = (float)*GetSrc<uint64_t*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
double Dst = (double)*GetSrc<uint32_t*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
double Dst = (double)*GetSrc<uint64_t*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_FLOAT_TOGPR_ZS: {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
int64_t Dst = (int64_t)*GetSrc<double*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
}
|
||||
else {
|
||||
int32_t Dst = (int32_t)*GetSrc<float*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_FLOAT_TOGPR_ZU: {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZU>();
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
uint64_t Dst = (uint64_t)*GetSrc<double*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
}
|
||||
else {
|
||||
uint32_t Dst = (uint32_t)*GetSrc<float*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // int64_t <- float
|
||||
int64_t Dst = (int64_t)std::trunc(*GetSrc<float*>(SSAData, Op->Header.Args[0]));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // int64_t <- double
|
||||
int64_t Dst = (int64_t)std::trunc(*GetSrc<double*>(SSAData, Op->Header.Args[0]));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // int32_t <- float
|
||||
int32_t Dst = (int32_t)std::trunc(*GetSrc<float*>(SSAData, Op->Header.Args[0]));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // int32_t <- double
|
||||
int32_t Dst = (int32_t)std::trunc(*GetSrc<double*>(SSAData, Op->Header.Args[0]));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_FLOAT_TOGPR_S: {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
int64_t Dst = (int64_t)*GetSrc<double*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
}
|
||||
else {
|
||||
int32_t Dst = (int32_t)*GetSrc<float*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_FLOAT_TOGPR_U: {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_U>();
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
uint64_t Dst = (uint64_t)*GetSrc<double*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
}
|
||||
else {
|
||||
uint32_t Dst = (uint32_t)*GetSrc<float*>(SSAData, Op->Header.Args[0]);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // int64_t <- float
|
||||
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<float*>(SSAData, Op->Header.Args[0]));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // int64_t <- double
|
||||
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<double*>(SSAData, Op->Header.Args[0]));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // int32_t <- float
|
||||
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<float*>(SSAData, Op->Header.Args[0]));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // int32_t <- double
|
||||
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<double*>(SSAData, Op->Header.Args[0]));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -4363,6 +4435,53 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_VECTOR_FTOI: {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
void *Src = GetSrc<void*>(SSAData, Op->Header.Args[0]);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
auto Func_Trunc = [](auto a) { return std::trunc(a); };
|
||||
auto Func_Host = [](auto a) { return std::rint(a); };
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Host)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Host)
|
||||
}
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case IR::OP_FCMP: {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
uint32_t ResultFlags{};
|
||||
|
||||
+29
-17
@@ -865,8 +865,7 @@ DEF_OP(Bfi) {
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_A(OpSize <= 8, "OpSize is too large for BFE: %d", OpSize);
|
||||
LOGMAN_THROW_A(IROp->Size <= 8, "OpSize is too large for BFE: %d", IROp->Size);
|
||||
LOGMAN_THROW_A(Op->Width != 0, "Invalid BFE width of 0");
|
||||
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
@@ -969,34 +968,49 @@ DEF_OP(VExtractToGPR) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_ToGPR_ZU) {
|
||||
LogMan::Msg::D("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(Float_ToGPR_ZS) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
fcvtzs(GetReg<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()).D());
|
||||
aarch64::Register Dst{};
|
||||
aarch64::VRegister Src{};
|
||||
if (Op->SrcElementSize == 8) {
|
||||
Src = GetSrc(Op->Header.Args[0].ID()).D();
|
||||
}
|
||||
else {
|
||||
fcvtzs(GetReg<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()).S());
|
||||
Src = GetSrc(Op->Header.Args[0].ID()).S();
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_ToGPR_U) {
|
||||
LogMan::Msg::D("Unimplemented");
|
||||
if (IROp->Size == 8) {
|
||||
Dst = GetReg<RA_64>(Node);
|
||||
}
|
||||
else {
|
||||
Dst = GetReg<RA_32>(Node);
|
||||
}
|
||||
|
||||
fcvtzs(Dst, Src);
|
||||
}
|
||||
|
||||
DEF_OP(Float_ToGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
|
||||
aarch64::Register Dst{};
|
||||
aarch64::VRegister Src{};
|
||||
if (Op->SrcElementSize == 8) {
|
||||
frinti(VTMP1.D(), GetSrc(Op->Header.Args[0].ID()).D());
|
||||
fcvtzs(GetReg<RA_64>(Node), VTMP1.D());
|
||||
Src = VTMP1.D();
|
||||
}
|
||||
else {
|
||||
frinti(VTMP1.S(), GetSrc(Op->Header.Args[0].ID()).S());
|
||||
fcvtzs(GetReg<RA_32>(Node), VTMP1.S());
|
||||
Src = VTMP1.S();
|
||||
}
|
||||
|
||||
if (IROp->Size == 8) {
|
||||
Dst = GetReg<RA_64>(Node);
|
||||
}
|
||||
else {
|
||||
Dst = GetReg<RA_32>(Node);
|
||||
}
|
||||
|
||||
fcvtzs(Dst, Src);
|
||||
}
|
||||
|
||||
DEF_OP(FCmp) {
|
||||
@@ -1087,9 +1101,7 @@ void Arm64JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(SBFE, Sbfe);
|
||||
REGISTER_OP(SELECT, Select);
|
||||
REGISTER_OP(VEXTRACTTOGPR, VExtractToGPR);
|
||||
REGISTER_OP(FLOAT_TOGPR_ZU, Float_ToGPR_ZU);
|
||||
REGISTER_OP(FLOAT_TOGPR_ZS, Float_ToGPR_ZS);
|
||||
REGISTER_OP(FLOAT_TOGPR_U, Float_ToGPR_U);
|
||||
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
|
||||
REGISTER_OP(FCMP, FCmp);
|
||||
|
||||
|
||||
@@ -257,7 +257,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
uint8_t *OldCode = (uint8_t *)&Op->CodeOriginalLow;
|
||||
const auto *OldCode = (const uint8_t *)&Op->CodeOriginalLow;
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
@@ -268,7 +268,7 @@ DEF_OP(ValidateCode) {
|
||||
while (len >= 8)
|
||||
{
|
||||
ldr(x2, MemOperand(x0, idx));
|
||||
LoadConstant(x3, *(uint32_t *)(OldCode + idx));
|
||||
LoadConstant(x3, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(x2, x3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
len -= 8;
|
||||
@@ -277,7 +277,7 @@ DEF_OP(ValidateCode) {
|
||||
while (len >= 4)
|
||||
{
|
||||
ldr(w2, MemOperand(x0, idx));
|
||||
LoadConstant(w3, *(uint32_t *)(OldCode + idx));
|
||||
LoadConstant(w3, *(const uint32_t *)(OldCode + idx));
|
||||
cmp(w2, w3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
len -= 4;
|
||||
@@ -286,7 +286,7 @@ DEF_OP(ValidateCode) {
|
||||
while (len >= 2)
|
||||
{
|
||||
ldrh(w2, MemOperand(x0, idx));
|
||||
LoadConstant(w3, *(uint16_t *)(OldCode + idx));
|
||||
LoadConstant(w3, *(const uint16_t *)(OldCode + idx));
|
||||
cmp(w2, w3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
len -= 2;
|
||||
@@ -295,7 +295,7 @@ DEF_OP(ValidateCode) {
|
||||
while (len >= 1)
|
||||
{
|
||||
ldrb(w2, MemOperand(x0, idx));
|
||||
LoadConstant(w3, *(uint8_t *)(OldCode + idx));
|
||||
LoadConstant(w3, *(const uint8_t *)(OldCode + idx));
|
||||
cmp(w2, w3);
|
||||
csel(GetReg<RA_64>(Node), GetReg<RA_64>(Node), x1, Condition::eq);
|
||||
len -= 1;
|
||||
|
||||
@@ -56,10 +56,6 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_U) {
|
||||
LOGMAN_MSG_A("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
@@ -99,19 +95,6 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_UToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_UToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
ucvtf(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
ucvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
@@ -125,19 +108,6 @@ DEF_OP(Vector_SToF) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZU) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZU>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzu(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzu(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
@@ -151,21 +121,6 @@ DEF_OP(Vector_FToZS) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToU) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToU>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
fcvtzu(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
fcvtzu(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unknown castGPR element size: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
@@ -198,21 +153,74 @@ DEF_OP(Vector_FToF) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterConversionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_U, Float_FromGPR_U);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_UTOF, Vector_UToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
REGISTER_OP(VECTOR_FTOZU, Vector_FToZU);
|
||||
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
|
||||
REGISTER_OP(VECTOR_FTOU, Vector_FToU);
|
||||
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
|
||||
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
|
||||
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+34
-24
@@ -23,6 +23,7 @@ $end_info$
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <sys/mman.h>
|
||||
@@ -43,8 +44,10 @@ using namespace vixl::aarch64;
|
||||
void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LOGMAN_MSG_A("Unhandled IR Op: %s", std::string(Name).c_str());
|
||||
#endif
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16:{
|
||||
@@ -291,8 +294,11 @@ void Arm64JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LOGMAN_MSG_A("Unhandled IR Fallback abi: %s %d", std::string(Name).c_str(), Info.ABI);
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -501,17 +507,17 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
|
||||
});
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->HandleSIGBUS(Signal, info, ucontext);
|
||||
});
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
});
|
||||
}, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
@@ -575,7 +581,7 @@ Arm64JITCore::~Arm64JITCore() {
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
}
|
||||
|
||||
IR::PhysicalRegister Arm64JITCore::GetPhys(uint32_t Node) {
|
||||
IR::PhysicalRegister Arm64JITCore::GetPhys(uint32_t Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa%d. Class: %d", Node, PhyReg.Class);
|
||||
@@ -584,7 +590,7 @@ IR::PhysicalRegister Arm64JITCore::GetPhys(uint32_t Node) {
|
||||
}
|
||||
|
||||
template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(uint32_t Node) {
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(uint32_t Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
@@ -594,11 +600,12 @@ aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(uint32_t Node) {
|
||||
} else {
|
||||
LOGMAN_THROW_A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
__builtin_unreachable();
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(uint32_t Node) {
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(uint32_t Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
@@ -608,22 +615,23 @@ aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(uint32_t Node) {
|
||||
} else {
|
||||
LOGMAN_THROW_A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
__builtin_unreachable();
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_32>(uint32_t Node) {
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_32>(uint32_t Node) const {
|
||||
uint32_t Reg = GetPhys(Node).Reg;
|
||||
return RA32Pair[Reg];
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_64>(uint32_t Node) {
|
||||
std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JITCore::RA_64>(uint32_t Node) const {
|
||||
uint32_t Reg = GetPhys(Node).Reg;
|
||||
return RA64Pair[Reg];
|
||||
}
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(uint32_t Node) {
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(uint32_t Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
@@ -633,10 +641,11 @@ aarch64::VRegister Arm64JITCore::GetSrc(uint32_t Node) {
|
||||
} else {
|
||||
LOGMAN_THROW_A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
__builtin_unreachable();
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
aarch64::VRegister Arm64JITCore::GetDst(uint32_t Node) {
|
||||
aarch64::VRegister Arm64JITCore::GetDst(uint32_t Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
@@ -646,10 +655,11 @@ aarch64::VRegister Arm64JITCore::GetDst(uint32_t Node) {
|
||||
} else {
|
||||
LOGMAN_THROW_A(false, "Unexpected Class: %d", Reg.Class);
|
||||
}
|
||||
__builtin_unreachable();
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
|
||||
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINECONSTANT) {
|
||||
@@ -663,7 +673,7 @@ bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_
|
||||
}
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
|
||||
bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
@@ -677,18 +687,18 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(uint32_t Node) {
|
||||
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(uint32_t Node) const {
|
||||
return FEXCore::IR::RegisterClassType {GetPhys(Node).Class};
|
||||
}
|
||||
|
||||
|
||||
bool Arm64JITCore::IsFPR(uint32_t Node) {
|
||||
bool Arm64JITCore::IsFPR(uint32_t Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPR(uint32_t Node) {
|
||||
bool Arm64JITCore::IsGPR(uint32_t Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
@@ -702,8 +712,6 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
#ifndef NDEBUG
|
||||
LoadConstant(x0, Entry);
|
||||
#endif
|
||||
@@ -781,8 +789,10 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
{
|
||||
uint32_t Node = IR->GetID(BlockNode);
|
||||
@@ -888,7 +898,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuSt
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
FEXCore::CPU::CPUBackend *CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return new Arm64JITCore(ctx, Thread, CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<Arm64JITCore>(ctx, Thread, CompileThread);
|
||||
}
|
||||
}
|
||||
+22
-19
@@ -96,35 +96,35 @@ private:
|
||||
constexpr static uint8_t RA_FPR = 2;
|
||||
|
||||
template<uint8_t RAType>
|
||||
aarch64::Register GetReg(uint32_t Node);
|
||||
aarch64::Register GetReg(uint32_t Node) const;
|
||||
|
||||
template<>
|
||||
aarch64::Register GetReg<RA_32>(uint32_t Node);
|
||||
aarch64::Register GetReg<RA_32>(uint32_t Node) const;
|
||||
template<>
|
||||
aarch64::Register GetReg<RA_64>(uint32_t Node);
|
||||
aarch64::Register GetReg<RA_64>(uint32_t Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair(uint32_t Node);
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair(uint32_t Node) const;
|
||||
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_32>(uint32_t Node);
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_32>(uint32_t Node) const;
|
||||
template<>
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_64>(uint32_t Node);
|
||||
std::pair<aarch64::Register, aarch64::Register> GetSrcPair<RA_64>(uint32_t Node) const;
|
||||
|
||||
aarch64::VRegister GetSrc(uint32_t Node);
|
||||
aarch64::VRegister GetDst(uint32_t Node);
|
||||
aarch64::VRegister GetSrc(uint32_t Node) const;
|
||||
aarch64::VRegister GetDst(uint32_t Node) const;
|
||||
|
||||
FEXCore::IR::RegisterClassType GetRegClass(uint32_t Node);
|
||||
FEXCore::IR::RegisterClassType GetRegClass(uint32_t Node) const;
|
||||
|
||||
IR::PhysicalRegister GetPhys(uint32_t Node);
|
||||
IR::PhysicalRegister GetPhys(uint32_t Node) const;
|
||||
|
||||
bool IsFPR(uint32_t Node);
|
||||
bool IsGPR(uint32_t Node);
|
||||
bool IsFPR(uint32_t Node) const;
|
||||
bool IsGPR(uint32_t Node) const;
|
||||
|
||||
MemOperand GenerateMemOperand(uint8_t AccessSize, aarch64::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr);
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value);
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
|
||||
struct LiveRange {
|
||||
uint32_t Begin;
|
||||
@@ -240,7 +240,6 @@ private:
|
||||
DEF_OP(VExtractToGPR);
|
||||
DEF_OP(Float_ToGPR_ZU);
|
||||
DEF_OP(Float_ToGPR_ZS);
|
||||
DEF_OP(Float_ToGPR_U);
|
||||
DEF_OP(Float_ToGPR_S);
|
||||
DEF_OP(FCmp);
|
||||
|
||||
@@ -277,16 +276,13 @@ private:
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(Float_FromGPR_U);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_UToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
DEF_OP(Vector_FToZU);
|
||||
DEF_OP(Vector_FToZS);
|
||||
DEF_OP(Vector_FToU);
|
||||
DEF_OP(Vector_FToS);
|
||||
DEF_OP(Vector_FToF);
|
||||
DEF_OP(Vector_FToI);
|
||||
|
||||
///< Flag ops
|
||||
DEF_OP(GetHostFlag);
|
||||
@@ -336,6 +332,7 @@ private:
|
||||
DEF_OP(SplatVector4);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
DEF_OP(VOr);
|
||||
DEF_OP(VXor);
|
||||
DEF_OP(VAdd);
|
||||
@@ -346,8 +343,10 @@ private:
|
||||
DEF_OP(VSQSub);
|
||||
DEF_OP(VAddP);
|
||||
DEF_OP(VAddV);
|
||||
DEF_OP(VUMinV);
|
||||
DEF_OP(VURAvg);
|
||||
DEF_OP(VAbs);
|
||||
DEF_OP(VPopcount);
|
||||
DEF_OP(VFAdd);
|
||||
DEF_OP(VFAddP);
|
||||
DEF_OP(VFSub);
|
||||
@@ -367,6 +366,8 @@ private:
|
||||
DEF_OP(VSMax);
|
||||
DEF_OP(VZip);
|
||||
DEF_OP(VZip2);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VUnZip2);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
@@ -389,6 +390,7 @@ private:
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
@@ -411,6 +413,7 @@ private:
|
||||
DEF_OP(VSMull);
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
|
||||
///< Encryption ops
|
||||
|
||||
@@ -5,6 +5,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -554,7 +555,8 @@ MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Registe
|
||||
}
|
||||
}
|
||||
}
|
||||
__builtin_unreachable();
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
|
||||
+177
-4
@@ -127,6 +127,11 @@ DEF_OP(VAnd) {
|
||||
and_(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(VBic) {
|
||||
auto Op = IROp->C<IR::IROp_VBic>();
|
||||
bic(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(VOr) {
|
||||
auto Op = IROp->C<IR::IROp_VOr>();
|
||||
orr(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
@@ -337,6 +342,21 @@ DEF_OP(VAddV) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUMinV) {
|
||||
auto Op = IROp->C<IR::IROp_VUMinV>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
// Vector
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
case 2:
|
||||
case 4:
|
||||
uminv(GetDst(Node).VCast(Op->Header.ElementSize * 8, 1), GetSrc(Op->Header.Args[0].ID()).VCast(OpSize * 8, Elements));
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VURAvg) {
|
||||
auto Op = IROp->C<IR::IROp_VURAvg>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
@@ -380,6 +400,30 @@ DEF_OP(VAbs) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VPopcount) {
|
||||
auto Op = IROp->C<IR::IROp_VPopcount>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 8) {
|
||||
// Scalar
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
cnt(GetDst(Node).V8B(), GetSrc(Op->Header.Args[0].ID()).V8B());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Vector
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
cnt(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFAdd) {
|
||||
auto Op = IROp->C<IR::IROp_VFAdd>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -722,7 +766,6 @@ DEF_OP(VFRSqrt) {
|
||||
|
||||
DEF_OP(VNeg) {
|
||||
auto Op = IROp->C<IR::IROp_VNeg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
neg(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
@@ -736,13 +779,12 @@ DEF_OP(VNeg) {
|
||||
case 8:
|
||||
neg(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unsupported Not size: %d", OpSize);
|
||||
default: LOGMAN_MSG_A("Unsupported Not size: %d", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFNeg) {
|
||||
auto Op = IROp->C<IR::IROp_VFNeg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fneg(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
@@ -750,7 +792,7 @@ DEF_OP(VFNeg) {
|
||||
case 8:
|
||||
fneg(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unsupported Not size: %d", OpSize);
|
||||
default: LOGMAN_MSG_A("Unsupported Not size: %d", IROp->Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -949,6 +991,92 @@ DEF_OP(VZip2) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUnZip) {
|
||||
auto Op = IROp->C<IR::IROp_VUnZip>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 8) {
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
uzp1(GetDst(Node).V8B(), GetSrc(Op->Header.Args[0].ID()).V8B(), GetSrc(Op->Header.Args[1].ID()).V8B());
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
uzp1(GetDst(Node).V4H(), GetSrc(Op->Header.Args[0].ID()).V4H(), GetSrc(Op->Header.Args[1].ID()).V4H());
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uzp1(GetDst(Node).V2S(), GetSrc(Op->Header.Args[0].ID()).V2S(), GetSrc(Op->Header.Args[1].ID()).V2S());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
uzp1(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
uzp1(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8H(), GetSrc(Op->Header.Args[1].ID()).V8H());
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uzp1(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S(), GetSrc(Op->Header.Args[1].ID()).V4S());
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uzp1(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D(), GetSrc(Op->Header.Args[1].ID()).V2D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUnZip2) {
|
||||
auto Op = IROp->C<IR::IROp_VUnZip2>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 8) {
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
uzp2(GetDst(Node).V8B(), GetSrc(Op->Header.Args[0].ID()).V8B(), GetSrc(Op->Header.Args[1].ID()).V8B());
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
uzp2(GetDst(Node).V4H(), GetSrc(Op->Header.Args[0].ID()).V4H(), GetSrc(Op->Header.Args[1].ID()).V4H());
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uzp2(GetDst(Node).V2S(), GetSrc(Op->Header.Args[0].ID()).V2S(), GetSrc(Op->Header.Args[1].ID()).V2S());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
uzp2(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
uzp2(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8H(), GetSrc(Op->Header.Args[1].ID()).V8H());
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uzp2(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S(), GetSrc(Op->Header.Args[1].ID()).V4S());
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uzp2(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D(), GetSrc(Op->Header.Args[1].ID()).V2D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VBSL) {
|
||||
auto Op = IROp->C<IR::IROp_VBSL>();
|
||||
if (IROp->Size == 16) {
|
||||
@@ -1648,6 +1776,25 @@ DEF_OP(VExtractElement) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VDupElement) {
|
||||
auto Op = IROp->C<IR::IROp_VDupElement>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
dup(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), Op->Index);
|
||||
break;
|
||||
case 2:
|
||||
dup(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8H(), Op->Index);
|
||||
break;
|
||||
case 4:
|
||||
dup(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S(), Op->Index);
|
||||
break;
|
||||
case 8:
|
||||
dup(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D(), Op->Index);
|
||||
break;
|
||||
default: LOGMAN_MSG_A("Unhandled DupElementSize: %d", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VExtr) {
|
||||
auto Op = IROp->C<IR::IROp_VExtr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -2135,6 +2282,25 @@ DEF_OP(VSMull2) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL) {
|
||||
auto Op = IROp->C<IR::IROp_VUABDL>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 2: {
|
||||
uabdl(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8B(), GetSrc(Op->Header.Args[1].ID()).V8B());
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uabdl(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4H(), GetSrc(Op->Header.Args[1].ID()).V4H());
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uabdl(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2S(), GetSrc(Op->Header.Args[1].ID()).V2S());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize >> 1); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -2163,6 +2329,7 @@ void Arm64JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(SPLATVECTOR4, SplatVector4);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
REGISTER_OP(VAND, VAnd);
|
||||
REGISTER_OP(VBIC, VBic);
|
||||
REGISTER_OP(VOR, VOr);
|
||||
REGISTER_OP(VXOR, VXor);
|
||||
REGISTER_OP(VADD, VAdd);
|
||||
@@ -2173,8 +2340,10 @@ void Arm64JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VSQSUB, VSQSub);
|
||||
REGISTER_OP(VADDP, VAddP);
|
||||
REGISTER_OP(VADDV, VAddV);
|
||||
REGISTER_OP(VUMINV, VUMinV);
|
||||
REGISTER_OP(VURAVG, VURAvg);
|
||||
REGISTER_OP(VABS, VAbs);
|
||||
REGISTER_OP(VPOPCOUNT, VPopcount);
|
||||
REGISTER_OP(VFADD, VFAdd);
|
||||
REGISTER_OP(VFADDP, VFAddP);
|
||||
REGISTER_OP(VFSUB, VFSub);
|
||||
@@ -2194,6 +2363,8 @@ void Arm64JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VSMAX, VSMax);
|
||||
REGISTER_OP(VZIP, VZip);
|
||||
REGISTER_OP(VZIP2, VZip2);
|
||||
REGISTER_OP(VUNZIP, VUnZip);
|
||||
REGISTER_OP(VUNZIP2, VUnZip2);
|
||||
REGISTER_OP(VBSL, VBSL);
|
||||
REGISTER_OP(VCMPEQ, VCMPEQ);
|
||||
REGISTER_OP(VCMPEQZ, VCMPEQZ);
|
||||
@@ -2216,6 +2387,7 @@ void Arm64JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VINSSCALARELEMENT, VInsScalarElement);
|
||||
REGISTER_OP(VEXTRACTELEMENT, VExtractElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VSLI, VSLI);
|
||||
REGISTER_OP(VSRI, VSRI);
|
||||
@@ -2239,6 +2411,7 @@ void Arm64JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VSMULL, VSMull);
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
+4
-2
@@ -1,5 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
@@ -11,6 +13,6 @@ struct InternalThreadState;
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
FEXCore::CPU::CPUBackend *CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
FEXCore::CPU::CPUBackend *CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread);
|
||||
}
|
||||
+43
-26
@@ -970,9 +970,7 @@ DEF_OP(Bfi) {
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_A(OpSize <= 8, "OpSize is too large for BFE: %d", OpSize);
|
||||
LOGMAN_THROW_A(IROp->Size <= 8, "OpSize is too large for BFE: %d", IROp->Size);
|
||||
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
|
||||
@@ -1108,42 +1106,63 @@ DEF_OP(VExtractToGPR) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_ToGPR_ZU) {
|
||||
LogMan::Msg::D("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(Float_ToGPR_ZS) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
cvttsd2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
}
|
||||
else {
|
||||
cvttss2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_ToGPR_U) {
|
||||
LogMan::Msg::D("Unimplemented");
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: // int64_t <- float
|
||||
cvttss2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
case 0x0808: // int64_t <- double
|
||||
cvttsd2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
case 0x0404: // int32_t <- float
|
||||
cvttss2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
case 0x0408: // int32_t <- double
|
||||
cvttsd2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_ToGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
cvtsd2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
}
|
||||
else {
|
||||
cvtss2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: // int64_t <- float
|
||||
cvtss2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
case 0x0808: // int64_t <- double
|
||||
cvtsd2si(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
case 0x0404: // int32_t <- float
|
||||
cvtss2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
case 0x0408: // int32_t <- double
|
||||
cvtsd2si(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
|
||||
if (Op->ElementSize == 4) {
|
||||
ucomiss(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED)) {
|
||||
if (Op->ElementSize == 4) {
|
||||
ucomiss(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
}
|
||||
else {
|
||||
ucomisd(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
}
|
||||
}
|
||||
else {
|
||||
ucomisd(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
if (Op->ElementSize == 4) {
|
||||
comiss(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
}
|
||||
else {
|
||||
comisd(GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
}
|
||||
}
|
||||
mov (rdx, 0);
|
||||
|
||||
@@ -1217,9 +1236,7 @@ void X86JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(SBFE, Sbfe);
|
||||
REGISTER_OP(SELECT, Select);
|
||||
REGISTER_OP(VEXTRACTTOGPR, VExtractToGPR);
|
||||
REGISTER_OP(FLOAT_TOGPR_ZU, Float_ToGPR_ZU);
|
||||
REGISTER_OP(FLOAT_TOGPR_ZS, Float_ToGPR_ZS);
|
||||
REGISTER_OP(FLOAT_TOGPR_U, Float_ToGPR_U);
|
||||
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
|
||||
REGISTER_OP(FCMP, FCmp);
|
||||
#undef REGISTER_OP
|
||||
|
||||
@@ -248,7 +248,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
uint8_t* OldCode = (uint8_t*)&Op->CodeOriginalLow;
|
||||
const auto* OldCode = (const uint8_t*)&Op->CodeOriginalLow;
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
@@ -256,20 +256,20 @@ DEF_OP(ValidateCode) {
|
||||
mov(rax, Entry + Op->Offset);
|
||||
mov(rbx, 1);
|
||||
while (len >= 4) {
|
||||
cmp(dword[rax + idx], *(uint32_t*)(OldCode + idx));
|
||||
cmp(dword[rax + idx], *(const uint32_t*)(OldCode + idx));
|
||||
cmovne(GetDst<RA_64>(Node), rbx);
|
||||
len-=4;
|
||||
idx+=4;
|
||||
}
|
||||
while (len >= 2) {
|
||||
mov(rcx, *(uint16_t*)(OldCode + idx));
|
||||
mov(rcx, *(const uint16_t*)(OldCode + idx));
|
||||
cmp(word[rax + idx], cx);
|
||||
cmovne(GetDst<RA_64>(Node), rbx);
|
||||
len-=2;
|
||||
idx+=2;
|
||||
}
|
||||
while (len >= 1) {
|
||||
cmp(byte[rax + idx], *(uint8_t*)(OldCode + idx));
|
||||
cmp(byte[rax + idx], *(const uint8_t*)(OldCode + idx));
|
||||
cmovne(GetDst<RA_64>(Node), rbx);
|
||||
len-=1;
|
||||
idx+=1;
|
||||
@@ -319,8 +319,9 @@ DEF_OP(CPUID) {
|
||||
//
|
||||
// Result: RAX, RDX. 4xi32
|
||||
|
||||
mov (rsi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov (rdx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
// rsi can be in the source registers, so copy argument to edx first
|
||||
mov (edx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov (esi, GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
mov (rdi, reinterpret_cast<uint64_t>(&CTX->CPUID));
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
@@ -56,10 +56,6 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_U) {
|
||||
LOGMAN_MSG_A("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
@@ -99,10 +95,6 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_UToF) {
|
||||
LOGMAN_MSG_A("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
@@ -125,10 +117,6 @@ DEF_OP(Vector_SToF) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZU) {
|
||||
LOGMAN_MSG_A("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
@@ -142,10 +130,6 @@ DEF_OP(Vector_FToZS) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToU) {
|
||||
LOGMAN_MSG_A("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
@@ -176,21 +160,50 @@ DEF_OP(Vector_FToF) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
uint8_t RoundMode{};
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
RoundMode = 0b0000'0'0'00;
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'01;
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'10;
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
RoundMode = 0b0000'0'0'11;
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
RoundMode = 0b0000'0'1'00;
|
||||
break;
|
||||
}
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
roundps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
|
||||
break;
|
||||
case 8:
|
||||
roundpd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterConversionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_U, Float_FromGPR_U);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_UTOF, Vector_UToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
REGISTER_OP(VECTOR_FTOZU, Vector_FToZU);
|
||||
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
|
||||
REGISTER_OP(VECTOR_FTOU, Vector_FToU);
|
||||
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
|
||||
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
|
||||
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+43
-36
@@ -84,8 +84,10 @@ void X86JITCore::PopRegs() {
|
||||
void X86JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LOGMAN_MSG_A("Unhandled IR Op: %s", std::string(Name).c_str());
|
||||
#endif
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16: {
|
||||
@@ -282,8 +284,11 @@ void X86JITCore::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, uint32_t Node) {
|
||||
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto Name = FEXCore::IR::GetName(IROp->Op);
|
||||
LOGMAN_MSG_A("Unhandled IR Fallback abi: %s %d", std::string(Name).c_str(), Info.ABI);
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -343,12 +348,12 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
|
||||
});
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
});
|
||||
}, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
@@ -411,7 +416,7 @@ void X86JITCore::ClearCache() {
|
||||
}
|
||||
}
|
||||
|
||||
IR::PhysicalRegister X86JITCore::GetPhys(uint32_t Node) {
|
||||
IR::PhysicalRegister X86JITCore::GetPhys(uint32_t Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa%d. Class: %d", Node, PhyReg.Class);
|
||||
@@ -419,98 +424,98 @@ IR::PhysicalRegister X86JITCore::GetPhys(uint32_t Node) {
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsFPR(uint32_t Node) {
|
||||
bool X86JITCore::IsFPR(uint32_t Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::FPRClass.Val;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsGPR(uint32_t Node) {
|
||||
bool X86JITCore::IsGPR(uint32_t Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRClass.Val;
|
||||
}
|
||||
|
||||
template<uint8_t RAType>
|
||||
Xbyak::Reg X86JITCore::GetSrc(uint32_t Node) {
|
||||
Xbyak::Reg X86JITCore::GetSrc(uint32_t Node) const {
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
// r10
|
||||
// Callee Saved
|
||||
// rbx, rbp, r12, r13, r14, r15
|
||||
auto PhyReg = GetPhys(Node);
|
||||
if (RAType == RA_64)
|
||||
if constexpr (RAType == RA_64)
|
||||
return RA64[PhyReg.Reg].cvt64();
|
||||
else if (RAType == RA_XMM)
|
||||
else if constexpr (RAType == RA_XMM)
|
||||
return RAXMM[PhyReg.Reg];
|
||||
else if (RAType == RA_32)
|
||||
else if constexpr (RAType == RA_32)
|
||||
return RA64[PhyReg.Reg].cvt32();
|
||||
else if (RAType == RA_16)
|
||||
else if constexpr (RAType == RA_16)
|
||||
return RA64[PhyReg.Reg].cvt16();
|
||||
else if (RAType == RA_8)
|
||||
else if constexpr (RAType == RA_8)
|
||||
return RA64[PhyReg.Reg].cvt8();
|
||||
}
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_64>(uint32_t Node);
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_64>(uint32_t Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_32>(uint32_t Node);
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_32>(uint32_t Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_16>(uint32_t Node);
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_16>(uint32_t Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_8>(uint32_t Node);
|
||||
Xbyak::Reg X86JITCore::GetSrc<X86JITCore::RA_8>(uint32_t Node) const;
|
||||
|
||||
Xbyak::Xmm X86JITCore::GetSrc(uint32_t Node) {
|
||||
Xbyak::Xmm X86JITCore::GetSrc(uint32_t Node) const {
|
||||
auto PhyReg = GetPhys(Node);
|
||||
return RAXMM_x[PhyReg.Reg];
|
||||
}
|
||||
|
||||
template<uint8_t RAType>
|
||||
Xbyak::Reg X86JITCore::GetDst(uint32_t Node) {
|
||||
Xbyak::Reg X86JITCore::GetDst(uint32_t Node) const {
|
||||
auto PhyReg = GetPhys(Node);
|
||||
if (RAType == RA_64)
|
||||
if constexpr (RAType == RA_64)
|
||||
return RA64[PhyReg.Reg].cvt64();
|
||||
else if (RAType == RA_XMM)
|
||||
else if constexpr (RAType == RA_XMM)
|
||||
return RAXMM[PhyReg.Reg];
|
||||
else if (RAType == RA_32)
|
||||
else if constexpr (RAType == RA_32)
|
||||
return RA64[PhyReg.Reg].cvt32();
|
||||
else if (RAType == RA_16)
|
||||
else if constexpr (RAType == RA_16)
|
||||
return RA64[PhyReg.Reg].cvt16();
|
||||
else if (RAType == RA_8)
|
||||
else if constexpr (RAType == RA_8)
|
||||
return RA64[PhyReg.Reg].cvt8();
|
||||
}
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_64>(uint32_t Node);
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_64>(uint32_t Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_32>(uint32_t Node);
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_32>(uint32_t Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_16>(uint32_t Node);
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_16>(uint32_t Node) const;
|
||||
|
||||
template
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_8>(uint32_t Node);
|
||||
Xbyak::Reg X86JITCore::GetDst<X86JITCore::RA_8>(uint32_t Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair(uint32_t Node) {
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair(uint32_t Node) const {
|
||||
auto PhyReg = GetPhys(Node);
|
||||
if (RAType == RA_64)
|
||||
if constexpr (RAType == RA_64)
|
||||
return RA64Pair[PhyReg.Reg];
|
||||
else if (RAType == RA_32)
|
||||
else if constexpr (RAType == RA_32)
|
||||
return {RA64Pair[PhyReg.Reg].first.cvt32(), RA64Pair[PhyReg.Reg].second.cvt32()};
|
||||
}
|
||||
|
||||
template
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_64>(uint32_t Node);
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_64>(uint32_t Node) const;
|
||||
|
||||
template
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_32>(uint32_t Node);
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> X86JITCore::GetSrcPair<X86JITCore::RA_32>(uint32_t Node) const;
|
||||
|
||||
Xbyak::Xmm X86JITCore::GetDst(uint32_t Node) {
|
||||
Xbyak::Xmm X86JITCore::GetDst(uint32_t Node) const {
|
||||
auto PhyReg = GetPhys(Node);
|
||||
return RAXMM_x[PhyReg.Reg];
|
||||
}
|
||||
|
||||
bool X86JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
|
||||
bool X86JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINECONSTANT) {
|
||||
@@ -524,7 +529,7 @@ bool X86JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t*
|
||||
}
|
||||
}
|
||||
|
||||
bool X86JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) {
|
||||
bool X86JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
@@ -662,8 +667,10 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
{
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
uint32_t Node = IR->GetID(BlockNode);
|
||||
auto IsTarget = JumpTargets.find(Node);
|
||||
@@ -764,7 +771,7 @@ uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateF
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
FEXCore::CPU::CPUBackend *CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return new X86JITCore(ctx, Thread, AllocateNewCodeBuffer(CompileThread ? X86JITCore::MAX_CODE_SIZE : X86JITCore::INITIAL_CODE_SIZE), CompileThread);
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(CompileThread ? X86JITCore::MAX_CODE_SIZE : X86JITCore::INITIAL_CODE_SIZE), CompileThread);
|
||||
}
|
||||
}
|
||||
+19
-16
@@ -112,26 +112,26 @@ private:
|
||||
constexpr static uint8_t RA_64 = 3;
|
||||
constexpr static uint8_t RA_XMM = 4;
|
||||
|
||||
IR::PhysicalRegister GetPhys(uint32_t Node);
|
||||
IR::PhysicalRegister GetPhys(uint32_t Node) const;
|
||||
|
||||
bool IsFPR(uint32_t Node);
|
||||
bool IsGPR(uint32_t Node);
|
||||
bool IsFPR(uint32_t Node) const;
|
||||
bool IsGPR(uint32_t Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
Xbyak::Reg GetSrc(uint32_t Node);
|
||||
Xbyak::Reg GetSrc(uint32_t Node) const;
|
||||
template<uint8_t RAType>
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> GetSrcPair(uint32_t Node);
|
||||
std::pair<Xbyak::Reg, Xbyak::Reg> GetSrcPair(uint32_t Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
Xbyak::Reg GetDst(uint32_t Node);
|
||||
Xbyak::Reg GetDst(uint32_t Node) const;
|
||||
|
||||
Xbyak::Xmm GetSrc(uint32_t Node);
|
||||
Xbyak::Xmm GetDst(uint32_t Node);
|
||||
Xbyak::Xmm GetSrc(uint32_t Node) const;
|
||||
Xbyak::Xmm GetDst(uint32_t Node) const;
|
||||
|
||||
Xbyak::RegExp GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
Xbyak::RegExp GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) const;
|
||||
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr);
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value);
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
FEXCore::IR::RegisterAllocationData *RAData;
|
||||
@@ -241,9 +241,7 @@ private:
|
||||
DEF_OP(Sbfe);
|
||||
DEF_OP(Select);
|
||||
DEF_OP(VExtractToGPR);
|
||||
DEF_OP(Float_ToGPR_ZU);
|
||||
DEF_OP(Float_ToGPR_ZS);
|
||||
DEF_OP(Float_ToGPR_U);
|
||||
DEF_OP(Float_ToGPR_S);
|
||||
DEF_OP(FCmp);
|
||||
DEF_OP(F80Cmp);
|
||||
@@ -281,16 +279,14 @@ private:
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(Float_FromGPR_U);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_UToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
DEF_OP(Vector_FToZU);
|
||||
DEF_OP(Vector_FToZS);
|
||||
DEF_OP(Vector_FToU);
|
||||
DEF_OP(Vector_FToS);
|
||||
DEF_OP(Vector_FToF);
|
||||
DEF_OP(Vector_FToI);
|
||||
|
||||
///< Flag ops
|
||||
DEF_OP(GetHostFlag);
|
||||
@@ -333,6 +329,7 @@ private:
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
DEF_OP(VOr);
|
||||
DEF_OP(VXor);
|
||||
DEF_OP(VAdd);
|
||||
@@ -343,8 +340,10 @@ private:
|
||||
DEF_OP(VSQSub);
|
||||
DEF_OP(VAddP);
|
||||
DEF_OP(VAddV);
|
||||
DEF_OP(VUMinV);
|
||||
DEF_OP(VURAvg);
|
||||
DEF_OP(VAbs);
|
||||
DEF_OP(VPopcount);
|
||||
DEF_OP(VFAdd);
|
||||
DEF_OP(VFAddP);
|
||||
DEF_OP(VFSub);
|
||||
@@ -364,6 +363,8 @@ private:
|
||||
DEF_OP(VSMax);
|
||||
DEF_OP(VZip);
|
||||
DEF_OP(VZip2);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VUnZip2);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
@@ -386,6 +387,7 @@ private:
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
DEF_OP(VSRI);
|
||||
@@ -408,6 +410,7 @@ private:
|
||||
DEF_OP(VSMull);
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VTBL1);
|
||||
|
||||
///< Encryption ops
|
||||
|
||||
@@ -425,7 +425,7 @@ DEF_OP(StoreFlag) {
|
||||
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
|
||||
}
|
||||
|
||||
Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) const {
|
||||
if (Offset.IsInvalid()) {
|
||||
return Base;
|
||||
} else {
|
||||
|
||||
+15
-13
@@ -12,6 +12,10 @@ static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::D("Value: 0x%lx", Value);
|
||||
}
|
||||
|
||||
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::D("Value: 0x%016lx'%016lx", ValueUpper, Value);
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(FEXCore::IR::IROp_Header *IROp, uint32_t Node)
|
||||
|
||||
DEF_OP(Fence) {
|
||||
@@ -119,24 +123,22 @@ DEF_OP(SetRoundingMode) {
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
push(Reg);
|
||||
PushRegs();
|
||||
if (IsGPR(Op->Header.Args[0].ID())) {
|
||||
mov (rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
mov(rax, reinterpret_cast<uintptr_t>(PrintValue));
|
||||
}
|
||||
else {
|
||||
pextrq(rdi, GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
pextrq(rsi, GetSrc(Op->Header.Args[0].ID()), 1);
|
||||
|
||||
mov (rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
|
||||
mov(rax, reinterpret_cast<uintptr_t>(PrintValue));
|
||||
mov(rax, reinterpret_cast<uintptr_t>(PrintVectorValue));
|
||||
}
|
||||
|
||||
call(rax);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
PopRegs();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+265
-3
@@ -139,6 +139,14 @@ DEF_OP(VAnd) {
|
||||
vpand(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
}
|
||||
|
||||
DEF_OP(VBic) {
|
||||
auto Op = IROp->C<IR::IROp_VBic>();
|
||||
// This doesn't map directly to ARM
|
||||
vpcmpeqd(xmm15, xmm15, xmm15);
|
||||
vpxor(xmm15, GetSrc(Op->Header.Args[1].ID()), xmm15);
|
||||
vpand(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), xmm15);
|
||||
}
|
||||
|
||||
DEF_OP(VOr) {
|
||||
auto Op = IROp->C<IR::IROp_VOr>();
|
||||
vpor(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
@@ -355,6 +363,23 @@ DEF_OP(VAddV) {
|
||||
movaps(Dest, xmm15);
|
||||
}
|
||||
|
||||
DEF_OP(VUMinV) {
|
||||
auto Op = IROp->C<IR::IROp_VUMinV>();
|
||||
|
||||
auto Src = GetSrc(Op->Header.Args[0].ID());
|
||||
auto Dest = GetDst(Node);
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 2: {
|
||||
phminposuw(Dest, Src);
|
||||
// Extract the upper bits which are zero, overwriting position
|
||||
pextrw(eax, Dest, 2);
|
||||
pinsrw(Dest, eax, 1);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VURAvg) {
|
||||
auto Op = IROp->C<IR::IROp_VURAvg>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
@@ -393,6 +418,32 @@ DEF_OP(VAbs) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VPopcount) {
|
||||
auto Op = IROp->C<IR::IROp_VPopcount>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
// This only supports 8bit popcount on 8byte to 16byte registers
|
||||
|
||||
auto Src = GetSrc(Op->Header.Args[0].ID());
|
||||
auto Dest = GetDst(Node);
|
||||
vpxor(xmm15, xmm15, xmm15);
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
// This is disgustingly bad on x86-64 but we only need it for compatibility
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
for (size_t i = 0; i < Elements; ++i) {
|
||||
pextrb(eax, Src, i);
|
||||
popcnt(eax, eax);
|
||||
pinsrb(xmm15, eax, i);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
|
||||
movaps(Dest, xmm15);
|
||||
}
|
||||
|
||||
DEF_OP(VFAdd) {
|
||||
auto Op = IROp->C<IR::IROp_VFAdd>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -933,6 +984,112 @@ DEF_OP(VZip2) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUnZip) {
|
||||
auto Op = IROp->C<IR::IROp_VUnZip>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
if (OpSize == 8) {
|
||||
LOGMAN_MSG_A("Unsupported registersize on VunZip");
|
||||
}
|
||||
else {
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
// Shuffle low bits
|
||||
mov(rax, 0x0E'0C'0A'08'06'04'02'00); // Lower
|
||||
mov(rcx, 0x80'80'80'80'80'80'80'80); // Upper
|
||||
vmovq(xmm15, rax);
|
||||
pinsrq(xmm15, rcx, 1);
|
||||
vpshufb(xmm14, GetSrc(Op->Header.Args[0].ID()), xmm15);
|
||||
vpshufb(xmm13, GetSrc(Op->Header.Args[1].ID()), xmm15);
|
||||
// movlhps back to combine
|
||||
vmovlhps(GetDst(Node), xmm14, xmm13);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
// Shuffle low bits
|
||||
mov(rax, 0x0D'0C'09'08'05'04'01'00); // Lower
|
||||
mov(rcx, 0x80'80'80'80'80'80'80'80); // Upper
|
||||
vmovq(xmm15, rax);
|
||||
pinsrq(xmm15, rcx, 1);
|
||||
vpshufb(xmm14, GetSrc(Op->Header.Args[0].ID()), xmm15);
|
||||
vpshufb(xmm13, GetSrc(Op->Header.Args[1].ID()), xmm15);
|
||||
// movlhps back to combine
|
||||
vmovlhps(GetDst(Node), xmm14, xmm13);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
vshufps(GetDst(Node),
|
||||
GetSrc(Op->Header.Args[0].ID()),
|
||||
GetSrc(Op->Header.Args[1].ID()),
|
||||
0b10'00'10'00);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
vshufpd(GetDst(Node),
|
||||
GetSrc(Op->Header.Args[0].ID()),
|
||||
GetSrc(Op->Header.Args[1].ID()),
|
||||
0b0'0);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUnZip2) {
|
||||
auto Op = IROp->C<IR::IROp_VUnZip2>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
|
||||
if (OpSize == 8) {
|
||||
LOGMAN_MSG_A("Unsupported registersize on VunZip");
|
||||
}
|
||||
else {
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
// Shuffle low bits
|
||||
mov(rax, 0x0F'0D'0B'09'07'05'03'01); // Lower
|
||||
mov(rcx, 0x80'80'80'80'80'80'80'80); // Upper
|
||||
vmovq(xmm15, rax);
|
||||
pinsrq(xmm15, rcx, 1);
|
||||
vpshufb(xmm14, GetSrc(Op->Header.Args[0].ID()), xmm15);
|
||||
vpshufb(xmm13, GetSrc(Op->Header.Args[1].ID()), xmm15);
|
||||
// movlhps back to combine
|
||||
vmovlhps(GetDst(Node), xmm14, xmm13);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
// Shuffle low bits
|
||||
mov(rax, 0x0F'0E'0B'0A'07'06'03'02); // Lower
|
||||
mov(rcx, 0x80'80'80'80'80'80'80'80); // Upper
|
||||
vmovq(xmm15, rax);
|
||||
pinsrq(xmm15, rcx, 1);
|
||||
vpshufb(xmm14, GetSrc(Op->Header.Args[0].ID()), xmm15);
|
||||
vpshufb(xmm13, GetSrc(Op->Header.Args[1].ID()), xmm15);
|
||||
// movlhps back to combine
|
||||
vmovlhps(GetDst(Node), xmm14, xmm13);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
vshufps(GetDst(Node),
|
||||
GetSrc(Op->Header.Args[0].ID()),
|
||||
GetSrc(Op->Header.Args[1].ID()),
|
||||
0b11'01'11'01);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
vshufpd(GetDst(Node),
|
||||
GetSrc(Op->Header.Args[0].ID()),
|
||||
GetSrc(Op->Header.Args[1].ID()),
|
||||
0b1'1);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(VBSL) {
|
||||
auto Op = IROp->C<IR::IROp_VBSL>();
|
||||
vpand(xmm0, GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
@@ -1407,6 +1564,61 @@ DEF_OP(VExtractElement) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VDupElement) {
|
||||
auto Op = IROp->C<IR::IROp_VDupElement>();
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
// First extract the index
|
||||
pextrb(eax, GetSrc(Op->Header.Args[0].ID()), Op->Index);
|
||||
// Insert it in to the first element of the destination
|
||||
pinsrb(GetDst(Node), eax, 0);
|
||||
pinsrb(GetDst(Node), eax, 1);
|
||||
// Shuffle low elements
|
||||
vpshuflw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
// Insert element in to the first upper 64bit element
|
||||
pinsrb(GetDst(Node), eax, 8);
|
||||
pinsrb(GetDst(Node), eax, 9);
|
||||
// Shuffle high elements
|
||||
vpshufhw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
// First extract the index
|
||||
pextrw(eax, GetSrc(Op->Header.Args[0].ID()), Op->Index);
|
||||
// Insert it in to the first element of the destination
|
||||
pinsrw(GetDst(Node), eax, 0);
|
||||
// Shuffle low elements
|
||||
vpshuflw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
// Insert element in to the first upper 64bit element
|
||||
pinsrw(GetDst(Node), eax, 4);
|
||||
// Shuffle high elements
|
||||
vpshufhw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
vpshufd(GetDst(Node),
|
||||
GetSrc(Op->Header.Args[0].ID()),
|
||||
(Op->Index << 0) |
|
||||
(Op->Index << 2) |
|
||||
(Op->Index << 4) |
|
||||
(Op->Index << 6));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
vshufpd(GetDst(Node),
|
||||
GetSrc(Op->Header.Args[0].ID()),
|
||||
GetSrc(Op->Header.Args[0].ID()),
|
||||
(Op->Index << 0) |
|
||||
(Op->Index << 1));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
DEF_OP(VExtr) {
|
||||
auto Op = IROp->C<IR::IROp_VExtr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -1460,14 +1672,36 @@ DEF_OP(VUShrI) {
|
||||
|
||||
DEF_OP(VSShrI) {
|
||||
auto Op = IROp->C<IR::IROp_VSShrI>();
|
||||
movapd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
auto Dest = GetDst(Node);
|
||||
movapd(Dest, GetSrc(Op->Header.Args[0].ID()));
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
// This isn't a native instruction on x86
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
for (int i = 0; i < Elements; ++i) {
|
||||
pextrb(eax, Dest, i);
|
||||
movsx(eax, al);
|
||||
sar(al, Op->BitShift);
|
||||
pinsrb(Dest, eax, i);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
psraw(GetDst(Node), Op->BitShift);
|
||||
psraw(Dest, Op->BitShift);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
psrad(GetDst(Node), Op->BitShift);
|
||||
psrad(Dest, Op->BitShift);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
// This isn't a native instruction on x86
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
pextrq(rax, Dest, i);
|
||||
sar(rax, Op->BitShift);
|
||||
pinsrq(Dest, rax, i);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
@@ -1872,6 +2106,27 @@ DEF_OP(VSMull2) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL) {
|
||||
auto Op = IROp->C<IR::IROp_VUABDL>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 2: {
|
||||
pmovzxbw(xmm14, GetSrc(Op->Header.Args[0].ID()));
|
||||
pmovzxbw(xmm15, GetSrc(Op->Header.Args[1].ID()));
|
||||
vpsubw(GetDst(Node), xmm14, xmm15);
|
||||
vpabsw(GetDst(Node), GetDst(Node));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pmovzxwd(xmm14, GetSrc(Op->Header.Args[0].ID()));
|
||||
pmovzxwd(xmm15, GetSrc(Op->Header.Args[1].ID()));
|
||||
vpsubd(GetDst(Node), xmm14, xmm15);
|
||||
vpabsd(GetDst(Node), GetDst(Node));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A("Unknown Element Size: %d", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -1901,6 +2156,7 @@ void X86JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(SPLATVECTOR4, SplatVector);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
REGISTER_OP(VAND, VAnd);
|
||||
REGISTER_OP(VBIC, VBic);
|
||||
REGISTER_OP(VOR, VOr);
|
||||
REGISTER_OP(VXOR, VXor);
|
||||
REGISTER_OP(VADD, VAdd);
|
||||
@@ -1911,8 +2167,10 @@ void X86JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VSQSUB, VSQSub);
|
||||
REGISTER_OP(VADDP, VAddP);
|
||||
REGISTER_OP(VADDV, VAddV);
|
||||
REGISTER_OP(VUMINV, VUMinV);
|
||||
REGISTER_OP(VURAVG, VURAvg);
|
||||
REGISTER_OP(VABS, VAbs);
|
||||
REGISTER_OP(VPOPCOUNT, VPopcount);
|
||||
REGISTER_OP(VFADD, VFAdd);
|
||||
REGISTER_OP(VFADDP, VFAddP);
|
||||
REGISTER_OP(VFSUB, VFSub);
|
||||
@@ -1932,6 +2190,8 @@ void X86JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VSMAX, VSMax);
|
||||
REGISTER_OP(VZIP, VZip);
|
||||
REGISTER_OP(VZIP2, VZip2);
|
||||
REGISTER_OP(VUNZIP, VUnZip);
|
||||
REGISTER_OP(VUNZIP2, VUnZip2);
|
||||
REGISTER_OP(VBSL, VBSL);
|
||||
REGISTER_OP(VCMPEQ, VCMPEQ);
|
||||
REGISTER_OP(VCMPEQZ, VCMPEQZ);
|
||||
@@ -1954,6 +2214,7 @@ void X86JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VINSSCALARELEMENT, VInsScalarElement);
|
||||
REGISTER_OP(VEXTRACTELEMENT, VExtractElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VSLI, VSLI);
|
||||
REGISTER_OP(VSRI, VSRI);
|
||||
@@ -1977,6 +2238,7 @@ void X86JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VSMULL, VSMull);
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
+6
-3
@@ -38,7 +38,10 @@ public:
|
||||
std::map<uint64_t, std::vector<uint64_t>> CodePages;
|
||||
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode, uint64_t Start, uint64_t Length) {
|
||||
auto InsertPoint = BlockList.emplace(Address, (uintptr_t)HostCode);
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto InsertPoint =
|
||||
#endif
|
||||
BlockList.emplace(Address, (uintptr_t)HostCode);
|
||||
LOGMAN_THROW_A(InsertPoint.second == true, "Dupplicate block mapping added");
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
@@ -98,8 +101,8 @@ public:
|
||||
|
||||
void HintUsedRange(uint64_t Address, uint64_t Size);
|
||||
|
||||
uintptr_t GetL1Pointer() { return L1Pointer; }
|
||||
uintptr_t GetPagePointer() { return PagePointer; }
|
||||
uintptr_t GetL1Pointer() const { return L1Pointer; }
|
||||
uintptr_t GetPagePointer() const { return PagePointer; }
|
||||
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
|
||||
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
|
||||
+842
-563
File diff suppressed because it is too large.
Load diff
+31
-28
@@ -85,7 +85,7 @@ public:
|
||||
auto it = JumpTargets.find(NextRIP);
|
||||
if (it == JumpTargets.end()) {
|
||||
|
||||
uint8_t GPRSize = CTX->Config.Is64BitMode ? 8 : 4;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
// If we don't have a jump target to a new block then we have to leave
|
||||
// Set the RIP to the next instruction and leave
|
||||
auto RelocatedNextRIP = _EntrypointOffset(NextRIP - Entry, GPRSize);
|
||||
@@ -105,7 +105,7 @@ public:
|
||||
|
||||
void ResetWorkingList();
|
||||
void ResetDecodeFailure() { DecodeFailure = false; }
|
||||
bool HadDecodeFailure() { return DecodeFailure; }
|
||||
bool HadDecodeFailure() const { return DecodeFailure; }
|
||||
|
||||
void BeginFunction(uint64_t RIP, std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
void Finalize();
|
||||
@@ -260,12 +260,6 @@ public:
|
||||
template<size_t ElementSize>
|
||||
void PSUBQOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PMINUOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PMAXUOp(OpcodeArgs);
|
||||
void PMINSWOp(OpcodeArgs);
|
||||
void PMAXSWOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void MOVMSKOp(OpcodeArgs);
|
||||
void MOVMSKOpOne(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
@@ -275,10 +269,6 @@ public:
|
||||
void PSHUFBOp(OpcodeArgs);
|
||||
template<size_t ElementSize, bool HalfSize, bool Low>
|
||||
void PSHUFDOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PCMPEQOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PCMPGTOp(OpcodeArgs);
|
||||
void MOVDOp(OpcodeArgs);
|
||||
template<size_t ElementSize, bool Scalar, uint32_t SrcIndex>
|
||||
void PSRLDOp(OpcodeArgs);
|
||||
@@ -297,21 +287,21 @@ public:
|
||||
template<size_t ElementSize>
|
||||
void PAVGOp(OpcodeArgs);
|
||||
void MOVDDUPOp(OpcodeArgs);
|
||||
template<size_t DstElementSize, bool Signed>
|
||||
template<size_t DstElementSize>
|
||||
void CVTGPR_To_FPR(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Signed, bool HostRoundingMode>
|
||||
template<size_t SrcElementSize, bool HostRoundingMode>
|
||||
void CVTFPR_To_GPR(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Signed, bool Widen>
|
||||
template<size_t SrcElementSize, bool Widen>
|
||||
void Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void Scalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void Vector_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Signed, bool Narrow, bool HostRoundingMode>
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Signed, bool Widen>
|
||||
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Signed, bool Narrow, bool HostRoundingMode>
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
void MASKMOVOp(OpcodeArgs);
|
||||
void MOVBetweenGPR_FPR(OpcodeArgs);
|
||||
@@ -325,17 +315,13 @@ public:
|
||||
void ANDNOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PINSROp(OpcodeArgs);
|
||||
void InsertPSOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void PExtrOp(OpcodeArgs);
|
||||
template<size_t ElementSize, bool Signed>
|
||||
void PMULOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void PSIGN(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void PABS(OpcodeArgs);
|
||||
|
||||
// X87 Ops
|
||||
template<size_t width>
|
||||
void FLD(OpcodeArgs);
|
||||
@@ -470,6 +456,23 @@ public:
|
||||
void AESDecLastOp(OpcodeArgs);
|
||||
void AESKeyGenAssist(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, size_t DstElementSize, bool Signed>
|
||||
void ExtendVectorElements(OpcodeArgs);
|
||||
template<size_t ElementSize, bool Scalar>
|
||||
void VectorRound(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VectorBlend(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VectorVariableBlend(OpcodeArgs);
|
||||
void PTestOp(OpcodeArgs);
|
||||
void PHMINPOSUWOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void DPPOp(OpcodeArgs);
|
||||
|
||||
void MPSADBWOp(OpcodeArgs);
|
||||
|
||||
void UnimplementedOp(OpcodeArgs);
|
||||
|
||||
#undef OpcodeArgs
|
||||
@@ -494,8 +497,8 @@ private:
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align);
|
||||
|
||||
uint8_t GetDstSize(FEXCore::X86Tables::DecodedOp Op);
|
||||
uint8_t GetSrcSize(FEXCore::X86Tables::DecodedOp Op);
|
||||
uint8_t GetDstSize(FEXCore::X86Tables::DecodedOp Op) const;
|
||||
uint8_t GetSrcSize(FEXCore::X86Tables::DecodedOp Op) const;
|
||||
|
||||
template<unsigned BitOffset>
|
||||
void SetRFLAG(OrderedNode *Value);
|
||||
@@ -525,12 +528,12 @@ private:
|
||||
OrderedNode * GetX87Top();
|
||||
void SetX87Top(OrderedNode *Value);
|
||||
|
||||
bool DestIsLockedMem(FEXCore::X86Tables::DecodedOp Op) {
|
||||
return Op->Dest.TypeNone.Type !=FEXCore::X86Tables::DecodedOperand::TYPE_GPR && (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK);
|
||||
bool DestIsLockedMem(FEXCore::X86Tables::DecodedOp Op) const {
|
||||
return DestIsMem(Op) && (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0;
|
||||
}
|
||||
|
||||
bool DestIsMem(FEXCore::X86Tables::DecodedOp Op) {
|
||||
return Op->Dest.TypeNone.Type !=FEXCore::X86Tables::DecodedOperand::TYPE_GPR;
|
||||
bool DestIsMem(FEXCore::X86Tables::DecodedOp Op) const {
|
||||
return !Op->Dest.IsGPR();
|
||||
}
|
||||
|
||||
void CreateJumpBlocks(std::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
|
||||
@@ -41,10 +41,10 @@ void InitializeH0F38Tables() {
|
||||
{OPD(PF_38_NONE, 0x0B), 1, X86InstInfo{"PMULHRSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x0B), 1, X86InstInfo{"PMULHRSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_66, 0x10), 1, X86InstInfo{"PBLENDVB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x14), 1, X86InstInfo{"BLENDVPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x15), 1, X86InstInfo{"BLENDVPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x17), 1, X86InstInfo{"PTEST", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x10), 1, X86InstInfo{"PBLENDVB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x14), 1, X86InstInfo{"BLENDVPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x15), 1, X86InstInfo{"BLENDVPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x17), 1, X86InstInfo{"PTEST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x1C), 1, X86InstInfo{"PABSB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x1C), 1, X86InstInfo{"PABSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x1D), 1, X86InstInfo{"PABSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
@@ -52,34 +52,34 @@ void InitializeH0F38Tables() {
|
||||
{OPD(PF_38_NONE, 0x1E), 1, X86InstInfo{"PABSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x1E), 1, X86InstInfo{"PABSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_66, 0x20), 1, X86InstInfo{"PMOVSXBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x21), 1, X86InstInfo{"PMOVSXBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x22), 1, X86InstInfo{"PMOVSXBQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x23), 1, X86InstInfo{"PMOVSXWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x24), 1, X86InstInfo{"PMOVSXWQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x25), 1, X86InstInfo{"PMOVSXDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x28), 1, X86InstInfo{"PMULDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x29), 1, X86InstInfo{"PCMPEQQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x20), 1, X86InstInfo{"PMOVSXBW", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x21), 1, X86InstInfo{"PMOVSXBD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x22), 1, X86InstInfo{"PMOVSXBQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x23), 1, X86InstInfo{"PMOVSXWD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x24), 1, X86InstInfo{"PMOVSXWQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x25), 1, X86InstInfo{"PMOVSXDQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x28), 1, X86InstInfo{"PMULDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x29), 1, X86InstInfo{"PCMPEQQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x2A), 1, X86InstInfo{"MOVNTDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x2B), 1, X86InstInfo{"PACKUSDW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x2B), 1, X86InstInfo{"PACKUSDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_66, 0x30), 1, X86InstInfo{"PMOVZXBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x31), 1, X86InstInfo{"PMOVZXBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x32), 1, X86InstInfo{"PMOVZXBQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x33), 1, X86InstInfo{"PMOVZXWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x34), 1, X86InstInfo{"PMOVZXWQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x35), 1, X86InstInfo{"PMOVZXDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x38), 1, X86InstInfo{"PMINSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x39), 1, X86InstInfo{"PMINSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3A), 1, X86InstInfo{"PMINUW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3B), 1, X86InstInfo{"PMINUD", TYPE_INST, FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3C), 1, X86InstInfo{"PMAXSB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3D), 1, X86InstInfo{"PMAXSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3E), 1, X86InstInfo{"PMAXUW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3F), 1, X86InstInfo{"PMAXUD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x30), 1, X86InstInfo{"PMOVZXBW", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x31), 1, X86InstInfo{"PMOVZXBD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x32), 1, X86InstInfo{"PMOVZXBQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x33), 1, X86InstInfo{"PMOVZXWD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x34), 1, X86InstInfo{"PMOVZXWQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x35), 1, X86InstInfo{"PMOVZXDQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x38), 1, X86InstInfo{"PMINSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x39), 1, X86InstInfo{"PMINSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3A), 1, X86InstInfo{"PMINUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3B), 1, X86InstInfo{"PMINUD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3C), 1, X86InstInfo{"PMAXSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3D), 1, X86InstInfo{"PMAXSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3E), 1, X86InstInfo{"PMAXUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3F), 1, X86InstInfo{"PMAXUD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_66, 0x40), 1, X86InstInfo{"PMULLD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x41), 1, X86InstInfo{"PHMINPOSUW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x40), 1, X86InstInfo{"PMULLD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x41), 1, X86InstInfo{"PHMINPOSUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_66, 0xDB), 1, X86InstInfo{"AESIMC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xDC), 1, X86InstInfo{"AESENC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
@@ -16,26 +16,26 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
|
||||
const U16U8InfoStruct H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_8BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_8BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
+168
-99
@@ -60,6 +60,12 @@
|
||||
"constexpr static uint8_t ROUND_MODE_TOWARDS_ZERO = 3",
|
||||
"constexpr static uint8_t ROUND_MODE_FLUSH_TO_ZERO = 1 << 2",
|
||||
|
||||
"static constexpr FEXCore::IR::RoundType Round_Nearest {ROUND_MODE_NEAREST}",
|
||||
"static constexpr FEXCore::IR::RoundType Round_Negative_Infinity {ROUND_MODE_NEGATIVE_INFINITY}",
|
||||
"static constexpr FEXCore::IR::RoundType Round_Positive_Infinity {ROUND_MODE_POSITIVE_INFINITY}",
|
||||
"static constexpr FEXCore::IR::RoundType Round_Towards_Zero {ROUND_MODE_TOWARDS_ZERO} /* Truncate */",
|
||||
"static constexpr FEXCore::IR::RoundType Round_Host {ROUND_MODE_TOWARDS_ZERO + 1}",
|
||||
|
||||
"constexpr static FEXCore::IR::MemOffsetType MEM_OFFSET_SXTX {0};",
|
||||
"constexpr static FEXCore::IR::MemOffsetType MEM_OFFSET_UXTW {1};",
|
||||
"constexpr static FEXCore::IR::MemOffsetType MEM_OFFSET_SXTW {2};"
|
||||
@@ -1477,24 +1483,6 @@
|
||||
]
|
||||
},
|
||||
|
||||
"Float_ToGPR_U": {
|
||||
"Desc": ["Moves the scalar element to a GPR with conversion",
|
||||
"Converts the 32bit or 64bit float to an unsigned integer",
|
||||
"Rounding mode determined by host flag's rounding mode"
|
||||
],
|
||||
"OpClass": "ALU",
|
||||
"HasDest": true,
|
||||
"DestClass": "GPR",
|
||||
"DestSize": "ElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Scalar"
|
||||
],
|
||||
"Args": [
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"Float_ToGPR_S": {
|
||||
"Desc": ["Moves the scalar element to a GPR with conversion",
|
||||
"Converts the 32bit or 64bit float to an signed integer",
|
||||
@@ -1503,30 +1491,16 @@
|
||||
"OpClass": "ALU",
|
||||
"HasDest": true,
|
||||
"DestClass": "GPR",
|
||||
"DestSize": "ElementSize",
|
||||
"DestSize": "DestElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Scalar"
|
||||
],
|
||||
"Args": [
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"Float_ToGPR_ZU": {
|
||||
"Desc": ["Moves the scalar element to a GPR with conversion",
|
||||
"Converts the 32bit or 64bit float to an unsigned integer rounding towards zero (Truncating)"
|
||||
],
|
||||
"OpClass": "ALU",
|
||||
"HasDest": true,
|
||||
"DestClass": "GPR",
|
||||
"DestSize": "ElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Scalar"
|
||||
"HelperArgs": [
|
||||
"uint8_t", "DestElementSize"
|
||||
],
|
||||
"Args": [
|
||||
"uint8_t", "ElementSize"
|
||||
"uint8_t", "SrcElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -1537,13 +1511,16 @@
|
||||
"OpClass": "ALU",
|
||||
"HasDest": true,
|
||||
"DestClass": "GPR",
|
||||
"DestSize": "ElementSize",
|
||||
"DestSize": "DestElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Scalar"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "DestElementSize"
|
||||
],
|
||||
"Args": [
|
||||
"uint8_t", "ElementSize"
|
||||
"uint8_t", "SrcElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -1664,6 +1641,23 @@
|
||||
]
|
||||
},
|
||||
|
||||
"VBic": {
|
||||
"OpClass": "Vector",
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "2",
|
||||
"SSANames": [
|
||||
"Vector1",
|
||||
"Vector2"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"VOr": {
|
||||
"OpClass": "Vector",
|
||||
"HasDest": true,
|
||||
@@ -1809,8 +1803,8 @@
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "2",
|
||||
"SSANames": [
|
||||
"Vector1",
|
||||
"Vector2"
|
||||
"VectorLower",
|
||||
"VectorUpper"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
@@ -1837,6 +1831,25 @@
|
||||
]
|
||||
},
|
||||
|
||||
"VUMinV": {
|
||||
"OpClass": "Vector",
|
||||
"Desc": ["Does a horizontal vector unsigned minimum of elements across the source vector",
|
||||
"Result is a zero extended scalar"
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Vector"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"VURAvg": {
|
||||
"OpClass": "Vector",
|
||||
"Desc": ["Does an unsigned rounded average", "dst_elem = (src1_elem + src2_elem + 1) >> 1"],
|
||||
@@ -1873,6 +1886,24 @@
|
||||
]
|
||||
},
|
||||
|
||||
"VPopcount": {
|
||||
"OpClass": "Vector",
|
||||
"Desc": ["Does a popcount for each element of the register"
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Vector"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"VFAdd": {
|
||||
"OpClass": "Vector",
|
||||
"HasDest": true,
|
||||
@@ -1899,8 +1930,8 @@
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "2",
|
||||
"SSANames": [
|
||||
"Vector1",
|
||||
"Vector2"
|
||||
"VectorLow",
|
||||
"VectorHigh"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
@@ -2192,6 +2223,40 @@
|
||||
]
|
||||
},
|
||||
|
||||
"VUnZip": {
|
||||
"OpClass": "Vector",
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "2",
|
||||
"SSANames": [
|
||||
"Lower",
|
||||
"Upper"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"VUnZip2": {
|
||||
"OpClass": "Vector",
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "2",
|
||||
"SSANames": [
|
||||
"Lower",
|
||||
"Upper"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"VBSL": {
|
||||
"Desc": ["Does a vector bitwise select.",
|
||||
"If the bit in the field is 1 then the corresponding bit is pulled from VectorTrue",
|
||||
@@ -2583,6 +2648,26 @@
|
||||
]
|
||||
},
|
||||
|
||||
"VDupElement": {
|
||||
"Desc": ["Duplicates one element from the source register across the whole register"],
|
||||
"OpClass": "Vector",
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Vector"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
],
|
||||
"Args": [
|
||||
"uint8_t", "Index"
|
||||
]
|
||||
},
|
||||
|
||||
"VExtr": {
|
||||
"Desc": ["Concats two vector registers together and extracts a full width register from the element index",
|
||||
"Index is an element index. So it is offset by ElementSize argument",
|
||||
@@ -2935,27 +3020,6 @@
|
||||
]
|
||||
},
|
||||
|
||||
"Float_FromGPR_U": {
|
||||
"OpClass": "Conv",
|
||||
"Desc": ["Scalar op: Converts unsigned GPR to Scalar float",
|
||||
"Zeroes the upper bits of the vector register"
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "DstElementSize",
|
||||
"NumElements": "1",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"GPR"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "DstElementSize"
|
||||
],
|
||||
"Args": [
|
||||
"uint8_t", "SrcElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"Float_FromGPR_S": {
|
||||
"OpClass": "Conv",
|
||||
"Desc": ["Scalar op: Converts signed GPR to Scalar float",
|
||||
@@ -3032,25 +3096,6 @@
|
||||
]
|
||||
},
|
||||
|
||||
"Vector_FToU": {
|
||||
"OpClass": "Conv",
|
||||
"Desc": ["Vector op: Converts float to unsigned integer",
|
||||
"Rounding mode determined by host rounding mode"
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Vector"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"Vector_FToS": {
|
||||
"OpClass": "Conv",
|
||||
"Desc": ["Vector op: Converts float to signed integer, rounding towards zero",
|
||||
@@ -3070,23 +3115,6 @@
|
||||
]
|
||||
},
|
||||
|
||||
"Vector_FToZU": {
|
||||
"OpClass": "Conv",
|
||||
"Desc": "Vector op: Converts float to unsigned integer, rounding towards zero",
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Vector"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"Vector_FToZS": {
|
||||
"OpClass": "Conv",
|
||||
"Desc": "Vector op: Converts float to signed integer, rounding towards zero",
|
||||
@@ -3124,6 +3152,28 @@
|
||||
]
|
||||
},
|
||||
|
||||
"Vector_FToI": {
|
||||
"OpClass": "Conv",
|
||||
"Desc": ["Vector op: Rounds float to integral",
|
||||
"Rounding mode determined by argument"
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Vector"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
],
|
||||
"Args":[
|
||||
"FEXCore::IR::RoundType", "Round"
|
||||
]
|
||||
},
|
||||
|
||||
"VUMul": {
|
||||
"OpClass": "Vector",
|
||||
"HasDest": true,
|
||||
@@ -3231,6 +3281,25 @@
|
||||
]
|
||||
},
|
||||
|
||||
"VUABDL": {
|
||||
"OpClass": "Vector",
|
||||
"Desc": ["Unsigned Absolute Difference Long"
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestClass": "FPR",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)",
|
||||
"SSAArgs": "2",
|
||||
"SSANames": [
|
||||
"Vector1",
|
||||
"Vector2"
|
||||
],
|
||||
"HelperArgs": [
|
||||
"uint8_t", "RegisterSize",
|
||||
"uint8_t", "ElementSize"
|
||||
]
|
||||
},
|
||||
|
||||
"VTBL1": {
|
||||
"Desc": ["Does a vector table lookup from one register in to the destination",
|
||||
"Lookup is byte sized per byte element.",
|
||||
|
||||
+13
-2
@@ -37,7 +37,7 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const*
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, CondClassType Arg) {
|
||||
std::array<std::string, 22> CondNames = {
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
"UGE",
|
||||
@@ -66,7 +66,7 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const*
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, MemOffsetType Arg) {
|
||||
std::array<std::string, 3> Names = {
|
||||
static constexpr std::array<std::string_view, 3> Names = {
|
||||
"SXTX",
|
||||
"UXTW",
|
||||
"SXTW",
|
||||
@@ -154,6 +154,17 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const*
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::RoundType Arg) {
|
||||
switch (Arg) {
|
||||
case FEXCore::IR::Round_Nearest: *out << "Nearest"; break;
|
||||
case FEXCore::IR::Round_Negative_Infinity: *out << "-Inf"; break;
|
||||
case FEXCore::IR::Round_Positive_Infinity: *out << "+Inf"; break;
|
||||
case FEXCore::IR::Round_Towards_Zero: *out << "Towards Zero"; break;
|
||||
case FEXCore::IR::Round_Host: *out << "Host"; break;
|
||||
default: *out << "<Unknown Round Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
|
||||
+106
-108
@@ -66,7 +66,8 @@ std::string DecodeErrorToString(DecodeFailure Failure) {
|
||||
case DecodeFailure::DECODE_INVALID_CONDFLAG: return "Invalid Conditional name";
|
||||
case DecodeFailure::DECODE_INVALID_MEMOFFSETTYPE: return "Invalid Memory Offset Type";
|
||||
case DecodeFailure::DECODE_INVALID_FENCETYPE: return "Invalid Fence Type";
|
||||
};
|
||||
}
|
||||
return "Unknown Error";
|
||||
}
|
||||
|
||||
std::unordered_map<std::string_view, FEXCore::IR::IROps> NameToOpMap;
|
||||
@@ -74,22 +75,22 @@ std::unordered_map<std::string_view, FEXCore::IR::IROps> NameToOpMap;
|
||||
class IRParser: public FEXCore::IR::IREmitter {
|
||||
public:
|
||||
template<typename Type>
|
||||
std::pair<DecodeFailure, Type> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, Type> DecodeValue(const std::string &Arg) {
|
||||
return {DecodeFailure::DECODE_UNKNOWN_TYPE, {}};
|
||||
}
|
||||
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint8_t> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, uint8_t> DecodeValue(const std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint8_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, bool> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, bool> DecodeValue(const std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint8_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
@@ -98,7 +99,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint16_t> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, uint16_t> DecodeValue(const std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint16_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
@@ -107,7 +108,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint32_t> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, uint32_t> DecodeValue(const std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint32_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
@@ -116,7 +117,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint64_t> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, uint64_t> DecodeValue(const std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
uint64_t Result = strtoull(&Arg.at(1), nullptr, 0);
|
||||
@@ -125,7 +126,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, int64_t> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, int64_t> DecodeValue(const std::string &Arg) {
|
||||
if (Arg.at(0) != '#') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
int64_t Result = (int64_t)strtoull(&Arg.at(1), nullptr, 0);
|
||||
@@ -134,7 +135,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, IR::SHA256Sum> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, IR::SHA256Sum> DecodeValue(const std::string &Arg) {
|
||||
IR::SHA256Sum Result;
|
||||
|
||||
if (Arg.at(0) != 's' || Arg.at(1) != 'h' || Arg.at(2) != 'a' || Arg.at(3) != '2' || Arg.at(4) != '5' || Arg.at(5) != '6' || Arg.at(6) != ':')
|
||||
@@ -165,7 +166,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::RegisterClassType> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, FEXCore::IR::RegisterClassType> DecodeValue(const std::string &Arg) {
|
||||
if (Arg == "GPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRClass};
|
||||
}
|
||||
@@ -183,7 +184,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::TypeDefinition> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, FEXCore::IR::TypeDefinition> DecodeValue(const std::string &Arg) {
|
||||
uint8_t Size{}, Elements{1};
|
||||
int NumArgs = sscanf(Arg.c_str(), "i%hhdv%hhd", &Size, &Elements);
|
||||
|
||||
@@ -195,8 +196,8 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::CondClassType> DecodeValue(std::string &Arg) {
|
||||
std::array<std::string, 22> CondNames = {
|
||||
std::pair<DecodeFailure, FEXCore::IR::CondClassType> DecodeValue(const std::string &Arg) {
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {
|
||||
"EQ",
|
||||
"NEQ",
|
||||
"UGE",
|
||||
@@ -230,8 +231,8 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::MemOffsetType> DecodeValue(std::string &Arg) {
|
||||
std::array<std::string, 3> Names = {
|
||||
std::pair<DecodeFailure, FEXCore::IR::MemOffsetType> DecodeValue(const std::string &Arg) {
|
||||
static constexpr std::array<std::string_view, 3> Names = {
|
||||
"SXTX",
|
||||
"UXTW",
|
||||
"SXTW",
|
||||
@@ -246,8 +247,8 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::FenceType> DecodeValue(std::string &Arg) {
|
||||
std::array<std::string, 3> Names = {
|
||||
std::pair<DecodeFailure, FEXCore::IR::FenceType> DecodeValue(const std::string &Arg) {
|
||||
static constexpr std::array<std::string_view, 3> Names = {
|
||||
"Loads",
|
||||
"Stores",
|
||||
"LoadStores",
|
||||
@@ -262,23 +263,22 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, OrderedNode*> DecodeValue(std::string &Arg) {
|
||||
std::pair<DecodeFailure, OrderedNode*> DecodeValue(const std::string &Arg) {
|
||||
if (Arg.at(0) != '%') return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
|
||||
// Strip off the type qualifier from the ssa value
|
||||
size_t ArgEnd = std::string::npos;
|
||||
std::string SSAName = trim(Arg);
|
||||
ArgEnd = SSAName.find_first_of(" ");
|
||||
const size_t ArgEnd = SSAName.find_first_of(' ');
|
||||
|
||||
if (ArgEnd != std::string::npos) {
|
||||
SSAName = SSAName.substr(0, ArgEnd);
|
||||
}
|
||||
SSAName = SSAName.substr(0, ArgEnd);
|
||||
}
|
||||
|
||||
// Forward declarations may make this not succed
|
||||
// Forward declarations may make this not succed
|
||||
auto Op = SSANameMapper.find(SSAName);
|
||||
if (Op == SSANameMapper.end()) {
|
||||
if (Op == SSANameMapper.end()) {
|
||||
return {DecodeFailure::DECODE_UNKNOWN_SSA, nullptr};
|
||||
}
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_OKAY, Op->second};
|
||||
}
|
||||
@@ -302,21 +302,21 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
|
||||
IRParser(std::istream *text) {
|
||||
InitializeStaticTables();
|
||||
|
||||
|
||||
std::string TmpLine;
|
||||
while (!text->eof()) {
|
||||
std::getline(*text, TmpLine);
|
||||
if (text->eof()) {
|
||||
break;
|
||||
}
|
||||
if (text->eof()) {
|
||||
break;
|
||||
}
|
||||
if (text->fail()) {
|
||||
LogMan::Msg::E("Failed to getline on line: %ld", Lines.size());
|
||||
LogMan::Msg::EFmt("Failed to getline on line: {}", Lines.size());
|
||||
return;
|
||||
}
|
||||
Lines.emplace_back(TmpLine);
|
||||
}
|
||||
|
||||
ResetWorkingList();
|
||||
ResetWorkingList();
|
||||
Loaded = Parse();
|
||||
}
|
||||
|
||||
@@ -327,11 +327,11 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
|
||||
|
||||
bool Parse() {
|
||||
auto CheckPrintError = [&](LineDefinition &Def, DecodeFailure Failure) -> bool {
|
||||
const auto CheckPrintError = [&](const LineDefinition &Def, DecodeFailure Failure) -> bool {
|
||||
if (Failure != DecodeFailure::DECODE_OKAY) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("Value Couldn't be decoded due to %s", DecodeErrorToString(Failure).c_str());
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("Value Couldn't be decoded due to {}", DecodeErrorToString(Failure));
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -339,13 +339,13 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
};
|
||||
|
||||
// String parse every line for our definitions
|
||||
for (size_t i = 0; i < Lines.size(); ++i) {
|
||||
std::string Line = Lines[i];
|
||||
for (size_t i = 0; i < Lines.size(); ++i) {
|
||||
std::string Line = Lines[i];
|
||||
LineDefinition Def{};
|
||||
CurrentDef = &Def;
|
||||
CurrentDef = &Def;
|
||||
Def.LineNumber = i;
|
||||
|
||||
Line = trim(Line);
|
||||
Line = trim(Line);
|
||||
|
||||
// Skip empty lines
|
||||
if (Line.empty()) {
|
||||
@@ -359,35 +359,37 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
size_t CurrentPos{};
|
||||
// Let's see if this node is assigning something first
|
||||
if (Line[0] == '%') {
|
||||
// Let's see if this node is assigning something first
|
||||
if (Line[0] == '%') {
|
||||
size_t DefinitionEnd = std::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of("=", CurrentPos)) != std::string::npos) {
|
||||
if ((DefinitionEnd = Line.find_first_of('=', CurrentPos)) != std::string::npos) {
|
||||
Def.Definition = Line.substr(0, DefinitionEnd);
|
||||
Def.Definition = trim(Def.Definition);
|
||||
Def.HasDefinition = true;
|
||||
CurrentPos = DefinitionEnd + 1; // +1 to ensure we go past then assignment
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("SSA declaration without assignment");
|
||||
LogMan::Msg::EFmt("Error on Line: {}", i);
|
||||
LogMan::Msg::EFmt("{}", Lines[i]);
|
||||
LogMan::Msg::EFmt("SSA declaration without assignment");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we are pulling in some IR from the IR Printer
|
||||
// Prints (%ssa%d) at the start of lines without a definition
|
||||
if (Line[0] == '(') {
|
||||
size_t DefinitionEnd = std::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of(")", CurrentPos)) != std::string::npos) {
|
||||
if ((DefinitionEnd = Line.find_first_of(')', CurrentPos)) != std::string::npos) {
|
||||
size_t SSAEnd = std::string::npos;
|
||||
if ((SSAEnd = Line.find_last_of(" ", DefinitionEnd)) != std::string::npos) {
|
||||
if ((SSAEnd = Line.find_last_of(' ', DefinitionEnd)) != std::string::npos) {
|
||||
std::string Type = Line.substr(SSAEnd + 1, DefinitionEnd - SSAEnd - 1);
|
||||
Type = trim(Type);
|
||||
|
||||
auto DefinitionSize = DecodeValue<FEXCore::IR::TypeDefinition>(Type);
|
||||
if (!CheckPrintError(Def, DefinitionSize.first)) return false;
|
||||
if (!CheckPrintError(Def, DefinitionSize.first)) {
|
||||
return false;
|
||||
}
|
||||
Def.Size = DefinitionSize.second;
|
||||
}
|
||||
|
||||
@@ -396,9 +398,9 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
CurrentPos = DefinitionEnd + 1;
|
||||
}
|
||||
else {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("SSA value with numbered SSA provided but no closing parentheses");
|
||||
LogMan::Msg::EFmt("Error on Line: {}", i);
|
||||
LogMan::Msg::EFmt("{}", Lines[i]);
|
||||
LogMan::Msg::EFmt("SSA value with numbered SSA provided but no closing parentheses");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -406,7 +408,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
if (Def.HasDefinition) {
|
||||
// Let's check if we have a size declared with this variable
|
||||
size_t NameEnd = std::string::npos;
|
||||
if ((NameEnd = Def.Definition.find_first_of(" ")) != std::string::npos) {
|
||||
if ((NameEnd = Def.Definition.find_first_of(' ')) != std::string::npos) {
|
||||
std::string Type = Def.Definition.substr(NameEnd + 1);
|
||||
Type = trim(Type);
|
||||
Def.Definition = trim(Def.Definition.substr(0, NameEnd));
|
||||
@@ -417,9 +419,9 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
if (Def.Definition == "%Invalid") {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("Definition tried to define reserved %Invalid ssa node");
|
||||
LogMan::Msg::EFmt("Error on Line: {}", i);
|
||||
LogMan::Msg::EFmt("{}", Lines[i]);
|
||||
LogMan::Msg::EFmt("Definition tried to define reserved %Invalid ssa node");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -436,9 +438,9 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
else {
|
||||
if (RemainingLine.empty()) {
|
||||
LogMan::Msg::E("Error on Line: %d", i);
|
||||
LogMan::Msg::E("%s", Lines[i].c_str());
|
||||
LogMan::Msg::E("Line without an IROp?");
|
||||
LogMan::Msg::EFmt("Error on Line: {}", i);
|
||||
LogMan::Msg::EFmt("{}", Lines[i]);
|
||||
LogMan::Msg::EFmt("Line without an IROp?");
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -455,12 +457,10 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
else {
|
||||
while (!RemainingLine.empty()) {
|
||||
size_t ArgEnd = std::string::npos;
|
||||
ArgEnd = RemainingLine.find_first_of(",");
|
||||
const size_t ArgEnd = RemainingLine.find(',');
|
||||
std::string Arg = trim(RemainingLine.substr(0, ArgEnd));
|
||||
|
||||
std::string Arg = RemainingLine.substr(0, ArgEnd);
|
||||
Arg = trim(Arg);
|
||||
Def.Args.emplace_back(Arg);
|
||||
Def.Args.emplace_back(std::move(Arg));
|
||||
|
||||
RemainingLine.erase(0, ArgEnd+1); // +1 to ensure we go past the ','
|
||||
if (ArgEnd == std::string::npos)
|
||||
@@ -469,17 +469,17 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
}
|
||||
|
||||
Defs.emplace_back(Def);
|
||||
}
|
||||
CurrentDef = &Defs.emplace_back(std::move(Def));
|
||||
}
|
||||
|
||||
// Ensure all of the ops are real ops
|
||||
for(size_t i = 0; i < Defs.size(); ++i) {
|
||||
auto &Def = Defs[i];
|
||||
auto Op = NameToOpMap.find(Def.IROp);
|
||||
if (Op == NameToOpMap.end()) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("IROp '%s' doesn't exist", Def.IROp.c_str());
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("IROp '{}' doesn't exist", Def.IROp);
|
||||
return false;
|
||||
}
|
||||
Def.OpEnum = Op->second;
|
||||
@@ -489,11 +489,11 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
IRPair<IROp_IRHeader> IRHeader;
|
||||
{
|
||||
auto &Def = Defs[0];
|
||||
CurrentDef = &Def;
|
||||
CurrentDef = &Def;
|
||||
if (Def.OpEnum != FEXCore::IR::IROps::OP_IRHEADER) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("First op needs to be IRHeader. Was '%s'", Def.IROp.c_str());
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("First op needs to be IRHeader. Was '{}'", Def.IROp);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -507,14 +507,14 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
SetWriteCursor(nullptr); // isolate the header from everything following
|
||||
|
||||
// Initialize SSANameMapper with Invalid value
|
||||
SSANameMapper["%Invalid"] = Invalid();
|
||||
SSANameMapper.insert_or_assign("%Invalid", Invalid());
|
||||
|
||||
// Spin through the blocks and generate basic block ops
|
||||
for(size_t i = 0; i < Defs.size(); ++i) {
|
||||
auto &Def = Defs[i];
|
||||
if (Def.OpEnum == FEXCore::IR::IROps::OP_CODEBLOCK) {
|
||||
auto CodeBlock = _CodeBlock(InvalidNode, InvalidNode);
|
||||
SSANameMapper[Def.Definition] = CodeBlock.Node;
|
||||
SSANameMapper.insert_or_assign(Def.Definition, CodeBlock.Node);
|
||||
Def.Node = CodeBlock.Node;
|
||||
|
||||
if (i == 1) {
|
||||
@@ -532,23 +532,22 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
FEXCore::IR::IROp_CodeBlock *CurrentBlockOp{};
|
||||
for(size_t i = 1; i < Defs.size(); ++i) {
|
||||
auto &Def = Defs[i];
|
||||
CurrentDef = &Def;
|
||||
|
||||
CurrentDef = &Def;
|
||||
|
||||
switch (Def.OpEnum) {
|
||||
// Special handled
|
||||
case FEXCore::IR::IROps::OP_IRHEADER:
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("IRHEADER used in the middle of the block!");
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("IRHEADER used in the middle of the block!");
|
||||
return false; // only one OP_IRHEADER allowed per block
|
||||
|
||||
case FEXCore::IR::IROps::OP_CODEBLOCK: {
|
||||
SetWriteCursor(nullptr); // isolate from previous block
|
||||
if (CurrentBlock != nullptr) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("CodeBlock being used inside of already existing codeblock!");
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("CodeBlock being used inside of already existing codeblock!");
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -560,15 +559,16 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
|
||||
case FEXCore::IR::IROps::OP_BEGINBLOCK: {
|
||||
if (CurrentBlock == nullptr) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("EndBlock being used outside of a block!");
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("EndBlock being used outside of a block!");
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Adjust = DecodeValue<OrderedNode*>(Def.Args[0]);
|
||||
|
||||
if (!CheckPrintError(Def, Adjust.first)) return false;
|
||||
if (!CheckPrintError(Def, Adjust.first)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Def.Node = _BeginBlock(Adjust.second);
|
||||
CurrentBlockOp->Begin = Def.Node->Wrapped(DualListData.ListBegin());
|
||||
@@ -577,15 +577,16 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
|
||||
case FEXCore::IR::IROps::OP_ENDBLOCK: {
|
||||
if (CurrentBlock == nullptr) {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("EndBlock being used outside of a block!");
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("EndBlock being used outside of a block!");
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Adjust = DecodeValue<OrderedNode*>(Def.Args[0]);
|
||||
|
||||
if (!CheckPrintError(Def, Adjust.first)) return false;
|
||||
if (!CheckPrintError(Def, Adjust.first)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Def.Node = _EndBlock(Adjust.second);
|
||||
CurrentBlockOp->Last = Def.Node->Wrapped(DualListData.ListBegin());
|
||||
@@ -597,20 +598,18 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
case FEXCore::IR::IROps::OP_DUMMY: {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("Dummy op must not be used");
|
||||
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("Dummy op must not be used");
|
||||
break;
|
||||
}
|
||||
#define IROP_PARSER_SWITCH_HELPERS
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
default: {
|
||||
LogMan::Msg::E("Error on Line: %d", Def.LineNumber);
|
||||
LogMan::Msg::E("%s", Lines[Def.LineNumber].c_str());
|
||||
LogMan::Msg::E("Unhandled Op enum '%s' in parser", Def.IROp.c_str());
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("Unhandled Op enum '{}' in parser", Def.IROp);
|
||||
return false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -624,7 +623,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
IROp->Size = Def.Size.Bytes();
|
||||
IROp->ElementSize = 0;
|
||||
}
|
||||
SSANameMapper[Def.Definition] = Def.Node;
|
||||
SSANameMapper.insert_or_assign(Def.Definition, Def.Node);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -632,11 +631,11 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
void InitializeStaticTables() {
|
||||
if (NameToOpMap.size() == 0) {
|
||||
if (NameToOpMap.empty()) {
|
||||
for (FEXCore::IR::IROps Op = FEXCore::IR::IROps::OP_DUMMY;
|
||||
Op <= FEXCore::IR::IROps::OP_LAST;
|
||||
Op = static_cast<FEXCore::IR::IROps>(static_cast<uint32_t>(Op) + 1)) {
|
||||
NameToOpMap[FEXCore::IR::GetName(Op)] = Op;
|
||||
NameToOpMap.insert_or_assign(FEXCore::IR::GetName(Op), Op);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -644,13 +643,12 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
|
||||
} // anon namespace
|
||||
|
||||
IREmitter* Parse(std::istream *in) {
|
||||
auto parser = new IRParser(in);
|
||||
std::unique_ptr<IREmitter> Parse(std::istream *in) {
|
||||
auto parser = std::make_unique<IRParser>(in);
|
||||
|
||||
if (parser->Loaded) {
|
||||
return parser;
|
||||
} else {
|
||||
delete parser;
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
+2
-4
@@ -45,10 +45,9 @@ void PassManager::AddDefaultPasses(bool InlineConstants, bool StaticRegisterAllo
|
||||
InsertPass(CreateStaticRegisterAllocationPass());
|
||||
}
|
||||
|
||||
CompactionPass = CreateIRCompaction();
|
||||
// If the IR is compacted post-RA then the node indexing gets messed up and the backend isn't able to find the register assigned to a node
|
||||
// Compact before IR, don't worry about RA generating spills/fills
|
||||
InsertPass(CompactionPass);
|
||||
CompactionPass = InsertPass(CreateIRCompaction());
|
||||
}
|
||||
|
||||
void PassManager::AddDefaultValidationPasses() {
|
||||
@@ -60,8 +59,7 @@ void PassManager::AddDefaultValidationPasses() {
|
||||
}
|
||||
|
||||
void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA) {
|
||||
RAPass = IR::CreateRegisterAllocationPass(CompactionPass, OptimizeSRA);
|
||||
InsertPass(RAPass);
|
||||
RAPass = InsertPass(IR::CreateRegisterAllocationPass(CompactionPass, OptimizeSRA));
|
||||
}
|
||||
|
||||
bool PassManager::Run(IREmitter *IREmit) {
|
||||
|
||||
+6
-6
@@ -42,9 +42,9 @@ class PassManager final {
|
||||
public:
|
||||
void AddDefaultPasses(bool InlineConstants, bool StaticRegisterAllocation);
|
||||
void AddDefaultValidationPasses();
|
||||
void InsertPass(Pass *Pass) {
|
||||
Pass* InsertPass(std::unique_ptr<Pass> Pass) {
|
||||
Pass->RegisterPassManager(this);
|
||||
Passes.emplace_back(Pass);
|
||||
return Passes.emplace_back(std::move(Pass)).get();
|
||||
}
|
||||
|
||||
void InsertRegisterAllocationPass(bool OptimizeSRA);
|
||||
@@ -52,7 +52,7 @@ public:
|
||||
bool Run(IREmitter *IREmit);
|
||||
|
||||
void RegisterExitHandler(ShouldExitHandler Handler) {
|
||||
ExitHandler = Handler;
|
||||
ExitHandler = std::move(Handler);
|
||||
}
|
||||
|
||||
bool HasRAPass() const {
|
||||
@@ -73,15 +73,15 @@ protected:
|
||||
|
||||
private:
|
||||
Pass *RAPass{};
|
||||
FEXCore::IR::Pass *CompactionPass{};
|
||||
Pass *CompactionPass{};
|
||||
|
||||
std::vector<std::unique_ptr<Pass>> Passes;
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
std::vector<std::unique_ptr<Pass>> ValidationPasses;
|
||||
void InsertValidationPass(Pass *Pass) {
|
||||
void InsertValidationPass(std::unique_ptr<Pass> Pass) {
|
||||
Pass->RegisterPassManager(this);
|
||||
ValidationPasses.emplace_back(Pass);
|
||||
ValidationPasses.emplace_back(std::move(Pass));
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
+15
-13
@@ -1,25 +1,27 @@
|
||||
#pragma once
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
FEXCore::IR::Pass* CreateConstProp(bool InlineConstants);
|
||||
FEXCore::IR::Pass* CreateContextLoadStoreElimination();
|
||||
FEXCore::IR::Pass* CreateSyscallOptimization();
|
||||
FEXCore::IR::Pass* CreateDeadFlagCalculationEliminination();
|
||||
FEXCore::IR::Pass* CreateDeadStoreElimination();
|
||||
FEXCore::IR::Pass* CreatePassDeadCodeElimination();
|
||||
FEXCore::IR::Pass* CreateIRCompaction();
|
||||
FEXCore::IR::RegisterAllocationPass* CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool OptimizeSRA);
|
||||
FEXCore::IR::Pass* CreateStaticRegisterAllocationPass();
|
||||
FEXCore::IR::Pass* CreateLongDivideEliminationPass();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateSyscallOptimization();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction();
|
||||
std::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool OptimizeSRA);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
|
||||
|
||||
namespace Validation {
|
||||
FEXCore::IR::Pass* CreateIRValidation();
|
||||
FEXCore::IR::Pass* CreatePhiValidation();
|
||||
FEXCore::IR::Pass* CreateValueDominanceValidation();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreatePhiValidation();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateValueDominanceValidation();
|
||||
}
|
||||
}
|
||||
|
||||
+370
-315
@@ -19,15 +19,6 @@ $end_info$
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class ConstProp final : public FEXCore::IR::Pass {
|
||||
std::unordered_map<uint64_t, OrderedNode*> ConstPool;
|
||||
std::map<OrderedNode*, uint64_t> AddressgenConsts;
|
||||
public:
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
bool InlineConstants;
|
||||
ConstProp(bool DoInlineConstants) : InlineConstants(DoInlineConstants) { }
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
uint64_t getMask(T Op) {
|
||||
uint64_t NumBits = Op->Header.Size * 8;
|
||||
@@ -69,8 +60,7 @@ static bool IsImmMemory(uint64_t imm, uint8_t AccessSize) {
|
||||
}
|
||||
}
|
||||
|
||||
std::tuple<MemOffsetType, uint8_t, OrderedNode*, OrderedNode*> MemExtendedAddressing(IREmitter *IREmit, uint8_t AccessSize, IROp_Header* AddressHeader) {
|
||||
|
||||
static std::tuple<MemOffsetType, uint8_t, OrderedNode*, OrderedNode*> MemExtendedAddressing(IREmitter *IREmit, uint8_t AccessSize, IROp_Header* AddressHeader) {
|
||||
auto Src0Header = IREmit->GetOpHeader(AddressHeader->Args[0]);
|
||||
if (Src0Header->Size == 8) {
|
||||
//Try to optimize: Base + MUL(Offset, Scale)
|
||||
@@ -124,7 +114,7 @@ std::tuple<MemOffsetType, uint8_t, OrderedNode*, OrderedNode*> MemExtendedAddres
|
||||
return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[0]), IREmit->UnwrapNode(AddressHeader->Args[1]) };
|
||||
}
|
||||
|
||||
OrderedNodeWrapper RemoveUselessMasking(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t mask) {
|
||||
static OrderedNodeWrapper RemoveUselessMasking(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t mask) {
|
||||
#if 1 // HOTFIX: We need to clear up the meaning of opsize and dest size. See #594
|
||||
return src;
|
||||
#else
|
||||
@@ -151,7 +141,7 @@ OrderedNodeWrapper RemoveUselessMasking(IREmitter *IREmit, OrderedNodeWrapper sr
|
||||
#endif
|
||||
}
|
||||
|
||||
bool IsBfeAlreadyDone(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t Width) {
|
||||
static bool IsBfeAlreadyDone(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t Width) {
|
||||
auto IROp = IREmit->GetOpHeader(src);
|
||||
if (IROp->Op == OP_BFE) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
@@ -162,31 +152,55 @@ bool IsBfeAlreadyDone(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t Width)
|
||||
return false;
|
||||
}
|
||||
|
||||
bool ConstProp::Run(IREmitter *IREmit) {
|
||||
class ConstProp final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit ConstProp(bool DoInlineConstants) : InlineConstants(DoInlineConstants) { }
|
||||
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
|
||||
bool InlineConstants;
|
||||
|
||||
private:
|
||||
bool HandleConstantPools(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
void LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
bool ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& CurrentIR,
|
||||
OrderedNode* CodeNode, IROp_Header* IROp);
|
||||
bool ConstantPropagation(IREmitter *IREmit, const IRListView& CurrentIR,
|
||||
OrderedNode* CodeNode, IROp_Header* IROp);
|
||||
bool ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR);
|
||||
|
||||
std::unordered_map<uint64_t, OrderedNode*> ConstPool;
|
||||
std::map<OrderedNode*, uint64_t> AddressgenConsts;
|
||||
};
|
||||
|
||||
bool ConstProp::HandleConstantPools(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
{
|
||||
|
||||
// constants are pooled per block
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
if (ConstPool.count(Op->Constant)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ConstPool[Op->Constant]);
|
||||
Changed = true;
|
||||
} else {
|
||||
ConstPool[Op->Constant] = CodeNode;
|
||||
}
|
||||
// constants are pooled per block
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
if (ConstPool.count(Op->Constant)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ConstPool[Op->Constant]);
|
||||
Changed = true;
|
||||
} else {
|
||||
ConstPool[Op->Constant] = CodeNode;
|
||||
}
|
||||
}
|
||||
ConstPool.clear();
|
||||
}
|
||||
ConstPool.clear();
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
// Code motion around selects
|
||||
// Moves unary ops that depend on a select before the select, if both inputs are constants
|
||||
// assumes that unary ops without side effects on constants will be constprop'd
|
||||
void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
// Code motion around selects
|
||||
// Moves unary ops that depend on a select before the select, if both inputs are constants
|
||||
// assumes that unary ops without side effects on constants will be constprop'd
|
||||
@@ -243,9 +257,9 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// FCMP optimization
|
||||
|
||||
void ConstProp::FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
// Make all FCMPs set no flags
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
if (IROp->Op == OP_FCMP) {
|
||||
@@ -266,10 +280,11 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// LoadMem / StoreMem imm pooling
|
||||
// If imms are close by, use address gen to generate the values instead of using a new imm
|
||||
// LoadMem / StoreMem imm pooling
|
||||
// If imms are close by, use address gen to generate the values instead of using a new imm
|
||||
void ConstProp::LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_LOADMEM || IROp->Op == OP_STOREMEM) {
|
||||
@@ -293,152 +308,163 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
}
|
||||
AddressgenConsts.clear();
|
||||
}
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
// zext / masking elimination
|
||||
switch (IROp->Op) {
|
||||
// Generic handling
|
||||
case OP_OR:
|
||||
case OP_XOR:
|
||||
case OP_NOT:
|
||||
case OP_ADD:
|
||||
case OP_SUB:
|
||||
case OP_MUL:
|
||||
case OP_UMUL:
|
||||
case OP_DIV:
|
||||
case OP_UDIV:
|
||||
case OP_LSHR:
|
||||
case OP_ASHR:
|
||||
case OP_LSHL:
|
||||
case OP_ROR: {
|
||||
for (int i = 0; i < IROp->NumArgs; i++) {
|
||||
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[i], getMask(IROp));
|
||||
if (newArg.ID() != IROp->Args[i].ID()) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(newArg));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& CurrentIR,
|
||||
OrderedNode* CodeNode, IROp_Header* IROp) {
|
||||
bool Changed = false;
|
||||
|
||||
case OP_AND: {
|
||||
// if AND's arguments are imms, they are masking
|
||||
for (int i = 0; i < IROp->NumArgs; i++) {
|
||||
auto mask = getMask(IROp);
|
||||
uint64_t imm = 0;
|
||||
if (IREmit->IsValueConstant(IROp->Args[i^1], &imm))
|
||||
mask = imm;
|
||||
|
||||
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[i], imm);
|
||||
|
||||
if (newArg.ID() != IROp->Args[i].ID()) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(newArg));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_BFE: {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
|
||||
// Is this value already BFE'd?
|
||||
if (IsBfeAlreadyDone(IREmit, IROp->Args[0], Op->Width)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
//printf("Removed BFE once \n");
|
||||
break;
|
||||
}
|
||||
|
||||
// Is this value already ZEXT'd?
|
||||
if (Op->lsb == 0) {
|
||||
//LoadMem, LoadMemTSO & LoadContext ZExt
|
||||
auto source = IROp->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
|
||||
if (Op->Width >= (sourceHeader->Size*8) &&
|
||||
(sourceHeader->Op == OP_LOADMEM || sourceHeader->Op == OP_LOADMEMTSO || sourceHeader->Op == OP_LOADCONTEXT)
|
||||
) {
|
||||
//printf("Eliminated needless zext bfe\n");
|
||||
// Load mem / load ctx zexts, no need to vmem
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// BFE does implicit masking, remove any masks leading to this, if possible
|
||||
uint64_t imm = 1ULL << (Op->Width-1);
|
||||
imm = (imm-1) *2 + 1;
|
||||
imm <<= Op->lsb;
|
||||
|
||||
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[0], imm);
|
||||
|
||||
if (newArg.ID() != IROp->Args[0].ID()) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(newArg));
|
||||
switch (IROp->Op) {
|
||||
// Generic handling
|
||||
case OP_OR:
|
||||
case OP_XOR:
|
||||
case OP_NOT:
|
||||
case OP_ADD:
|
||||
case OP_SUB:
|
||||
case OP_MUL:
|
||||
case OP_UMUL:
|
||||
case OP_DIV:
|
||||
case OP_UDIV:
|
||||
case OP_LSHR:
|
||||
case OP_ASHR:
|
||||
case OP_LSHL:
|
||||
case OP_ROR: {
|
||||
for (int i = 0; i < IROp->NumArgs; i++) {
|
||||
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[i], getMask(IROp));
|
||||
if (newArg.ID() != IROp->Args[i].ID()) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(newArg));
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_SBFE: {
|
||||
auto Op = IROp->C<IR::IROp_Sbfe>();
|
||||
case OP_AND: {
|
||||
// if AND's arguments are imms, they are masking
|
||||
for (int i = 0; i < IROp->NumArgs; i++) {
|
||||
auto mask = getMask(IROp);
|
||||
uint64_t imm = 0;
|
||||
if (IREmit->IsValueConstant(IROp->Args[i^1], &imm))
|
||||
mask = imm;
|
||||
|
||||
// BFE does implicit masking
|
||||
uint64_t imm = 1ULL << (Op->Width-1);
|
||||
imm = (imm-1) *2 + 1;
|
||||
imm <<= Op->lsb;
|
||||
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[i], imm);
|
||||
|
||||
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[0], imm);
|
||||
|
||||
if (newArg.ID() != IROp->Args[0].ID()) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(newArg));
|
||||
if (newArg.ID() != IROp->Args[i].ID()) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(newArg));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_BFE: {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
|
||||
// Is this value already BFE'd?
|
||||
if (IsBfeAlreadyDone(IREmit, IROp->Args[0], Op->Width)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
|
||||
//printf("Removed BFE once \n");
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VFADD:
|
||||
case OP_VFSUB:
|
||||
case OP_VFMUL:
|
||||
case OP_VFDIV:
|
||||
case OP_FCMP: {
|
||||
auto flopSize = IROp->Size;
|
||||
for (int i = 0; i < IROp->NumArgs; i++) {
|
||||
auto argHeader = IREmit->GetOpHeader(IROp->Args[i]);
|
||||
|
||||
if (argHeader->Op == OP_VMOV) {
|
||||
auto source = argHeader->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
if (sourceHeader->Size >= flopSize) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(source));
|
||||
//printf("VMOV bypassed\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VMOV: {
|
||||
// elim from load mem
|
||||
// Is this value already ZEXT'd?
|
||||
if (Op->lsb == 0) {
|
||||
//LoadMem, LoadMemTSO & LoadContext ZExt
|
||||
auto source = IROp->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
|
||||
if (IROp->Size >= sourceHeader->Size &&
|
||||
if (Op->Width >= (sourceHeader->Size*8) &&
|
||||
(sourceHeader->Op == OP_LOADMEM || sourceHeader->Op == OP_LOADMEMTSO || sourceHeader->Op == OP_LOADCONTEXT)
|
||||
) {
|
||||
//printf("Eliminated needless zext VMOV\n");
|
||||
) {
|
||||
//printf("Eliminated needless zext bfe\n");
|
||||
// Load mem / load ctx zexts, no need to vmem
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source));
|
||||
} else if (IROp->Size == sourceHeader->Size) {
|
||||
// VMOV of same size
|
||||
//printf("printf vmov of same size?!\n");
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
|
||||
// BFE does implicit masking, remove any masks leading to this, if possible
|
||||
uint64_t imm = 1ULL << (Op->Width-1);
|
||||
imm = (imm-1) *2 + 1;
|
||||
imm <<= Op->lsb;
|
||||
|
||||
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[0], imm);
|
||||
|
||||
if (newArg.ID() != IROp->Args[0].ID()) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(newArg));
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// constprop + some more per instruction logic
|
||||
case OP_SBFE: {
|
||||
auto Op = IROp->C<IR::IROp_Sbfe>();
|
||||
|
||||
// BFE does implicit masking
|
||||
uint64_t imm = 1ULL << (Op->Width-1);
|
||||
imm = (imm-1) *2 + 1;
|
||||
imm <<= Op->lsb;
|
||||
|
||||
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[0], imm);
|
||||
|
||||
if (newArg.ID() != IROp->Args[0].ID()) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(newArg));
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VFADD:
|
||||
case OP_VFSUB:
|
||||
case OP_VFMUL:
|
||||
case OP_VFDIV:
|
||||
case OP_FCMP: {
|
||||
auto flopSize = IROp->Size;
|
||||
for (int i = 0; i < IROp->NumArgs; i++) {
|
||||
auto argHeader = IREmit->GetOpHeader(IROp->Args[i]);
|
||||
|
||||
if (argHeader->Op == OP_VMOV) {
|
||||
auto source = argHeader->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
if (sourceHeader->Size >= flopSize) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(source));
|
||||
//printf("VMOV bypassed\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VMOV: {
|
||||
// elim from load mem
|
||||
auto source = IROp->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
|
||||
if (IROp->Size >= sourceHeader->Size &&
|
||||
(sourceHeader->Op == OP_LOADMEM || sourceHeader->Op == OP_LOADMEMTSO || sourceHeader->Op == OP_LOADCONTEXT)
|
||||
) {
|
||||
//printf("Eliminated needless zext VMOV\n");
|
||||
// Load mem / load ctx zexts, no need to vmem
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source));
|
||||
} else if (IROp->Size == sourceHeader->Size) {
|
||||
// VMOV of same size
|
||||
//printf("printf vmov of same size?!\n");
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source));
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
// constprop + some more per instruction logic
|
||||
bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& CurrentIR,
|
||||
OrderedNode* CodeNode, IROp_Header* IROp) {
|
||||
bool Changed = false;
|
||||
|
||||
switch (IROp->Op) {
|
||||
/*
|
||||
case OP_UMUL:
|
||||
@@ -490,7 +516,6 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Header.Args[0]);
|
||||
|
||||
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
|
||||
|
||||
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, Op->Size, AddressHeader);
|
||||
|
||||
Op->OffsetType = OffsetType;
|
||||
@@ -530,7 +555,6 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
uint64_t NewConstant = (Constant1 + Constant2) & getMask(Op) ;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
continue;
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -711,9 +735,9 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
uint64_t NewConstant = (Constant1 * Constant2) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
} else if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2) && __builtin_popcountl(Constant2) == 1) {
|
||||
} else if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2) && std::popcount(Constant2) == 1) {
|
||||
if (IROp->Size == 4 || IROp->Size == 8) {
|
||||
uint64_t amt = __builtin_ctzl(Constant2);
|
||||
uint64_t amt = std::countr_zero(Constant2);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto shift = IREmit->_Lshl(CurrentIR.GetNode(Op->Header.Args[0]), IREmit->_Constant(amt));
|
||||
shift.first->Header.Size = IROp->Size; // force Lshl to be the same size as the original Mul
|
||||
@@ -753,192 +777,223 @@ bool ConstProp::Run(IREmitter *IREmit) {
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// constant inlining
|
||||
if (InlineConstants) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
switch(IROp->Op) {
|
||||
case OP_LSHR:
|
||||
case OP_ASHR:
|
||||
case OP_ROR:
|
||||
case OP_LSHL:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Lshr>();
|
||||
return Changed;
|
||||
}
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR) {
|
||||
bool Changed = false;
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
switch(IROp->Op) {
|
||||
case OP_LSHR:
|
||||
case OP_ASHR:
|
||||
case OP_ROR:
|
||||
case OP_LSHL:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Lshr>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
// this shouldn't be here, but rather on the emitter themselves or the constprop transformation?
|
||||
if (IROp->Size <=4)
|
||||
Constant2 &= 31;
|
||||
else
|
||||
Constant2 &= 63;
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_ADD:
|
||||
case OP_SUB:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
// this shouldn't be here, but rather on the emitter themselves or the constprop transformation?
|
||||
if (IROp->Size <=4)
|
||||
Constant2 &= 31;
|
||||
else
|
||||
Constant2 &= 63;
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_SELECT:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
|
||||
if (IsImmAddSub(Constant1)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant1));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
case OP_ADD:
|
||||
case OP_SUB:
|
||||
uint64_t Constant2{};
|
||||
uint64_t Constant3{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[2], &Constant2) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[3], &Constant3) &&
|
||||
Constant2 == 1 &&
|
||||
Constant3 == 0)
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[2]));
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, IREmit->_InlineConstant(Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 3, IREmit->_InlineConstant(Constant3));
|
||||
}
|
||||
|
||||
case OP_SELECT:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
break;
|
||||
}
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
|
||||
if (IsImmAddSub(Constant1)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
case OP_CONDJUMP:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant1));
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
uint64_t Constant2{};
|
||||
uint64_t Constant3{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[2], &Constant2) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[3], &Constant3) &&
|
||||
Constant2 == 1 &&
|
||||
Constant3 == 0)
|
||||
{
|
||||
case OP_EXITFUNCTION:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
|
||||
uint64_t Constant{};
|
||||
if (IREmit->IsValueConstant(Op->NewRIP, &Constant)) {
|
||||
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(Constant));
|
||||
|
||||
Changed = true;
|
||||
} else {
|
||||
auto NewRIP = IREmit->GetOpHeader(Op->NewRIP);
|
||||
if (NewRIP->Op == OP_ENTRYPOINTOFFSET) {
|
||||
auto EO = NewRIP->C<IR::IROp_EntrypointOffset>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(EO->Offset, EO->Header.Size));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_OR:
|
||||
case OP_XOR:
|
||||
case OP_AND:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_Or>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmLogical(Constant2, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LOADMEM:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmMemory(Constant2, Op->Size)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_STOREMEM:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Header.Args[2], &Constant2)) {
|
||||
if (IsImmMemory(Constant2, Op->Size)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[2]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, IREmit->_InlineConstant(Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 3, IREmit->_InlineConstant(Constant3));
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_CONDJUMP:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmAddSub(Constant2)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_EXITFUNCTION:
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
|
||||
uint64_t Constant{};
|
||||
if (IREmit->IsValueConstant(Op->NewRIP, &Constant)) {
|
||||
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(Constant));
|
||||
|
||||
Changed = true;
|
||||
} else {
|
||||
auto NewRIP = IREmit->GetOpHeader(Op->NewRIP);
|
||||
if (NewRIP->Op == OP_ENTRYPOINTOFFSET) {
|
||||
auto EO = NewRIP->C<IR::IROp_EntrypointOffset>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(EO->Offset, EO->Header.Size));
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_OR:
|
||||
case OP_XOR:
|
||||
case OP_AND:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_Or>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmLogical(Constant2, IROp->Size * 8)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LOADMEM:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsImmMemory(Constant2, Op->Size)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_STOREMEM:
|
||||
{
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
|
||||
uint64_t Constant2{};
|
||||
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Header.Args[2], &Constant2)) {
|
||||
if (IsImmMemory(Constant2, Op->Size)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[2]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, IREmit->_InlineConstant(Constant2));
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default: break;
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(OriginalWriteCursor);
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateConstProp(bool InlineConstants) {
|
||||
return new ConstProp(InlineConstants);
|
||||
bool ConstProp::Run(IREmitter *IREmit) {
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
if (HandleConstantPools(IREmit, CurrentIR)) {
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
CodeMotionAroundSelects(IREmit, CurrentIR);
|
||||
FCMPOptimization(IREmit, CurrentIR);
|
||||
LoadMemStoreMemImmediatePooling(IREmit, CurrentIR);
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
if (ZextAndMaskingElimination(IREmit, CurrentIR, CodeNode, IROp)) {
|
||||
Changed = true;
|
||||
}
|
||||
if (ConstantPropagation(IREmit, CurrentIR, CodeNode, IROp)) {
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (InlineConstants && ConstantInlining(IREmit, CurrentIR)) {
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(OriginalWriteCursor);
|
||||
return Changed;
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants) {
|
||||
return std::make_unique<ConstProp>(InlineConstants);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -59,10 +59,8 @@ void DeadCodeElimination::markUsed(OrderedNodeWrapper *CodeOp, IROp_Header *IROp
|
||||
|
||||
}
|
||||
|
||||
|
||||
FEXCore::IR::Pass* CreatePassDeadCodeElimination() {
|
||||
return new DeadCodeElimination{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination() {
|
||||
return std::make_unique<DeadCodeElimination>();
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
+3
-3
@@ -283,7 +283,7 @@ class RCLSE final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
RCLSE() {
|
||||
ClassifyContextStruct(&ClassifiedStruct);
|
||||
DCE.reset(FEXCore::IR::CreatePassDeadCodeElimination());
|
||||
DCE = FEXCore::IR::CreatePassDeadCodeElimination();
|
||||
}
|
||||
bool Run(FEXCore::IR::IREmitter *IREmit) override;
|
||||
private:
|
||||
@@ -636,8 +636,8 @@ bool RCLSE::Run(FEXCore::IR::IREmitter *IREmit) {
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
FEXCore::IR::Pass* CreateContextLoadStoreElimination() {
|
||||
return new RCLSE{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination() {
|
||||
return std::make_unique<RCLSE>();
|
||||
}
|
||||
|
||||
}
|
||||
@@ -331,8 +331,8 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateDeadStoreElimination() {
|
||||
return new DeadStoreElimination{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination() {
|
||||
return std::make_unique<DeadStoreElimination>();
|
||||
}
|
||||
|
||||
}
|
||||
@@ -153,8 +153,10 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
{
|
||||
// Fixup the arguments of all the IROps
|
||||
for (auto &Block : GeneratedCodeBlocks) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = LocalIR.GetOp<FEXCore::IR::IROp_CodeBlock>(Block.NewNode);
|
||||
LOGMAN_THROW_A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
for (auto [LocalNode, LocalIROp] : LocalIR.GetCode(Block.NewNode)) {
|
||||
|
||||
@@ -199,8 +201,8 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
return true;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateIRCompaction() {
|
||||
return new IRCompaction{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction() {
|
||||
return std::make_unique<IRCompaction>();
|
||||
}
|
||||
|
||||
}
|
||||
@@ -11,7 +11,7 @@ $end_info$
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Common/BitSet.h"
|
||||
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
|
||||
namespace {
|
||||
struct BlockInfo {
|
||||
@@ -54,8 +54,10 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
std::vector<uint32_t> Uses(CurrentIR.GetSSACount(), 0);
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto HeaderOp = CurrentIR.GetHeader();
|
||||
LOGMAN_THROW_A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
#endif
|
||||
|
||||
IR::RegisterAllocationData * RAData{};
|
||||
if (Manager->HasRAPass()) {
|
||||
@@ -279,13 +281,13 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
Out << "Warnings:" << std::endl << Warnings.str() << std::endl;
|
||||
}
|
||||
|
||||
LogMan::Msg::E("%s", Out.str().c_str());
|
||||
LogMan::Msg::EFmt("{}", Out.str());
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateIRValidation() {
|
||||
return new IRValidation{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateIRValidation() {
|
||||
return std::make_unique<IRValidation>();
|
||||
}
|
||||
}
|
||||
@@ -106,7 +106,7 @@ bool LongDivideEliminationPass::Run(IREmitter *IREmit) {
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateLongDivideEliminationPass() {
|
||||
return new LongDivideEliminationPass{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass() {
|
||||
return std::make_unique<LongDivideEliminationPass>();
|
||||
}
|
||||
}
|
||||
@@ -8,7 +8,7 @@ $end_info$
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
|
||||
namespace FEXCore::IR::Validation {
|
||||
|
||||
@@ -59,15 +59,15 @@ bool PhiValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
Out << "Errors:" << std::endl << Errors.str() << std::endl;
|
||||
|
||||
LogMan::Msg::E(Out.str().c_str());
|
||||
LogMan::Msg::EFmt("{}", Out.str());
|
||||
}
|
||||
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreatePhiValidation() {
|
||||
return new PhiValidation{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreatePhiValidation() {
|
||||
return std::make_unique<PhiValidation>();
|
||||
}
|
||||
|
||||
}
|
||||
+2
-2
@@ -59,8 +59,8 @@ bool DeadFlagCalculationEliminination::Run(IREmitter *IREmit) {
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateDeadFlagCalculationEliminination() {
|
||||
return new DeadFlagCalculationEliminination{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination() {
|
||||
return std::make_unique<DeadFlagCalculationEliminination>();
|
||||
}
|
||||
|
||||
}
|
||||
@@ -46,7 +46,7 @@ namespace {
|
||||
for (int i = 1; i < Size; i++)
|
||||
Items[i] = 0xDEADBEEF;
|
||||
#endif
|
||||
Next.release();
|
||||
Next.reset();
|
||||
}
|
||||
|
||||
BucketList() {
|
||||
@@ -144,7 +144,7 @@ namespace {
|
||||
}
|
||||
else if (++i == Size) {
|
||||
if (that->Next->Items[0] == 0) {
|
||||
that->Next.release();
|
||||
that->Next.reset();
|
||||
foundThat->Items[foundI] = that->Items[Size-1];
|
||||
that->Items[Size-1] = 0;
|
||||
break;
|
||||
@@ -1399,11 +1399,13 @@ namespace FEXCore::IR {
|
||||
if (InterferenceNode != ~0U) {
|
||||
FEXCore::IR::RegisterClassType InterferenceRegClass = FEXCore::IR::RegisterClassType{Graph->AllocData->Map[InterferenceNode].Class};
|
||||
uint32_t SpillSlot = FindSpillSlot(InterferenceNode, InterferenceRegClass);
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
RegisterNode *InterferenceRegisterNode = &Graph->Nodes[InterferenceNode];
|
||||
LOGMAN_THROW_A(SpillSlot != ~0U, "Interference Node doesn't have a spill slot!");
|
||||
//LOGMAN_THROW_A(InterferenceRegisterNode->Head.RegAndClass.Reg != INVALID_REG, "Interference node never assigned a register?");
|
||||
LOGMAN_THROW_A(InterferenceRegClass != ~0U, "Interference node never assigned a register class?");
|
||||
LOGMAN_THROW_A(InterferenceRegisterNode->Head.PhiPartner == nullptr, "We don't support spilling PHI nodes currently");
|
||||
#endif
|
||||
|
||||
// This is the op that we need to dump
|
||||
auto [InterferenceOrderedNode, InterferenceIROp] = IR.at(InterferenceNode)();
|
||||
@@ -1538,7 +1540,7 @@ namespace FEXCore::IR {
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterAllocationPass* CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool OptimizeSRA) {
|
||||
return new ConstrainedRAPass{CompactionPass, OptimizeSRA};
|
||||
std::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool OptimizeSRA) {
|
||||
return std::make_unique<ConstrainedRAPass>(CompactionPass, OptimizeSRA);
|
||||
}
|
||||
}
|
||||
+2
-2
@@ -94,8 +94,8 @@ bool StaticRegisterAllocationPass::Run(IREmitter *IREmit) {
|
||||
return true;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateStaticRegisterAllocationPass() {
|
||||
return new StaticRegisterAllocationPass{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass() {
|
||||
return std::make_unique<StaticRegisterAllocationPass>();
|
||||
}
|
||||
|
||||
}
|
||||
@@ -44,12 +44,11 @@ bool SyscallOptimization::Run(IREmitter *IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateSyscallOptimization() {
|
||||
return new SyscallOptimization{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateSyscallOptimization() {
|
||||
return std::make_unique<SyscallOptimization>();
|
||||
}
|
||||
|
||||
}
|
||||
@@ -8,9 +8,9 @@ $end_info$
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
#include <iostream>
|
||||
#include <map>
|
||||
#include <list>
|
||||
#include <sstream>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace {
|
||||
@@ -206,14 +206,14 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
Out << "Warnings:" << std::endl << Warnings.str() << std::endl;
|
||||
}
|
||||
|
||||
LogMan::Msg::E(Out.str().c_str());
|
||||
LogMan::Msg::EFmt("{}", Out.str());
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
FEXCore::IR::Pass* CreateValueDominanceValidation() {
|
||||
return new ValueDominanceValidation{};
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateValueDominanceValidation() {
|
||||
return std::make_unique<ValueDominanceValidation>();
|
||||
}
|
||||
|
||||
}
|
||||
+46
-5
@@ -3,6 +3,7 @@
|
||||
#include <sys/mman.h>
|
||||
#include <jemalloc/jemalloc.h>
|
||||
#include <memory>
|
||||
#include <malloc.h>
|
||||
|
||||
extern "C" {
|
||||
extern void *__libc_malloc(size_t size);
|
||||
@@ -16,6 +17,15 @@ extern "C" {
|
||||
|
||||
extern mmap_hook_type __mmap_hook;
|
||||
extern munmap_hook_type __munmap_hook;
|
||||
|
||||
static FEXCore::Allocator::MALLOC_Hook global_malloc {::__libc_malloc};
|
||||
static FEXCore::Allocator::REALLOC_Hook global_realloc {::__libc_realloc};
|
||||
static FEXCore::Allocator::FREE_Hook global_free {::__libc_free};
|
||||
|
||||
// Override the global functions
|
||||
FEX_DEFAULT_VISIBILITY void *malloc(size_t size) { return global_malloc(size); }
|
||||
FEX_DEFAULT_VISIBILITY void *realloc(void *ptr, size_t size) { return global_realloc(ptr, size); }
|
||||
FEX_DEFAULT_VISIBILITY void free(void *ptr) { return global_free(ptr); }
|
||||
}
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
@@ -25,6 +35,10 @@ namespace FEXCore::Allocator {
|
||||
REALLOC_Hook realloc {::__libc_realloc};
|
||||
FREE_Hook free {::__libc_free};
|
||||
|
||||
using GLIBC_MALLOC_Hook = void*(*)(size_t, const void *caller);
|
||||
using GLIBC_REALLOC_Hook = void*(*)(void*, size_t, const void *caller);
|
||||
using GLIBC_FREE_Hook = void(*)(void*, const void *caller);
|
||||
|
||||
std::unique_ptr<Alloc::HostAllocator> Alloc64{};
|
||||
|
||||
void *FEX_mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
@@ -57,8 +71,10 @@ namespace FEXCore::Allocator {
|
||||
return ::je_free(ptr);
|
||||
}
|
||||
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wdeprecated-declarations"
|
||||
void SetupHooks() {
|
||||
Alloc64.reset(Alloc::OSAllocator::Create64BitAllocator());
|
||||
Alloc64 = Alloc::OSAllocator::Create64BitAllocator();
|
||||
__mmap_hook = FEX_mmap;
|
||||
__munmap_hook = FEX_munmap;
|
||||
FEXCore::Allocator::mmap = FEX_mmap;
|
||||
@@ -66,12 +82,37 @@ namespace FEXCore::Allocator {
|
||||
FEXCore::Allocator::malloc = ::je_malloc;
|
||||
FEXCore::Allocator::realloc = ::je_realloc;
|
||||
FEXCore::Allocator::free = ::je_free;
|
||||
|
||||
global_malloc = ::je_malloc;
|
||||
global_realloc = ::je_realloc;
|
||||
global_free = ::je_free;
|
||||
|
||||
__malloc_hook = FEXCore::Allocator::FEX_malloc_hook;
|
||||
__realloc_hook = FEXCore::Allocator::FEX_realloc_hook;
|
||||
__free_hook = FEXCore::Allocator::FEX_free_hook;
|
||||
}
|
||||
|
||||
void ClearHooks() {
|
||||
__mmap_hook = ::mmap;
|
||||
__munmap_hook = ::munmap;
|
||||
FEXCore::Allocator::mmap = ::mmap;
|
||||
FEXCore::Allocator::munmap = ::munmap;
|
||||
FEXCore::Allocator::malloc = ::__libc_malloc;
|
||||
FEXCore::Allocator::realloc = ::__libc_realloc;
|
||||
FEXCore::Allocator::free = ::__libc_free;
|
||||
|
||||
global_malloc = ::__libc_malloc;
|
||||
global_realloc = ::__libc_realloc;
|
||||
global_free = ::__libc_free;
|
||||
|
||||
// Reset's glibc hooks
|
||||
__malloc_hook = 0;
|
||||
__realloc_hook = 0;
|
||||
__free_hook = 0;
|
||||
}
|
||||
#pragma GCC diagnostic pop
|
||||
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
// Override the global functions
|
||||
void *malloc(size_t size) { return FEXCore::Allocator::malloc(size); }
|
||||
void *realloc(void *ptr, size_t size) { return FEXCore::Allocator::realloc(ptr, size); }
|
||||
void free(void *ptr) { return FEXCore::Allocator::free(ptr); }
|
||||
}
|
||||
+11
-2
@@ -666,6 +666,15 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
LiveRegions = ObjectAlloc->new_construct(LiveRegions, ObjectAlloc);
|
||||
}
|
||||
else {
|
||||
|
||||
// If the allocation size is large than a page, then try allowing it to be a huge page
|
||||
// This enables the kernel to use transparent large pages in the allocator which can reduce memory pressure
|
||||
// Considering we are allocating the entire VA space, this is a good thing
|
||||
// If MADV_HUGEPAGE isn't support then this will fail harmlessly
|
||||
if (AllocationSize > 4096) {
|
||||
::madvise(Ptr, AllocationSize, MADV_HUGEPAGE);
|
||||
}
|
||||
|
||||
bool Merged = false;
|
||||
if (PrevReserved) {
|
||||
Merged = MergeReservedRegionIfPossible(PrevReserved, reinterpret_cast<uint64_t>(Ptr), AllocationSize);
|
||||
@@ -709,7 +718,7 @@ OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
}
|
||||
}
|
||||
|
||||
Alloc::HostAllocator *Create64BitAllocator() {
|
||||
return new OSAllocator_64Bit{};
|
||||
std::unique_ptr<Alloc::HostAllocator> Create64BitAllocator() {
|
||||
return std::make_unique<OSAllocator_64Bit>();
|
||||
}
|
||||
}
|
||||
+4
-3
@@ -1,6 +1,8 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <sys/types.h>
|
||||
|
||||
constexpr static uint64_t PAGE_SIZE = 4096;
|
||||
@@ -29,16 +31,15 @@ static inline uint64_t AlignUp(uint64_t value, uint64_t size) {
|
||||
GlobalAllocator(HostAllocator *_Alloc)
|
||||
: Alloc {_Alloc} {}
|
||||
|
||||
virtual ~GlobalAllocator() = default;
|
||||
virtual void *malloc(size_t Size) = 0;
|
||||
virtual void *calloc(size_t num, size_t size) = 0;
|
||||
virtual void *realloc(void *ptr, size_t size) = 0;
|
||||
virtual void *memalign(size_t alignment, size_t size) = 0;
|
||||
virtual void free(void *ptr) = 0;
|
||||
};
|
||||
|
||||
GlobalAllocator *CreateBasicAllocator(HostAllocator *Alloc);
|
||||
}
|
||||
|
||||
namespace Alloc::OSAllocator {
|
||||
Alloc::HostAllocator *Create64BitAllocator();
|
||||
std::unique_ptr<Alloc::HostAllocator> Create64BitAllocator();
|
||||
}
|
||||
+19
-1
@@ -37,7 +37,17 @@ void UnInstallHandlers() { Handlers.clear(); }
|
||||
Handler(Buffer);
|
||||
}
|
||||
|
||||
__builtin_trap();
|
||||
FEX_TRAP_EXECUTION;
|
||||
}
|
||||
|
||||
void MFmt(const char *fmt, const fmt::format_args& args) {
|
||||
auto msg = fmt::vformat(fmt, args);
|
||||
|
||||
for (auto& Handler : Handlers) {
|
||||
Handler(msg.c_str());
|
||||
}
|
||||
|
||||
FEX_TRAP_EXECUTION;
|
||||
}
|
||||
} // namespace Throw
|
||||
|
||||
@@ -67,5 +77,13 @@ void M(DebugLevels Level, const char *fmt, va_list args) {
|
||||
}
|
||||
}
|
||||
|
||||
void MFmtImpl(DebugLevels level, const char* fmt, const fmt::format_args& args) {
|
||||
const auto msg = fmt::vformat(fmt, args);
|
||||
|
||||
for (auto& Handler : Handlers) {
|
||||
Handler(level, msg.c_str());
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace Msg
|
||||
} // namespace LogMan
|
||||
+65
-11
@@ -14,30 +14,48 @@ namespace FEXCore::Threads {
|
||||
void *Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
std::mutex StackPoolMutex{};
|
||||
std::deque<StackPoolItem> StackPool;
|
||||
std::mutex DeadStackPoolMutex{};
|
||||
std::mutex LiveStackPoolMutex{};
|
||||
|
||||
std::deque<StackPoolItem> DeadStackPool;
|
||||
std::deque<StackPoolItem> LiveStackPool;
|
||||
|
||||
void *AllocateStackObject(size_t Size) {
|
||||
std::unique_lock<std::mutex> lk{StackPoolMutex};
|
||||
if (StackPool.size() == 0) {
|
||||
std::lock_guard lk{DeadStackPoolMutex};
|
||||
if (DeadStackPool.size() == 0) {
|
||||
// Nothing in the pool, just allocate
|
||||
return FEXCore::Allocator::mmap(nullptr, Size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_GROWSDOWN, -1, 0);
|
||||
}
|
||||
|
||||
// Keep the first item in the stack pool
|
||||
auto Result = StackPool.front().Ptr;
|
||||
StackPool.pop_front();
|
||||
auto Result = DeadStackPool.front().Ptr;
|
||||
DeadStackPool.pop_front();
|
||||
|
||||
// Erase the rest as a garbage collection step
|
||||
for (auto &Item : StackPool) {
|
||||
for (auto &Item : DeadStackPool) {
|
||||
FEXCore::Allocator::munmap(Item.Ptr, Item.Size);
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
void AddStackToPool(void *Ptr, size_t Size) {
|
||||
std::unique_lock<std::mutex> lk{StackPoolMutex};
|
||||
StackPool.emplace_back(StackPoolItem{Ptr, Size});
|
||||
void AddStackToDeadPool(void *Ptr, size_t Size) {
|
||||
std::lock_guard lk{DeadStackPoolMutex};
|
||||
DeadStackPool.emplace_back(StackPoolItem{Ptr, Size});
|
||||
}
|
||||
|
||||
void AddStackToLivePool(void *Ptr, size_t Size) {
|
||||
std::lock_guard lk{LiveStackPoolMutex};
|
||||
LiveStackPool.emplace_back(StackPoolItem{Ptr, Size});
|
||||
}
|
||||
|
||||
void RemoveStackFromLivePool(void *Ptr) {
|
||||
std::lock_guard lk{LiveStackPoolMutex};
|
||||
for (auto it = LiveStackPool.begin(); it != LiveStackPool.end(); ++it) {
|
||||
if (it->Ptr == Ptr) {
|
||||
LiveStackPool.erase(it);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void *InitializeThread(void *Ptr);
|
||||
@@ -49,6 +67,7 @@ namespace FEXCore::Threads {
|
||||
, UserArg {Arg} {
|
||||
pthread_attr_t Attr{};
|
||||
Stack = AllocateStackObject(STACK_SIZE);
|
||||
AddStackToLivePool(Stack, STACK_SIZE);
|
||||
pthread_attr_init(&Attr);
|
||||
pthread_attr_setstack(&Attr, Stack, STACK_SIZE);
|
||||
pthread_create(&Thread, &Attr, Func, Arg);
|
||||
@@ -87,7 +106,8 @@ namespace FEXCore::Threads {
|
||||
}
|
||||
|
||||
void FreeStack() {
|
||||
AddStackToPool(Stack, STACK_SIZE);
|
||||
RemoveStackFromLivePool(Stack);
|
||||
AddStackToDeadPool(Stack, STACK_SIZE);
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -115,8 +135,38 @@ namespace FEXCore::Threads {
|
||||
return std::make_unique<PThread>(Func, Arg);
|
||||
}
|
||||
|
||||
void CleanupAfterFork_PThread() {
|
||||
// We don't need to pull the mutex here
|
||||
// After a fork we are the only thread running
|
||||
// Just need to make sure not to delete our own stack
|
||||
uintptr_t StackLocation = reinterpret_cast<uintptr_t>(alloca(0));
|
||||
|
||||
auto ClearStackPool = [&](auto &StackPool) {
|
||||
for (auto it = StackPool.begin(); it != StackPool.end(); ) {
|
||||
StackPoolItem &Item = *it;
|
||||
uintptr_t ItemStack = reinterpret_cast<uintptr_t>(Item.Ptr);
|
||||
if (ItemStack <= StackLocation && (ItemStack + Item.Size) > StackLocation) {
|
||||
// This is our stack item, skip it
|
||||
++it;
|
||||
}
|
||||
else {
|
||||
// Untracked stack. Clean it up
|
||||
FEXCore::Allocator::munmap(Item.Ptr, Item.Size);
|
||||
it = StackPool.erase(it);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Clear both dead stacks and live stacks
|
||||
ClearStackPool(DeadStackPool);
|
||||
ClearStackPool(LiveStackPool);
|
||||
|
||||
LogMan::Throw::A((DeadStackPool.size() + LiveStackPool.size()) <= 1, "After fork we should only have zero or one tracked stacks!");
|
||||
}
|
||||
|
||||
static FEXCore::Threads::Pointers Ptrs = {
|
||||
.CreateThread = CreateThread_PThread,
|
||||
.CleanupAfterFork = CleanupAfterFork_PThread,
|
||||
};
|
||||
|
||||
std::unique_ptr<FEXCore::Threads::Thread> FEXCore::Threads::Thread::Create(
|
||||
@@ -125,6 +175,10 @@ namespace FEXCore::Threads {
|
||||
return Ptrs.CreateThread(Func, Arg);
|
||||
}
|
||||
|
||||
void FEXCore::Threads::Thread::CleanupAfterFork() {
|
||||
return Ptrs.CleanupAfterFork();
|
||||
}
|
||||
|
||||
void FEXCore::Threads::Thread::SetInternalPointers(Pointers const &_Ptrs) {
|
||||
memcpy(&Ptrs, &_Ptrs, sizeof(FEXCore::Threads::Pointers));
|
||||
}
|
||||
|
||||
+31
-29
@@ -1,5 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <list>
|
||||
@@ -54,15 +56,15 @@ namespace Type {
|
||||
#undef P
|
||||
}
|
||||
|
||||
__attribute__((visibility("default"))) std::string GetDataDirectory();
|
||||
__attribute__((visibility("default"))) std::string GetConfigDirectory(bool Global);
|
||||
__attribute__((visibility("default"))) std::string GetConfigFileLocation();
|
||||
__attribute__((visibility("default"))) std::string GetApplicationConfig(std::string &Filename, bool Global);
|
||||
FEX_DEFAULT_VISIBILITY std::string GetDataDirectory();
|
||||
FEX_DEFAULT_VISIBILITY std::string GetConfigDirectory(bool Global);
|
||||
FEX_DEFAULT_VISIBILITY std::string GetConfigFileLocation();
|
||||
FEX_DEFAULT_VISIBILITY std::string GetApplicationConfig(const std::string &Filename, bool Global);
|
||||
|
||||
using LayerValue = std::list<std::string>;
|
||||
using LayerOptions = std::unordered_map<ConfigOption, LayerValue>;
|
||||
|
||||
class __attribute__((visibility("default"))) Layer {
|
||||
class FEX_DEFAULT_VISIBILITY Layer {
|
||||
public:
|
||||
explicit Layer(const LayerType _Type);
|
||||
virtual ~Layer();
|
||||
@@ -94,57 +96,57 @@ namespace Type {
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string Data) {
|
||||
OptionMap[Option].emplace_back(Data);
|
||||
OptionMap[Option].emplace_back(std::move(Data));
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string Data) {
|
||||
OptionMap.erase(Option);
|
||||
OptionMap[Option].emplace_back(Data);
|
||||
Erase(Option);
|
||||
Set(Option, std::move(Data));
|
||||
}
|
||||
|
||||
void Erase(ConfigOption Option) {
|
||||
OptionMap.erase(Option);
|
||||
}
|
||||
|
||||
const LayerType GetLayerType() const { return Type; }
|
||||
const LayerOptions &GetOptionMap() { return OptionMap; }
|
||||
LayerType GetLayerType() const { return Type; }
|
||||
const LayerOptions &GetOptionMap() const { return OptionMap; }
|
||||
|
||||
protected:
|
||||
const LayerType Type;
|
||||
LayerOptions OptionMap;
|
||||
};
|
||||
|
||||
__attribute__((visibility("default"))) void Initialize();
|
||||
__attribute__((visibility("default"))) void Shutdown();
|
||||
FEX_DEFAULT_VISIBILITY void Initialize();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown();
|
||||
|
||||
__attribute__((visibility("default"))) void Load();
|
||||
__attribute__((visibility("default"))) void ReloadMetaLayer();
|
||||
FEX_DEFAULT_VISIBILITY void Load();
|
||||
FEX_DEFAULT_VISIBILITY void ReloadMetaLayer();
|
||||
|
||||
__attribute__((visibility("default"))) void AddLayer(std::unique_ptr<FEXCore::Config::Layer> _Layer);
|
||||
FEX_DEFAULT_VISIBILITY void AddLayer(std::unique_ptr<FEXCore::Config::Layer> _Layer);
|
||||
|
||||
__attribute__((visibility("default"))) bool Exists(ConfigOption Option);
|
||||
__attribute__((visibility("default"))) std::optional<LayerValue*> All(ConfigOption Option);
|
||||
__attribute__((visibility("default"))) std::optional<std::string*> Get(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY bool Exists(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<LayerValue*> All(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<std::string*> Get(ConfigOption Option);
|
||||
|
||||
__attribute__((visibility("default"))) void Set(ConfigOption Option, std::string Data);
|
||||
__attribute__((visibility("default"))) void Erase(ConfigOption Option);
|
||||
__attribute__((visibility("default"))) void EraseSet(ConfigOption Option, std::string Data);
|
||||
FEX_DEFAULT_VISIBILITY void Set(ConfigOption Option, std::string Data);
|
||||
FEX_DEFAULT_VISIBILITY void Erase(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY void EraseSet(ConfigOption Option, std::string Data);
|
||||
|
||||
template<typename T>
|
||||
class __attribute__((visibility("default"))) Value {
|
||||
class FEX_DEFAULT_VISIBILITY Value {
|
||||
public:
|
||||
template <typename TT = T,
|
||||
typename std::enable_if<!std::is_same<TT, std::string>::value, int>::type = 0>
|
||||
Value(FEXCore::Config::ConfigOption _Option, T Default)
|
||||
: Option {_Option} {
|
||||
ValueData = FEXCore::Config::Value<T>::GetIfExists(Option, Default);
|
||||
ValueData = GetIfExists(Option, Default);
|
||||
}
|
||||
|
||||
template <typename TT = T,
|
||||
typename std::enable_if<std::is_same<TT, std::string>::value, int>::type = 0>
|
||||
Value(FEXCore::Config::ConfigOption _Option, T Default)
|
||||
: Option {_Option} {
|
||||
ValueData = FEXCore::Config::Value<T>::GetIfExists(Option, Default);
|
||||
ValueData = GetIfExists(Option, Default);
|
||||
GetListIfExists(Option, &AppendList);
|
||||
}
|
||||
|
||||
@@ -156,7 +158,7 @@ namespace Type {
|
||||
ERROR_AND_DIE("FEXCore::Config::Value has no value");
|
||||
}
|
||||
|
||||
ValueData = FEXCore::Config::Value<T>::Get(Option);
|
||||
ValueData = Get(Option);
|
||||
}
|
||||
|
||||
template <typename TT = T,
|
||||
@@ -167,13 +169,13 @@ namespace Type {
|
||||
ERROR_AND_DIE("FEXCore::Config::Value has no value");
|
||||
}
|
||||
|
||||
ValueData = FEXCore::Config::Value<T>::GetIfExists(Option);
|
||||
ValueData = GetIfExists(Option);
|
||||
GetListIfExists(Option, &AppendList);
|
||||
}
|
||||
|
||||
operator T() { return ValueData; }
|
||||
T operator()() { return ValueData; }
|
||||
Value<T>(T Value) { ValueData = Value; }
|
||||
operator T() const { return ValueData; }
|
||||
T operator()() const { return ValueData; }
|
||||
Value<T>(T Value) { ValueData = std::move(Value); }
|
||||
std::list<T> &All() { return AppendList; }
|
||||
|
||||
private:
|
||||
|
||||
+6
-3
@@ -6,7 +6,10 @@ $end_info$
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
#include <stdint.h>
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -85,8 +88,8 @@ class LLVMCore;
|
||||
virtual void ClearCache() {}
|
||||
virtual void CopyNecessaryDataForCompileThread(CPUBackend *Original) {}
|
||||
|
||||
using AsmDispatch = __attribute__((naked)) void(*)(FEXCore::Core::CpuStateFrame *Frame);
|
||||
using JITCallback = __attribute__((naked)) void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
using AsmDispatch = FEX_NAKED void(*)(FEXCore::Core::CpuStateFrame *Frame);
|
||||
using JITCallback = FEX_NAKED void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
|
||||
JITCallback CallbackPtr{};
|
||||
protected:
|
||||
|
||||
+5
-1
@@ -16,6 +16,10 @@ class IREmitter;
|
||||
*/
|
||||
class CodeLoader {
|
||||
public:
|
||||
using MapperFn = std::function<void *(void *addr, size_t length, int prot, int flags, int fd, off_t offset)>;
|
||||
using UnmapperFn = std::function<int(void *addr, size_t length)>;
|
||||
|
||||
virtual ~CodeLoader() = default;
|
||||
|
||||
/**
|
||||
* @brief CPU Core uses this to choose what the stack size should be for this code
|
||||
@@ -35,7 +39,7 @@ public:
|
||||
/**
|
||||
* @brief Maps and copies the executable, also sets up stack
|
||||
*/
|
||||
virtual bool MapMemory(std::function<void *(void *addr, size_t length, int prot, int flags, int fd, off_t offset)> Mapper, std::function<int(void *addr, size_t length)> Unmapper) { return false; }
|
||||
virtual bool MapMemory(const MapperFn& Mapper, const UnmapperFn& Unmapper) { return false; }
|
||||
|
||||
virtual std::vector<std::string> const *GetApplicationArguments() { return nullptr; }
|
||||
virtual void GetExecveArguments(std::vector<char const*> *Args) {}
|
||||
|
||||
+47
-41
@@ -5,6 +5,7 @@
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <istream>
|
||||
#include <ostream>
|
||||
@@ -45,12 +46,16 @@ namespace FEXCore::Context {
|
||||
MODE_32BIT,
|
||||
MODE_64BIT,
|
||||
};
|
||||
using CustomCPUFactoryType = std::function<FEXCore::CPU::CPUBackend* (FEXCore::Context::Context*, FEXCore::Core::InternalThreadState *Thread)>;
|
||||
|
||||
using CustomCPUFactoryType = std::function<std::unique_ptr<FEXCore::CPU::CPUBackend> (FEXCore::Context::Context*, FEXCore::Core::InternalThreadState *Thread)>;
|
||||
|
||||
using ExitHandler = std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)>;
|
||||
|
||||
/**
|
||||
* @brief This initializes internal FEXCore state that is shared between contexts and requires overhead to setup
|
||||
*/
|
||||
__attribute__((visibility("default"))) void InitializeStaticTables(OperatingMode Mode = MODE_64BIT);
|
||||
FEX_DEFAULT_VISIBILITY void InitializeStaticTables(OperatingMode Mode = MODE_64BIT);
|
||||
FEX_DEFAULT_VISIBILITY void ShutdownStaticTables();
|
||||
|
||||
/**
|
||||
* @brief [[threadsafe]] Create a new FEXCore context object
|
||||
@@ -59,7 +64,7 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return a new context object
|
||||
*/
|
||||
__attribute__((visibility("default"))) FEXCore::Context::Context *CreateNewContext();
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::Context::Context *CreateNewContext();
|
||||
|
||||
/**
|
||||
* @brief Post creation context initialization
|
||||
@@ -69,14 +74,14 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return true if we managed to initialize correctly
|
||||
*/
|
||||
__attribute__((visibility("default"))) bool InitializeContext(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY bool InitializeContext(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Destroy the context object
|
||||
*
|
||||
* @param CTX
|
||||
*/
|
||||
__attribute__((visibility("default"))) void DestroyContext(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY void DestroyContext(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Allows setting up in memory code and other things prior to launchign code execution
|
||||
@@ -86,17 +91,17 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return true if we loaded code
|
||||
*/
|
||||
__attribute__((visibility("default"))) bool InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader);
|
||||
FEX_DEFAULT_VISIBILITY bool InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader);
|
||||
|
||||
__attribute__((visibility("default"))) void SetExitHandler(FEXCore::Context::Context *CTX, std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> handler);
|
||||
__attribute__((visibility("default"))) std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)> GetExitHandler(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler);
|
||||
FEX_DEFAULT_VISIBILITY ExitHandler GetExitHandler(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Pauses execution on the CPU core
|
||||
*
|
||||
* Blocks until all threads have paused.
|
||||
*/
|
||||
__attribute__((visibility("default"))) void Pause(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY void Pause(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Starts (or continues) the CPU core
|
||||
@@ -105,7 +110,7 @@ namespace FEXCore::Context {
|
||||
* Use RunUntilExit() for synchonous executions
|
||||
*
|
||||
*/
|
||||
__attribute__((visibility("default"))) void Run(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY void Run(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Runs the CPU core until it exits
|
||||
@@ -117,9 +122,9 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return The ExitReason for the parentthread.
|
||||
*/
|
||||
__attribute__((visibility("default"))) ExitReason RunUntilExit(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY ExitReason RunUntilExit(FEXCore::Context::Context *CTX);
|
||||
|
||||
__attribute__((visibility("default"))) void CompileRIP(FEXCore::Context::Context *CTX, uint64_t GuestRIP);
|
||||
FEX_DEFAULT_VISIBILITY void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
|
||||
/**
|
||||
* @brief Gets the program exit status
|
||||
@@ -129,21 +134,21 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return The program exit status
|
||||
*/
|
||||
__attribute__((visibility("default"))) int GetProgramStatus(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY int GetProgramStatus(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Tells the core to shutdown
|
||||
*
|
||||
* Blocks until shutdown
|
||||
*/
|
||||
__attribute__((visibility("default"))) void Stop(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY void Stop(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Executes one instruction
|
||||
*
|
||||
* Returns once execution is complete.
|
||||
*/
|
||||
__attribute__((visibility("default"))) void Step(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY void Step(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief [[threadsafe]] Returns the ExitReason of the parent thread. Typically used for async result status
|
||||
@@ -152,7 +157,7 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return The ExitReason for the parentthread
|
||||
*/
|
||||
__attribute__((visibility("default"))) ExitReason GetExitReason(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY ExitReason GetExitReason(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief [[theadsafe]] Checks if the Context is either done working or paused(in the case of single stepping)
|
||||
@@ -163,7 +168,7 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return true if the core is done or paused
|
||||
*/
|
||||
__attribute__((visibility("default"))) bool IsDone(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY bool IsDone(FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Gets a copy the CPUState of the parent thread
|
||||
@@ -171,7 +176,7 @@ namespace FEXCore::Context {
|
||||
* @param CTX The context that we created
|
||||
* @param State The state object to populate
|
||||
*/
|
||||
__attribute__((visibility("default"))) void GetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State);
|
||||
FEX_DEFAULT_VISIBILITY void GetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State);
|
||||
|
||||
/**
|
||||
* @brief Copies the CPUState provided to the parent thread
|
||||
@@ -179,7 +184,7 @@ namespace FEXCore::Context {
|
||||
* @param CTX The context that we created
|
||||
* @param State The satate object to copy from
|
||||
*/
|
||||
__attribute__((visibility("default"))) void SetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State);
|
||||
FEX_DEFAULT_VISIBILITY void SetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State);
|
||||
|
||||
/**
|
||||
* @brief Allows the frontend to pass in a custom CPUBackend creation factory
|
||||
@@ -189,7 +194,7 @@ namespace FEXCore::Context {
|
||||
* @param CTX The context that we created
|
||||
* @param Factory The factory that the context will call if the DefaultCore config ise set to CUSTOM
|
||||
*/
|
||||
__attribute__((visibility("default"))) void SetCustomCPUBackendFactory(FEXCore::Context::Context *CTX, CustomCPUFactoryType Factory);
|
||||
FEX_DEFAULT_VISIBILITY void SetCustomCPUBackendFactory(FEXCore::Context::Context *CTX, CustomCPUFactoryType Factory);
|
||||
|
||||
/**
|
||||
* @brief Sets up memory regions on the guest for mirroring within the guest's VM space
|
||||
@@ -200,7 +205,7 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return true when successfully mapped. false if there was an error adding
|
||||
*/
|
||||
__attribute__((visibility("default"))) bool AddVirtualMemoryMapping(FEXCore::Context::Context *CTX, uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size);
|
||||
FEX_DEFAULT_VISIBILITY bool AddVirtualMemoryMapping(FEXCore::Context::Context *CTX, uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size);
|
||||
|
||||
/**
|
||||
* @brief Allows the frontend to set a custom syscall handler
|
||||
@@ -210,29 +215,30 @@ namespace FEXCore::Context {
|
||||
* @param Syscall Which syscall ID to install a visitor to
|
||||
* @param Visitor The Visitor to install
|
||||
*/
|
||||
__attribute__((visibility("default"))) void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, uint64_t Syscall, FEXCore::HLE::SyscallVisitor *Visitor);
|
||||
FEX_DEFAULT_VISIBILITY void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, uint64_t Syscall, FEXCore::HLE::SyscallVisitor *Visitor);
|
||||
|
||||
__attribute__((visibility("default"))) void HandleCallback(FEXCore::Context::Context *CTX, uint64_t RIP);
|
||||
FEX_DEFAULT_VISIBILITY void HandleCallback(FEXCore::Context::Context *CTX, uint64_t RIP);
|
||||
|
||||
__attribute__((visibility("default"))) void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func);
|
||||
__attribute__((visibility("default"))) void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func);
|
||||
FEX_DEFAULT_VISIBILITY void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
FEX_DEFAULT_VISIBILITY void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
__attribute__((visibility("default"))) FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
|
||||
__attribute__((visibility("default"))) void InitializeThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
__attribute__((visibility("default"))) void RunThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
__attribute__((visibility("default"))) void StopThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
__attribute__((visibility("default"))) void DestroyThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
__attribute__((visibility("default"))) void CleanupAfterFork(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
__attribute__((visibility("default"))) void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation);
|
||||
__attribute__((visibility("default"))) void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler);
|
||||
__attribute__((visibility("default"))) FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf);
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
|
||||
FEX_DEFAULT_VISIBILITY void InitializeThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
FEX_DEFAULT_VISIBILITY void RunThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
FEX_DEFAULT_VISIBILITY void StopThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
FEX_DEFAULT_VISIBILITY void DestroyThread(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
FEX_DEFAULT_VISIBILITY void CleanupAfterFork(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread);
|
||||
FEX_DEFAULT_VISIBILITY void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation);
|
||||
FEX_DEFAULT_VISIBILITY void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler);
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf);
|
||||
|
||||
__attribute__((visibility("default"))) void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name);
|
||||
__attribute__((visibility("default"))) void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length);
|
||||
__attribute__((visibility("default"))) void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader);
|
||||
__attribute__((visibility("default"))) bool WriteAOTIR(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
|
||||
__attribute__((visibility("default"))) void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer);
|
||||
__attribute__((visibility("default"))) void FlushCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length);
|
||||
FEX_DEFAULT_VISIBILITY void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name);
|
||||
FEX_DEFAULT_VISIBILITY void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length);
|
||||
FEX_DEFAULT_VISIBILITY void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader);
|
||||
FEX_DEFAULT_VISIBILITY void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
|
||||
FEX_DEFAULT_VISIBILITY void FinalizeAOTIRCache(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer);
|
||||
FEX_DEFAULT_VISIBILITY void FlushCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length);
|
||||
|
||||
__attribute__((visibility("default"))) void ConfigureAOTGen(FEXCore::Context::Context *CTX, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress);
|
||||
FEX_DEFAULT_VISIBILITY void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress);
|
||||
}
|
||||
+6
-3
@@ -1,12 +1,15 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstddef>
|
||||
#include <stdint.h>
|
||||
#include <string_view>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct __attribute__((packed)) CPUState {
|
||||
struct FEX_PACKED CPUState {
|
||||
uint64_t rip; ///< Current core's RIP. May not be entirely accurate while JIT is active
|
||||
uint64_t gregs[16];
|
||||
uint64_t : 64;
|
||||
@@ -51,6 +54,6 @@ namespace FEXCore::Core {
|
||||
|
||||
constexpr uint64_t PAGE_SIZE = 4096;
|
||||
|
||||
__attribute__((visibility("default"))) std::string_view const& GetFlagName(unsigned Flag);
|
||||
__attribute__((visibility("default"))) std::string_view const& GetGRegName(unsigned Reg);
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetFlagName(unsigned Flag);
|
||||
FEX_DEFAULT_VISIBILITY std::string_view const& GetGRegName(unsigned Reg);
|
||||
}
|
||||
+9
-4
@@ -1,4 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <signal.h>
|
||||
@@ -7,11 +10,11 @@ namespace FEXCore {
|
||||
namespace Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
struct __attribute__((packed)) GuestSAMask {
|
||||
struct FEX_PACKED GuestSAMask {
|
||||
uint64_t Val;
|
||||
};
|
||||
|
||||
struct __attribute__((packed)) GuestSigAction {
|
||||
struct FEX_PACKED GuestSigAction {
|
||||
union {
|
||||
void (*handler)(int);
|
||||
void (*sigaction)(int, siginfo_t *, void*);
|
||||
@@ -27,6 +30,8 @@ namespace Core {
|
||||
|
||||
class SignalDelegator {
|
||||
public:
|
||||
virtual ~SignalDelegator() = default;
|
||||
|
||||
/**
|
||||
* @brief Registers an emulated thread's object to a TLS object
|
||||
*
|
||||
@@ -49,8 +54,8 @@ namespace Core {
|
||||
*
|
||||
* It's a process level signal handler so one must be careful
|
||||
*/
|
||||
virtual void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func) = 0;
|
||||
virtual void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func) = 0;
|
||||
virtual void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) = 0;
|
||||
virtual void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal specifically for guest handling
|
||||
|
||||
+28
-8
@@ -1,4 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
@@ -13,7 +16,7 @@ namespace FEXCore {
|
||||
constexpr uint64_t UC_STRICT_RESTORE_SS = (1ULL << 2);
|
||||
|
||||
///< Describes the signal stack
|
||||
struct __attribute__((packed)) stack_t {
|
||||
struct FEX_PACKED stack_t {
|
||||
void *ss_sp;
|
||||
int32_t ss_flags;
|
||||
uint32_t : 32;
|
||||
@@ -21,7 +24,7 @@ namespace FEXCore {
|
||||
};
|
||||
static_assert(sizeof(FEXCore::x86_64::stack_t) == 24, "This needs to be the right size");
|
||||
|
||||
struct __attribute__((packed)) _libc_fpstate {
|
||||
struct FEX_PACKED _libc_fpstate {
|
||||
// This is in FXSAVE format
|
||||
uint16_t fcw;
|
||||
uint16_t fsw;
|
||||
@@ -65,19 +68,19 @@ namespace FEXCore {
|
||||
};
|
||||
static_assert(FEX_REG_CR2 == 22, "Oops");
|
||||
|
||||
struct __attribute__((packed)) mcontext_t {
|
||||
struct FEX_PACKED mcontext_t {
|
||||
uint64_t gregs[23];
|
||||
FEXCore::x86_64::_libc_fpstate *fpregs;
|
||||
uint64_t __reserved[8];
|
||||
};
|
||||
static_assert(sizeof(FEXCore::x86_64::mcontext_t) == 256, "This needs to be the right size");
|
||||
|
||||
struct __attribute__((packed)) sigset_t {
|
||||
struct FEX_PACKED sigset_t {
|
||||
uint64_t val[16];
|
||||
};
|
||||
static_assert(sizeof(FEXCore::x86_64::sigset_t) == 128, "This needs to be the right size");
|
||||
|
||||
struct __attribute__((packed)) ucontext_t {
|
||||
struct FEX_PACKED ucontext_t {
|
||||
uint64_t uc_flags;
|
||||
FEXCore::x86_64::ucontext_t *uc_link;
|
||||
FEXCore::x86_64::stack_t uc_stack;
|
||||
@@ -92,12 +95,29 @@ namespace FEXCore {
|
||||
}
|
||||
|
||||
namespace x86 {
|
||||
struct __attribute__((packed)) siginfo_t {
|
||||
uint32_t pad[32];
|
||||
struct FEX_PACKED siginfo_t {
|
||||
int si_signo;
|
||||
int si_errno;
|
||||
int si_code;
|
||||
union {
|
||||
uint32_t pad[29];
|
||||
/* SIGILL, SIGFPE, SIGSEGV, SIBUS */
|
||||
struct {
|
||||
uint32_t addr;
|
||||
} _sigfault;
|
||||
/* SIGCHLD */
|
||||
struct {
|
||||
int32_t pid;
|
||||
int32_t uid;
|
||||
int32_t status;
|
||||
int32_t utime;
|
||||
int32_t stime;
|
||||
} _sigchld;
|
||||
} _sifields;
|
||||
};
|
||||
static_assert(sizeof(FEXCore::x86::siginfo_t) == 128, "This needs to be the right size");
|
||||
|
||||
struct __attribute__((packed)) ucontext_t {
|
||||
struct FEX_PACKED ucontext_t {
|
||||
uint32_t pad[91];
|
||||
};
|
||||
static_assert(sizeof(FEXCore::x86::ucontext_t) == 364, "This needs to be the right size");
|
||||
|
||||
@@ -54,11 +54,11 @@ namespace FEXCore::Core {
|
||||
std::vector<DebugDataSubblock> Subblocks;
|
||||
};
|
||||
|
||||
enum SignalEvent {
|
||||
SIGNALEVENT_NONE, // If the guest uses our signal we need to know it was errant on our end
|
||||
SIGNALEVENT_PAUSE,
|
||||
SIGNALEVENT_STOP,
|
||||
SIGNALEVENT_RETURN,
|
||||
enum class SignalEvent {
|
||||
Nothing, // If the guest uses our signal we need to know it was errant on our end
|
||||
Pause,
|
||||
Stop,
|
||||
Return,
|
||||
};
|
||||
|
||||
struct LocalIREntry {
|
||||
@@ -78,7 +78,7 @@ namespace FEXCore::Core {
|
||||
} RunningEvents;
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
std::atomic<SignalEvent> SignalReason {SignalEvent::SIGNALEVENT_NONE};
|
||||
std::atomic<SignalEvent> SignalReason{SignalEvent::Nothing};
|
||||
|
||||
std::unique_ptr<FEXCore::Threads::Thread> ExecutionThread;
|
||||
Event StartRunning;
|
||||
@@ -107,7 +107,7 @@ namespace FEXCore::Core {
|
||||
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState{};
|
||||
|
||||
};
|
||||
static_assert(std::is_standard_layout<InternalThreadState>::value, "This needs to be standard layout");
|
||||
// static_assert(std::is_standard_layout<InternalThreadState>::value, "This needs to be standard layout");
|
||||
}
|
||||
|
||||
|
||||
+80
-58
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
@@ -99,54 +100,72 @@ inline void PopOpAddrIf(uint32_t *Flags, uint32_t Flag) {
|
||||
|
||||
}
|
||||
|
||||
union DecodedOperand {
|
||||
enum {
|
||||
TYPE_NONE,
|
||||
TYPE_GPR,
|
||||
TYPE_GPR_DIRECT,
|
||||
TYPE_GPR_INDIRECT,
|
||||
TYPE_RIP_RELATIVE,
|
||||
TYPE_LITERAL,
|
||||
TYPE_SIB,
|
||||
struct DecodedOperand {
|
||||
enum class OpType : uint8_t {
|
||||
Nothing,
|
||||
GPR,
|
||||
GPRDirect,
|
||||
GPRIndirect,
|
||||
RIPRelative,
|
||||
Literal,
|
||||
SIB,
|
||||
};
|
||||
|
||||
struct {
|
||||
uint8_t Type;
|
||||
} TypeNone;
|
||||
bool IsNone() const {
|
||||
return Type == OpType::Nothing;
|
||||
}
|
||||
bool IsGPR() const {
|
||||
return Type == OpType::GPR;
|
||||
}
|
||||
bool IsGPRDirect() const {
|
||||
return Type == OpType::GPRDirect;
|
||||
}
|
||||
bool IsGPRIndirect() const {
|
||||
return Type == OpType::GPRIndirect;
|
||||
}
|
||||
bool IsRIPRelative() const {
|
||||
return Type == OpType::RIPRelative;
|
||||
}
|
||||
bool IsLiteral() const {
|
||||
return Type == OpType::Literal;
|
||||
}
|
||||
bool IsSIB() const {
|
||||
return Type == OpType::SIB;
|
||||
}
|
||||
|
||||
struct {
|
||||
uint8_t Type;
|
||||
bool HighBits;
|
||||
uint8_t GPR;
|
||||
} TypeGPR;
|
||||
union TypeUnion {
|
||||
struct {
|
||||
bool HighBits;
|
||||
uint8_t GPR;
|
||||
} GPR;
|
||||
|
||||
struct {
|
||||
uint8_t Type;
|
||||
uint8_t GPR;
|
||||
int32_t Displacement;
|
||||
} TypeGPRIndirect;
|
||||
struct {
|
||||
uint8_t GPR;
|
||||
int32_t Displacement;
|
||||
} GPRIndirect;
|
||||
|
||||
struct {
|
||||
uint8_t Type;
|
||||
union {
|
||||
int32_t s;
|
||||
uint32_t u;
|
||||
struct {
|
||||
union {
|
||||
int32_t s;
|
||||
uint32_t u;
|
||||
} Value;
|
||||
} RIPLiteral;
|
||||
|
||||
struct {
|
||||
uint8_t Size;
|
||||
uint64_t Value;
|
||||
} Literal;
|
||||
} TypeRIPLiteral;
|
||||
|
||||
struct {
|
||||
uint8_t Type;
|
||||
uint8_t Size;
|
||||
uint64_t Literal;
|
||||
} TypeLiteral;
|
||||
struct {
|
||||
uint8_t Index; // ~0 invalid
|
||||
uint8_t Base; // ~0 invalid
|
||||
uint32_t Scale : 8;
|
||||
int32_t Offset;
|
||||
} SIB;
|
||||
};
|
||||
|
||||
struct {
|
||||
uint8_t Type;
|
||||
uint8_t Index; // ~0 invalid
|
||||
uint8_t Base; // ~0 invalid
|
||||
uint32_t Scale : 8;
|
||||
int32_t Offset;
|
||||
} TypeSIB;
|
||||
OpType Type;
|
||||
TypeUnion Data;
|
||||
};
|
||||
|
||||
struct DecodedInst {
|
||||
@@ -418,6 +437,9 @@ struct X86InstInfo {
|
||||
// We don't care if the opcode dispatcher differs
|
||||
return true;
|
||||
}
|
||||
bool operator!=(const X86InstInfo &b) const {
|
||||
return !operator==(b);
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial<X86InstInfo>::value, "X86InstInfo needs to be trivial");
|
||||
@@ -455,29 +477,29 @@ constexpr size_t MAX_XOP_GROUP_TABLE_SIZE = (1 << 6);
|
||||
|
||||
constexpr size_t MAX_EVEX_TABLE_SIZE = 256;
|
||||
|
||||
extern __attribute__((visibility("default"))) X86InstInfo BaseOps[MAX_PRIMARY_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo SecondBaseOps[MAX_SECOND_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo RepModOps[MAX_REP_MOD_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo RepNEModOps[MAX_REPNE_MOD_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo OpSizeModOps[MAX_OPSIZE_MOD_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo PrimaryInstGroupOps[MAX_INST_GROUP_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo SecondInstGroupOps[MAX_INST_SECOND_GROUP_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo SecondModRMTableOps[MAX_SECOND_MODRM_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo X87Ops[MAX_X87_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo DDDNowOps[MAX_3DNOW_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo H0F38TableOps[MAX_0F_38_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo H0F3ATableOps[MAX_0F_3A_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo BaseOps[MAX_PRIMARY_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo SecondBaseOps[MAX_SECOND_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo RepModOps[MAX_REP_MOD_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo RepNEModOps[MAX_REPNE_MOD_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo OpSizeModOps[MAX_OPSIZE_MOD_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo PrimaryInstGroupOps[MAX_INST_GROUP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo SecondInstGroupOps[MAX_INST_SECOND_GROUP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo SecondModRMTableOps[MAX_SECOND_MODRM_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo X87Ops[MAX_X87_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo DDDNowOps[MAX_3DNOW_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo H0F38TableOps[MAX_0F_38_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo H0F3ATableOps[MAX_0F_3A_TABLE_SIZE];
|
||||
|
||||
// VEX
|
||||
extern __attribute__((visibility("default"))) X86InstInfo VEXTableOps[MAX_VEX_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo VEXTableGroupOps[MAX_VEX_GROUP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo VEXTableOps[MAX_VEX_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo VEXTableGroupOps[MAX_VEX_GROUP_TABLE_SIZE];
|
||||
|
||||
// XOP
|
||||
extern __attribute__((visibility("default"))) X86InstInfo XOPTableOps[MAX_XOP_TABLE_SIZE];
|
||||
extern __attribute__((visibility("default"))) X86InstInfo XOPTableGroupOps[MAX_XOP_GROUP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo XOPTableOps[MAX_XOP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo XOPTableGroupOps[MAX_XOP_GROUP_TABLE_SIZE];
|
||||
|
||||
// EVEX
|
||||
extern __attribute__((visibility("default"))) X86InstInfo EVEXTableOps[MAX_EVEX_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo EVEXTableOps[MAX_EVEX_TABLE_SIZE];
|
||||
|
||||
__attribute__((visibility("default"))) void InitializeInfoTables(Context::OperatingMode Mode);
|
||||
FEX_DEFAULT_VISIBILITY void InitializeInfoTables(Context::OperatingMode Mode);
|
||||
}
|
||||
@@ -7,12 +7,12 @@ namespace FEXCore::HLE {
|
||||
// Tracking relationships between thread IDs and such
|
||||
class ThreadManagement {
|
||||
public:
|
||||
uint64_t GetUID() { return UID; }
|
||||
uint64_t GetGID() { return GID; }
|
||||
uint64_t GetEUID() { return EUID; }
|
||||
uint64_t GetEGID() { return EGID; }
|
||||
uint64_t GetTID() { return TID; }
|
||||
uint64_t GetPID() { return PID; }
|
||||
uint64_t GetUID() const { return UID; }
|
||||
uint64_t GetGID() const { return GID; }
|
||||
uint64_t GetEUID() const { return EUID; }
|
||||
uint64_t GetEGID() const { return EGID; }
|
||||
uint64_t GetTID() const { return TID; }
|
||||
uint64_t GetPID() const { return PID; }
|
||||
|
||||
uint64_t UID{1000};
|
||||
uint64_t GID{1000};
|
||||
|
||||
+37
-29
@@ -1,8 +1,12 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <array>
|
||||
#include <cassert>
|
||||
#include <cstdint>
|
||||
#include <string.h>
|
||||
#include <cstring>
|
||||
#include <memory>
|
||||
#include <sstream>
|
||||
#include <tuple>
|
||||
|
||||
@@ -70,11 +74,10 @@ struct NodeWrapperBase final {
|
||||
Type const *GetNode(uintptr_t Base) const { return reinterpret_cast<Type*>(Base + NodeOffset); }
|
||||
|
||||
void SetOffset(uintptr_t Base, uintptr_t Value) { NodeOffset = Value - Base; }
|
||||
constexpr bool operator==(NodeWrapperBase<Type> const &rhs) const { return NodeOffset == rhs.NodeOffset; }
|
||||
constexpr bool operator!=(NodeWrapperBase<Type> const &rhs) const { return !operator==(rhs); }
|
||||
friend constexpr bool operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial<NodeWrapperBase<OrderedNode>>::value);
|
||||
static_assert(std::is_trivial_v<NodeWrapperBase<OrderedNode>>);
|
||||
|
||||
static_assert(sizeof(NodeWrapperBase<OrderedNode>) == sizeof(uint32_t));
|
||||
|
||||
@@ -252,76 +255,81 @@ class OrderedNode final {
|
||||
void SetUses(uint32_t Uses) { NumUses = Uses; }
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial<OrderedNode>::value);
|
||||
static_assert(std::is_trivially_copyable<OrderedNode>::value);
|
||||
static_assert(std::is_trivial_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_copyable_v<OrderedNode>);
|
||||
static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
|
||||
struct RegisterClassType final {
|
||||
uint32_t Val;
|
||||
operator uint32_t() {
|
||||
constexpr operator uint32_t() const {
|
||||
return Val;
|
||||
}
|
||||
constexpr bool operator==(RegisterClassType const &rhs) const { return Val == rhs.Val; }
|
||||
constexpr bool operator!=(RegisterClassType const &rhs) const { return !operator==(rhs); }
|
||||
friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
|
||||
};
|
||||
|
||||
struct CondClassType final {
|
||||
uint8_t Val;
|
||||
operator uint8_t() {
|
||||
constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
friend constexpr bool operator==(const CondClassType&, const CondClassType&) = default;
|
||||
};
|
||||
|
||||
struct MemOffsetType final {
|
||||
uint8_t Val;
|
||||
operator uint8_t() {
|
||||
constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
int operator ==(const MemOffsetType other) {
|
||||
return Val == other.Val;
|
||||
}
|
||||
int operator !=(const MemOffsetType other) {
|
||||
return Val != other.Val;
|
||||
}
|
||||
friend constexpr bool operator==(const MemOffsetType&, const MemOffsetType&) = default;
|
||||
};
|
||||
|
||||
struct TypeDefinition final {
|
||||
uint16_t Val;
|
||||
operator uint16_t() const {
|
||||
|
||||
constexpr operator uint16_t() const {
|
||||
return Val;
|
||||
}
|
||||
|
||||
static TypeDefinition Create(uint8_t Bytes) {
|
||||
static constexpr TypeDefinition Create(uint8_t Bytes) {
|
||||
TypeDefinition Type{};
|
||||
Type.Val = Bytes << 8;
|
||||
return Type;
|
||||
}
|
||||
|
||||
static TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
|
||||
static constexpr TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
|
||||
TypeDefinition Type{};
|
||||
Type.Val = (Bytes << 8) | (Elements & 255);
|
||||
return Type;
|
||||
}
|
||||
|
||||
uint8_t Bytes() const {
|
||||
constexpr uint8_t Bytes() const {
|
||||
return Val >> 8;
|
||||
}
|
||||
|
||||
uint8_t Elements() const {
|
||||
constexpr uint8_t Elements() const {
|
||||
return Val & 255;
|
||||
}
|
||||
|
||||
friend constexpr bool operator==(const TypeDefinition&, const TypeDefinition&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial<TypeDefinition>::value);
|
||||
static_assert(std::is_trivial_v<TypeDefinition>);
|
||||
|
||||
struct FenceType final {
|
||||
uint8_t Val;
|
||||
operator uint8_t() const {
|
||||
constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
constexpr bool operator==(FenceType const &rhs) const { return Val == rhs.Val; }
|
||||
constexpr bool operator!=(FenceType const &rhs) const { return !operator==(rhs); }
|
||||
friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
|
||||
};
|
||||
|
||||
struct RoundType final {
|
||||
uint8_t Val;
|
||||
constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
friend constexpr bool operator==(const RoundType&, const RoundType&) = default;
|
||||
};
|
||||
|
||||
struct SHA256Sum final {
|
||||
@@ -383,7 +391,7 @@ public:
|
||||
return { RealNode, RealNode->Op(IRList) };
|
||||
}
|
||||
|
||||
uint32_t ID() {
|
||||
uint32_t ID() const {
|
||||
return Node.ID();
|
||||
}
|
||||
|
||||
@@ -463,8 +471,8 @@ public:
|
||||
class IRListView;
|
||||
class IREmitter;
|
||||
|
||||
__attribute__((visibility("default"))) void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData);
|
||||
__attribute__((visibility("default"))) IREmitter* Parse(std::istream *in);
|
||||
FEX_DEFAULT_VISIBILITY void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData);
|
||||
FEX_DEFAULT_VISIBILITY std::unique_ptr<IREmitter> Parse(std::istream *in);
|
||||
|
||||
template<typename Type>
|
||||
inline uint32_t NodeWrapperBase<Type>::ID() const { return NodeOffset / sizeof(IR::OrderedNode); }
|
||||
|
||||
+29
-4
@@ -111,9 +111,15 @@ friend class FEXCore::IR::PassManager;
|
||||
IRPair<IROp_VExtractElement> _VExtractElement(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t Index) {
|
||||
return _VExtractElement(ssa0, Index, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VDupElement> _VDupElement(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t Index) {
|
||||
return _VDupElement(ssa0, Index, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VAnd> _VAnd(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VAnd(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VBic> _VBic(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VBic(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VOr> _VOr(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VOr(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
@@ -144,12 +150,21 @@ friend class FEXCore::IR::PassManager;
|
||||
IRPair<IROp_VAddV> _VAddV(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
|
||||
return _VAddV(ssa0, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VUMinV> _VUMinV(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
|
||||
return _VUMinV(ssa0, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VURAvg> _VURAvg(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VURAvg(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VAbs> _VAbs(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
|
||||
return _VAbs(ssa0, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VPopcount> _VPopcount(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
|
||||
return _VPopcount(ssa0, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VFMul> _VFMul(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VFMul(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VUMin> _VUMin(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VUMin(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
@@ -168,6 +183,12 @@ friend class FEXCore::IR::PassManager;
|
||||
IRPair<IROp_VZip2> _VZip2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VZip2(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VUnZip> _VUnZip(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VUnZip(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VUnZip2> _VUnZip2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VUnZip2(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VCMPEQ> _VCMPEQ(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VCMPEQ(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
@@ -294,9 +315,6 @@ friend class FEXCore::IR::PassManager;
|
||||
IRPair<IROp_Vector_FToF> _Vector_FToF(uint8_t RegisterSize, uint8_t DstElementSize, uint8_t SrcElementSize, OrderedNode *ssa0) {
|
||||
return _Vector_FToF(ssa0, SrcElementSize, RegisterSize, DstElementSize);
|
||||
}
|
||||
IRPair<IROp_Float_FromGPR_U> _Float_FromGPR_U(uint8_t DstElementSize, uint8_t SrcElementSize, OrderedNode *ssa0) {
|
||||
return _Float_FromGPR_U(ssa0, SrcElementSize, DstElementSize);
|
||||
}
|
||||
IRPair<IROp_Float_FromGPR_S> _Float_FromGPR_S(uint8_t DstElementSize, uint8_t SrcElementSize, OrderedNode *ssa0) {
|
||||
return _Float_FromGPR_S(ssa0, SrcElementSize, DstElementSize);
|
||||
}
|
||||
@@ -321,6 +339,9 @@ friend class FEXCore::IR::PassManager;
|
||||
IRPair<IROp_VSMull2> _VSMull2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VSMull2(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VUABDL> _VUABDL(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
|
||||
return _VUABDL(ssa0, ssa1, RegisterSize, ElementSize);
|
||||
}
|
||||
IRPair<IROp_VSXTL> _VSXTL(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
|
||||
return _VSXTL(ssa0, RegisterSize, ElementSize);
|
||||
}
|
||||
@@ -562,7 +583,11 @@ friend class FEXCore::IR::PassManager;
|
||||
* @{ */
|
||||
/** @} */
|
||||
void LinkCodeBlocks(OrderedNode *CodeNode, OrderedNode *Next) {
|
||||
FEXCore::IR::IROp_CodeBlock *CurrentIROp = CodeNode->Op(DualListData.DataBegin())->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
FEXCore::IR::IROp_CodeBlock *CurrentIROp =
|
||||
#endif
|
||||
CodeNode->Op(DualListData.DataBegin())->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
LOGMAN_THROW_A(CurrentIROp->Header.Op == IROps::OP_CODEBLOCK, "Invalid");
|
||||
|
||||
CodeNode->append(DualListData.ListBegin(), Next);
|
||||
|
||||
+7
-7
@@ -35,12 +35,12 @@ class DualIntrusiveAllocator final {
|
||||
FEXCore::Allocator::free(reinterpret_cast<void*>(Data));
|
||||
}
|
||||
|
||||
bool DataCheckSize(size_t Size) {
|
||||
bool DataCheckSize(size_t Size) const {
|
||||
size_t NewOffset = DataCurrentOffset + Size;
|
||||
return NewOffset <= MemorySize;
|
||||
}
|
||||
|
||||
bool ListCheckSize(size_t Size) {
|
||||
bool ListCheckSize(size_t Size) const {
|
||||
size_t NewOffset = ListCurrentOffset + Size;
|
||||
return NewOffset <= MemorySize;
|
||||
}
|
||||
@@ -69,8 +69,8 @@ class DualIntrusiveAllocator final {
|
||||
size_t ListSize() const { return ListCurrentOffset; }
|
||||
size_t ListBackingSize() const { return MemorySize; }
|
||||
|
||||
uintptr_t const DataBegin() const { return Data; }
|
||||
uintptr_t const ListBegin() const { return List; }
|
||||
uintptr_t DataBegin() const { return Data; }
|
||||
uintptr_t ListBegin() const { return List; }
|
||||
|
||||
void Reset() { DataCurrentOffset = 0; ListCurrentOffset = 0; }
|
||||
|
||||
@@ -159,7 +159,7 @@ public:
|
||||
stream.write((char*)GetListData(), ListSize);
|
||||
}
|
||||
|
||||
size_t GetInlineSize() {
|
||||
size_t GetInlineSize() const {
|
||||
static_assert(sizeof(*this) == 40);
|
||||
return sizeof(*this) + DataSize + ListSize;
|
||||
}
|
||||
@@ -317,11 +317,11 @@ public:
|
||||
return iterator(reinterpret_cast<uintptr_t>(GetListData()), reinterpret_cast<uintptr_t>(GetData()), Wrapped);
|
||||
}
|
||||
|
||||
uintptr_t const GetData() const {
|
||||
uintptr_t GetData() const {
|
||||
return reinterpret_cast<uintptr_t>(IRDataInternal ? IRDataInternal : InlineData);
|
||||
}
|
||||
|
||||
uintptr_t const GetListData() const {
|
||||
uintptr_t GetListData() const {
|
||||
return reinterpret_cast<uintptr_t>(ListDataInternal ? ListDataInternal : &InlineData[DataSize]);
|
||||
}
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ union PhysicalRegister {
|
||||
return PhysicalRegister(InvalidClass, InvalidReg);
|
||||
}
|
||||
|
||||
bool IsInvalid() {
|
||||
bool IsInvalid() const {
|
||||
return *this == Invalid();
|
||||
}
|
||||
};
|
||||
|
||||
+8
-5
@@ -1,5 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
|
||||
@@ -11,11 +13,12 @@ namespace FEXCore::Allocator {
|
||||
using REALLOC_Hook = void*(*)(void*, size_t);
|
||||
using FREE_Hook = void(*)(void*);
|
||||
|
||||
__attribute__((visibility("default"))) extern MMAP_Hook mmap;
|
||||
__attribute__((visibility("default"))) extern MUNMAP_Hook munmap;
|
||||
__attribute__((visibility("default"))) extern MALLOC_Hook malloc;
|
||||
__attribute__((visibility("default"))) extern REALLOC_Hook realloc;
|
||||
__attribute__((visibility("default"))) extern FREE_Hook free;
|
||||
FEX_DEFAULT_VISIBILITY extern MMAP_Hook mmap;
|
||||
FEX_DEFAULT_VISIBILITY extern MUNMAP_Hook munmap;
|
||||
FEX_DEFAULT_VISIBILITY extern MALLOC_Hook malloc;
|
||||
FEX_DEFAULT_VISIBILITY extern REALLOC_Hook realloc;
|
||||
FEX_DEFAULT_VISIBILITY extern FREE_Hook free;
|
||||
|
||||
void SetupHooks();
|
||||
void ClearHooks();
|
||||
}
|
||||
@@ -0,0 +1,81 @@
|
||||
#pragma once
|
||||
|
||||
// Header for various utilities that operate on bits and bytes.
|
||||
|
||||
#include <bit>
|
||||
#include <climits>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <type_traits>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
// Determines the number of bits inside of a given type.
|
||||
template <typename T>
|
||||
[[nodiscard]] constexpr size_t BitSize() noexcept {
|
||||
return sizeof(T) * CHAR_BIT;
|
||||
}
|
||||
|
||||
// Swaps the bytes of a 16-bit unsigned value.
|
||||
[[nodiscard]] inline uint16_t BSwap16(uint16_t value) noexcept {
|
||||
#ifdef __GNUC__
|
||||
return __builtin_bswap16(value);
|
||||
#else
|
||||
return (value >> 8) | (value << 8);
|
||||
#endif
|
||||
}
|
||||
|
||||
// Swaps the bytes of a 32-bit unsigned value.
|
||||
[[nodiscard]] inline uint32_t BSwap32(uint32_t value) noexcept {
|
||||
#ifdef __GNUC__
|
||||
return __builtin_bswap32(value);
|
||||
#else
|
||||
return ((value & 0xFF000000U) >> 24) | ((value & 0x00FF0000U) >> 8) |
|
||||
((value & 0x0000FF00U) << 8) | ((value & 0x000000FFU) << 24);
|
||||
#endif
|
||||
}
|
||||
|
||||
// Swaps the bytes of a 64-bit unsigned value.
|
||||
[[nodiscard]] inline uint64_t BSwap64(uint64_t value) noexcept {
|
||||
#ifdef __GNUC__
|
||||
return __builtin_bswap64(value);
|
||||
#else
|
||||
return ((value & 0xFF00000000000000ULL) >> 56) | ((value & 0x00FF000000000000ULL) >> 40) |
|
||||
((value & 0x0000FF0000000000ULL) >> 24) | ((value & 0x000000FF00000000ULL) >> 8) |
|
||||
((value & 0x00000000FF000000ULL) << 8) | ((value & 0x0000000000FF0000ULL) << 24) |
|
||||
((value & 0x000000000000FF00ULL) << 40) | ((value & 0x00000000000000FFULL) << 56);
|
||||
#endif
|
||||
}
|
||||
|
||||
// Finds the first least-significant set bit within a given value.
|
||||
// Note that all returned indices are 1-based, not 0-based.
|
||||
template <typename T>
|
||||
[[nodiscard]] constexpr int FindFirstSetBit(T value) noexcept {
|
||||
static_assert(std::is_unsigned_v<T>, "Type must be unsigned.");
|
||||
|
||||
if (value == 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
const int trailing_zeroes = std::countr_zero(value);
|
||||
return trailing_zeroes + 1;
|
||||
}
|
||||
|
||||
// Stand-in for std::bit_cast until libc++ implements it.
|
||||
template <typename To, typename From>
|
||||
[[nodiscard]] inline To BitCast(const From& source) noexcept
|
||||
{
|
||||
static_assert(sizeof(From) == sizeof(To),
|
||||
"BitCast source and destination types must be equal in size.");
|
||||
static_assert(std::is_trivially_copyable_v<From>,
|
||||
"BitCast source type must be trivially copyable.");
|
||||
static_assert(std::is_trivially_copyable_v<To>,
|
||||
"BitCast destination type must be trivially copyable.");
|
||||
|
||||
std::aligned_storage_t<sizeof(To), alignof(To)> storage;
|
||||
std::memcpy(&storage, &source, sizeof(storage));
|
||||
return reinterpret_cast<To&>(storage);
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -0,0 +1,29 @@
|
||||
#pragma once
|
||||
|
||||
// Contains general abstractions related to compilers used to build FEX.
|
||||
|
||||
// Specifies the minimum alignment for a variable or structure field, measured in bytes.
|
||||
#define FEX_ALIGNED(alignment) __attribute__((aligned(alignment)))
|
||||
|
||||
// Allows annotating declarations with extra information.
|
||||
#define FEX_ANNOTATE(annotation_str) __attribute__((annotate(annotation_str)))
|
||||
|
||||
// Makes the attributed entity have the default DSO visibility level.
|
||||
// Compiler options can affect the visibility of symbols. This attribute
|
||||
// overrides said changes. This gives entities external linkage.
|
||||
#define FEX_DEFAULT_VISIBILITY __attribute__((visibility("default")))
|
||||
|
||||
// Indicates that the specified function doesn't need a function prologue/epilogue.
|
||||
// emitted for it by the compiler.
|
||||
#define FEX_NAKED __attribute__((naked))
|
||||
|
||||
// Specifies that a structure member or structure itself should have the smallest possible alignment.
|
||||
#define FEX_PACKED __attribute__((packed))
|
||||
|
||||
// Causes execution to exit abnormally.
|
||||
#define FEX_TRAP_EXECUTION __builtin_trap()
|
||||
|
||||
// Dictates to the compiler that the path this is on should not be reachable
|
||||
// from normal execution control flow. If normal execution does reach this,
|
||||
// then program behavior is undefined.
|
||||
#define FEX_UNREACHABLE __builtin_unreachable()
|
||||
+105
-8
@@ -1,7 +1,12 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <functional>
|
||||
#include <cstdarg>
|
||||
#include <sstream>
|
||||
#include <stdarg.h>
|
||||
|
||||
#include <fmt/format.h>
|
||||
|
||||
namespace LogMan {
|
||||
enum DebugLevels {
|
||||
@@ -18,8 +23,8 @@ constexpr DebugLevels MSG_LEVEL = INFO;
|
||||
|
||||
namespace Throw {
|
||||
using ThrowHandler = void(*)(char const *Message);
|
||||
__attribute__((visibility("default"))) void InstallHandler(ThrowHandler Handler);
|
||||
__attribute__((visibility("default"))) void UnInstallHandlers();
|
||||
FEX_DEFAULT_VISIBILITY void InstallHandler(ThrowHandler Handler);
|
||||
FEX_DEFAULT_VISIBILITY void UnInstallHandlers();
|
||||
|
||||
[[noreturn]] void M(const char *fmt, va_list args);
|
||||
|
||||
@@ -38,14 +43,32 @@ static inline void A(bool, const char*, ...) {}
|
||||
#define LOGMAN_THROW_A(pred, ...) do {} while (0)
|
||||
#endif
|
||||
|
||||
// Fmt interface
|
||||
|
||||
[[noreturn]] void MFmt(const char *fmt, const fmt::format_args& args);
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
template <typename... Args>
|
||||
static inline void AFmt(bool Value, const char *fmt, const Args&... args) {
|
||||
if (MSG_LEVEL < ASSERT || Value) {
|
||||
return;
|
||||
}
|
||||
MFmt(fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
#define LOGMAN_THROW_A_FMT(pred, ...) do { LogMan::Throw::AFmt(pred, __VA_ARGS__); } while (0)
|
||||
#else
|
||||
static inline void AFmt(bool, const char*, ...) {}
|
||||
#define LOGMAN_THROW_A_FMT(pred, ...) do {} while (0)
|
||||
#endif
|
||||
|
||||
} // namespace Throw
|
||||
|
||||
namespace Msg {
|
||||
using MsgHandler = void(*)(DebugLevels Level, char const *Message);
|
||||
__attribute__((visibility("default"))) void InstallHandler(MsgHandler Handler);
|
||||
__attribute__((visibility("default"))) void UnInstallHandlers();
|
||||
FEX_DEFAULT_VISIBILITY void InstallHandler(MsgHandler Handler);
|
||||
FEX_DEFAULT_VISIBILITY void UnInstallHandlers();
|
||||
|
||||
__attribute__((visibility("default"))) void M(DebugLevels Level, const char *fmt, va_list args);
|
||||
FEX_DEFAULT_VISIBILITY void M(DebugLevels Level, const char *fmt, va_list args);
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
static inline void A(const char *fmt, ...) {
|
||||
@@ -55,7 +78,7 @@ static inline void A(const char *fmt, ...) {
|
||||
M(ASSERT, fmt, args);
|
||||
va_end(args);
|
||||
}
|
||||
__builtin_trap();
|
||||
FEX_TRAP_EXECUTION;
|
||||
}
|
||||
#define LOGMAN_MSG_A(...) do { LogMan::Msg::A(__VA_ARGS__); } while (0)
|
||||
#else
|
||||
@@ -114,7 +137,81 @@ static inline void ERR(const char *fmt, ...) {
|
||||
#define ERROR_AND_DIE(...) \
|
||||
do { \
|
||||
LogMan::Msg::E(__VA_ARGS__); \
|
||||
__builtin_trap(); \
|
||||
FEX_TRAP_EXECUTION; \
|
||||
} while(0)
|
||||
|
||||
// Fmt-capable interface.
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void MFmtImpl(DebugLevels level, const char* fmt, const fmt::format_args& args);
|
||||
|
||||
template <typename... Args>
|
||||
static inline void MFmt(DebugLevels level, const char* fmt, const Args&... args) {
|
||||
MFmtImpl(level, fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
template <typename... Args>
|
||||
static inline void EFmt(const char* fmt, const Args&... args) {
|
||||
if (MSG_LEVEL < ERROR) {
|
||||
return;
|
||||
}
|
||||
MFmtImpl(ERROR, fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
template <typename... Args>
|
||||
static inline void DFmt(const char* fmt, const Args&... args) {
|
||||
if (MSG_LEVEL < DEBUG) {
|
||||
return;
|
||||
}
|
||||
MFmtImpl(DEBUG, fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
template <typename... Args>
|
||||
static inline void IFmt(const char* fmt, const Args&... args) {
|
||||
if (MSG_LEVEL < INFO) {
|
||||
return;
|
||||
}
|
||||
MFmtImpl(INFO, fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
template <typename... Args>
|
||||
static inline void OutFmt(const char* fmt, const Args&... args) {
|
||||
MFmtImpl(STDOUT, fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
template <typename... Args>
|
||||
static inline void ErrFmt(const char* fmt, const Args&... args) {
|
||||
MFmtImpl(STDERR, fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
template <typename... Args>
|
||||
static inline void AFmt(const char *fmt, const Args&... args) {
|
||||
if (MSG_LEVEL < ASSERT) {
|
||||
return;
|
||||
}
|
||||
MFmtImpl(ASSERT, fmt, fmt::make_format_args(args...));
|
||||
FEX_TRAP_EXECUTION;
|
||||
}
|
||||
#define LOGMAN_MSG_A_FMT(...) do { LogMan::Msg::AFmt(__VA_ARGS__); } while (0)
|
||||
#else
|
||||
template <typename... Args>
|
||||
static inline void AFmt(const char*, const Args&...) {}
|
||||
#define LOGMAN_MSG_A_FMT(...) do {} while(0)
|
||||
#endif
|
||||
|
||||
#define WARN_ONCE_FMT(...) \
|
||||
do { \
|
||||
static bool Warned{}; \
|
||||
if (!Warned) { \
|
||||
LogMan::Msg::DFmt(__VA_ARGS__); \
|
||||
Warned = true; \
|
||||
} \
|
||||
} while (0);
|
||||
|
||||
#define ERROR_AND_DIE_FMT(...) \
|
||||
do { \
|
||||
LogMan::Msg::EFmt(__VA_ARGS__); \
|
||||
FEX_TRAP_EXECUTION; \
|
||||
} while(0)
|
||||
|
||||
} // namespace Msg
|
||||
|
||||
@@ -7,8 +7,11 @@ namespace FEXCore::Threads {
|
||||
|
||||
class Thread;
|
||||
using CreateThreadFunc = std::function<std::unique_ptr<Thread>(ThreadFunc Func, void* Arg)>;
|
||||
using CleanupAfterForkFunc = std::function<void()>;
|
||||
|
||||
struct Pointers {
|
||||
CreateThreadFunc CreateThread;
|
||||
CleanupAfterForkFunc CleanupAfterFork;
|
||||
};
|
||||
|
||||
// API
|
||||
@@ -19,10 +22,19 @@ namespace FEXCore::Threads {
|
||||
virtual bool join(void **ret) = 0;
|
||||
virtual bool detach() = 0;
|
||||
virtual bool IsSelf() = 0;
|
||||
|
||||
/**
|
||||
* @name Calls provided API functions
|
||||
* @{ */
|
||||
|
||||
static std::unique_ptr<Thread> Create(
|
||||
ThreadFunc Func,
|
||||
void* Arg);
|
||||
|
||||
static void CleanupAfterFork();
|
||||
/** @} */
|
||||
|
||||
// Set API functions
|
||||
static void SetInternalPointers(Pointers const &_Ptrs);
|
||||
};
|
||||
}
|
||||
+1
Submodule External/drm-headers added at 2b23749e35.
+1
Submodule External/fmt added at 7bdf0628b1.
Vendored
+1
-1
Submodule External/imgui updated: 8b412457dc...df6054ff26.
@@ -17,7 +17,6 @@ See the [Source Outline](docs/SourceOutline.md) for more information.
|
||||
* cmake (version 3.14 minimum)
|
||||
* ninja-build
|
||||
* clang (version 10 minimum for C++20)
|
||||
* libnuma-dev
|
||||
* libglfw3-dev (For GUI)
|
||||
* libsdl2-dev (For GUI)
|
||||
* libepoxy-dev (For GUI)
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
add_subdirectory(Common/)
|
||||
add_subdirectory(CommonCore/)
|
||||
add_subdirectory(Linux/)
|
||||
add_subdirectory(Tests/)
|
||||
add_subdirectory(Tools/)
|
||||
|
||||
@@ -1,8 +1,10 @@
|
||||
set(NAME Common)
|
||||
set(SRCS
|
||||
ArgumentLoader.cpp
|
||||
EnvironmentLoader.cpp
|
||||
Config.cpp
|
||||
EnvironmentLoader.cpp
|
||||
FileFormatCheck.cpp
|
||||
RootFSSetup.cpp
|
||||
StringUtil.cpp)
|
||||
|
||||
add_library(${NAME} STATIC ${SRCS})
|
||||
|
||||
Loaded 100 of 248 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user