mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 12:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d3399a261b | ||
|
|
d2437e6a21 | ||
|
|
95dd6ceba8 | ||
|
|
1a0d135201 | ||
|
|
f453e1523e | ||
|
|
622b0bfbc9 | ||
|
|
ad52514b97 | ||
|
|
02a218c6e3 | ||
|
|
2d617ad173 | ||
|
|
0d06e3e47d | ||
|
|
78aee4d96e | ||
|
|
ba04da87e5 | ||
|
|
2e6b08cbcb | ||
|
|
472a373861 | ||
|
|
a451420911 | ||
|
|
2e84f21c18 | ||
|
|
fb7167c2d2 | ||
|
|
d884eb9287 | ||
|
|
8b9b1a90e4 | ||
|
|
e2d4010b59 | ||
|
|
babde31bf0 | ||
|
|
c282239077 | ||
|
|
8d28a441ab | ||
|
|
5821054d91 | ||
|
|
4626145374 | ||
|
|
a786d3621d | ||
|
|
1393dc2a5b | ||
|
|
c4604465ba | ||
|
|
cffae9cb0f | ||
|
|
cf24d3c33f | ||
|
|
672e885e40 | ||
|
|
7d05610da7 | ||
|
|
a843ecf4c8 | ||
|
|
f4ff1b0688 | ||
|
|
2b4cec8385 | ||
|
|
8ab4ab29f8 | ||
|
|
cc0509c0f3 | ||
|
|
ebfa65fedc | ||
|
|
a34ae24b3f | ||
|
|
58ea76eb24 | ||
|
|
e9a17b19c5 | ||
|
|
ce8d111453 | ||
|
|
47fd73f6cf | ||
|
|
76f3391ebc | ||
|
|
be6ff52709 | ||
|
|
e99e252188 | ||
|
|
98b980f7e3 | ||
|
|
f2f90eeb82 | ||
|
|
f267fd2250 | ||
|
|
500ad34769 | ||
|
|
1700d54012 | ||
|
|
70d8a10484 | ||
|
|
9e94784e26 | ||
|
|
4060f4018e | ||
|
|
739ac0f18f | ||
|
|
98d62a7eb1 | ||
|
|
aba7a3a830 | ||
|
|
9027d1eee7 | ||
|
|
4e5da4946d | ||
|
|
a70e3e42b2 | ||
|
|
09f476924f | ||
|
|
230e3245fd | ||
|
|
8de876daf2 | ||
|
|
53b1d155cc | ||
|
|
b0eb63ab9a | ||
|
|
2e3242682d | ||
|
|
ad4d4c9e67 | ||
|
|
3250d4e405 | ||
|
|
196a0531e0 | ||
|
|
e61cb5b2c3 | ||
|
|
f9b53c6b51 | ||
|
|
58e949e148 | ||
|
|
dad47b7bda | ||
|
|
e519bf5978 | ||
|
|
fc50e52157 | ||
|
|
7669df0e16 | ||
|
|
4d56fec5f1 | ||
|
|
8181552b16 | ||
|
|
c6c147daf6 | ||
|
|
975069825e | ||
|
|
5133f480d1 | ||
|
|
ce4b252e5c | ||
|
|
031d56de35 | ||
|
|
3cdaf6736b | ||
|
|
b5e696b3cb | ||
|
|
43aef377d7 | ||
|
|
add0e7a8db | ||
|
|
52e541d453 | ||
|
|
a031a49546 | ||
|
|
4d821b8dd8 | ||
|
|
f277025c9a | ||
|
|
ba28e6f82e | ||
|
|
3a89df9bed | ||
|
|
f6a0866fbb | ||
|
|
756fa2ecc5 | ||
|
|
cf834aa6da | ||
|
|
d2324f4a93 | ||
|
|
6226c7f4f3 | ||
|
|
991ecd558e | ||
|
|
a4fa3a460e | ||
|
|
77ba708933 | ||
|
|
662d50a966 | ||
|
|
5472d1cc04 | ||
|
|
d1d41f5645 | ||
|
|
94fd100fc7 | ||
|
|
b9ff36b5d9 | ||
|
|
cd5a809ec9 | ||
|
|
045a8efbeb | ||
|
|
54a1f7d833 | ||
|
|
1b496cda8f | ||
|
|
a5b24bfe4c | ||
|
|
46676ca376 | ||
|
|
7d939a3b3d | ||
|
|
a515061465 | ||
|
|
1c24d63f73 | ||
|
|
7e10dba5e2 | ||
|
|
e2d73014f1 | ||
|
|
53aa30596e | ||
|
|
122ae5b710 | ||
|
|
45c27b2965 | ||
|
|
832b247fc1 | ||
|
|
0e8b53d566 | ||
|
|
d03d69273b | ||
|
|
efa05ba19d | ||
|
|
5da205d91a | ||
|
|
41923bac99 | ||
|
|
c6148f6bf1 | ||
|
|
77aaa9af4d | ||
|
|
00cf8d530c | ||
|
|
98aa58e9f5 | ||
|
|
6911917819 | ||
|
|
a8255aa475 | ||
|
|
48e7aae38f | ||
|
|
7069643ae6 | ||
|
|
34272fc134 | ||
|
|
1d41002dfe | ||
|
|
563bf342d5 | ||
|
|
efd5fabb95 | ||
|
|
c1da525110 | ||
|
|
5ce6c88a88 | ||
|
|
64cce7c6fa | ||
|
|
4544e5b51f | ||
|
|
eb3e314946 | ||
|
|
8b65c3de10 | ||
|
|
c2beb27a9d | ||
|
|
05fdec9e72 | ||
|
|
e8e3c95349 | ||
|
|
a87fa3f246 | ||
|
|
b31ad523f5 | ||
|
|
34bce540ff | ||
|
|
8ea38e1d80 | ||
|
|
ce591a9541 | ||
|
|
a48c65cd65 | ||
|
|
c283f80f48 | ||
|
|
d6bf276b5a | ||
|
|
96a51650b1 | ||
|
|
f35a9c74a2 | ||
|
|
e2457943f5 | ||
|
|
a05644172a | ||
|
|
cc168ce0fb | ||
|
|
76bd22d279 | ||
|
|
18574f3cf1 | ||
|
|
665215ab47 | ||
|
|
6009f36403 | ||
|
|
2580efda0d | ||
|
|
3a310b8815 | ||
|
|
7ff96227c0 | ||
|
|
3e8d78051c | ||
|
|
bd24ebc96a | ||
|
|
6d3745b8f1 | ||
|
|
d0f0b975be | ||
|
|
ff2e6ed59f | ||
|
|
99b2018d0e | ||
|
|
f0d9c8c10a | ||
|
|
dce1b24c00 | ||
|
|
b47e981932 | ||
|
|
dc44eb4caf | ||
|
|
dfda6733f0 | ||
|
|
21c6986dc7 | ||
|
|
635720fe12 | ||
|
|
8c751d7423 | ||
|
|
702ecf7637 | ||
|
|
0595f1e044 | ||
|
|
cebb032bd3 | ||
|
|
8e32763ada | ||
|
|
7532337231 | ||
|
|
4a66d4570e | ||
|
|
6f5e99d47d | ||
|
|
9ee9f5bddd | ||
|
|
ddb9f6d3ad | ||
|
|
d29139d88a | ||
|
|
317575ba99 | ||
|
|
d4f2638a2e | ||
|
|
b67d9be227 | ||
|
|
d52add8fad | ||
|
|
aa9159d25c | ||
|
|
94c777259e | ||
|
|
c9f8fa5662 | ||
|
|
64ee6b119e | ||
|
|
d2ec9a8936 | ||
|
|
2a927453f7 | ||
|
|
c19d489c9a | ||
|
|
6012eb051b | ||
|
|
3974746473 | ||
|
|
e1bcdcf387 | ||
|
|
fd5fbddae9 | ||
|
|
8ff72beddb | ||
|
|
cba5f7877b | ||
|
|
9d7e9fd9fc | ||
|
|
082a0baff3 | ||
|
|
3a4914315b | ||
|
|
448b5a338a | ||
|
|
9b68617fa8 | ||
|
|
4c9890d7f8 | ||
|
|
b2db04f5d7 | ||
|
|
be8ff9ccb9 | ||
|
|
9c531d97b0 | ||
|
|
055d8d75a2 | ||
|
|
9fcf79ce0e | ||
|
|
6edf4619d4 | ||
|
|
8f769ce5a3 | ||
|
|
96ac71750a | ||
|
|
d0852cf1bb | ||
|
|
f5fea8af96 | ||
|
|
d52a1da501 | ||
|
|
abdcaa7c86 | ||
|
|
ad122cf463 | ||
|
|
b58a57d225 | ||
|
|
28d679de98 | ||
|
|
d1dd055e6a | ||
|
|
3045578da4 | ||
|
|
9566dda73e | ||
|
|
a0ced2b685 | ||
|
|
df232f567b | ||
|
|
2a6d6a9d13 | ||
|
|
cd03932bd1 | ||
|
|
2e5fa1ef1b | ||
|
|
0c6c4cd532 | ||
|
|
25f8a87429 | ||
|
|
edf1a7970d | ||
|
|
fac9972bad | ||
|
|
9ecb960f3a | ||
|
|
3d26e23891 | ||
|
|
7bbbd95775 | ||
|
|
bb308899b9 | ||
|
|
e95c8d703c | ||
|
|
903d6a742e | ||
|
|
424218e327 | ||
|
|
17dc03d414 | ||
|
|
baf699c6e1 | ||
|
|
1431af1ff5 | ||
|
|
775a41b903 | ||
|
|
2da1e90dd5 | ||
|
|
e614340c0c | ||
|
|
3c293b9aed | ||
|
|
283c2861c9 | ||
|
|
757dc95116 | ||
|
|
6192250b8a | ||
|
|
f489135b1d | ||
|
|
4d00a52761 | ||
|
|
6941a59223 | ||
|
|
3f232e631e | ||
|
|
6e3643c3ef | ||
|
|
d7348c8aff | ||
|
|
e7bdb8679d | ||
|
|
c28824f94d | ||
|
|
664d766b45 | ||
|
|
fce694ed92 | ||
|
|
96aafb4f07 | ||
|
|
a474f86ea8 | ||
|
|
dbaf95a8f3 | ||
|
|
e67df96ad9 | ||
|
|
56de94578d | ||
|
|
06fc2f5ef0 | ||
|
|
b3ba315cbd | ||
|
|
e5a531e683 | ||
|
|
e2de57bd04 | ||
|
|
4eebca93e3 | ||
|
|
3919ec9692 | ||
|
|
02aeb0ac1a | ||
|
|
206544ad09 | ||
|
|
3854cd2b2f | ||
|
|
b2eb8aaf66 | ||
|
|
acbd920c9a | ||
|
|
db0bdd48e5 | ||
|
|
da21ee3cda | ||
|
|
d2baef2b36 | ||
|
|
df96bc83cc | ||
|
|
ec03831a21 | ||
|
|
9ca821316a | ||
|
|
025a060337 | ||
|
|
371d6f0730 | ||
|
|
643bc10d52 | ||
|
|
8fb801069f | ||
|
|
542ed8b6ad | ||
|
|
053620f4f5 | ||
|
|
88b01a0ca9 | ||
|
|
197140498b | ||
|
|
9acd325aa4 | ||
|
|
2483329ef6 | ||
|
|
f6b58b4219 | ||
|
|
6c6d86f761 | ||
|
|
359221b379 | ||
|
|
87fe1d672e | ||
|
|
9257221b3b | ||
|
|
f9b38a1de7 | ||
|
|
67e1ac0442 | ||
|
|
c57e9e008f | ||
|
|
b34c23fe3d | ||
|
|
29f644235d | ||
|
|
01da5972fc | ||
|
|
643e964edd | ||
|
|
30e3d795da | ||
|
|
89b05a2ea4 | ||
|
|
2e009be27c | ||
|
|
32150cf7b5 | ||
|
|
bf812aae8f | ||
|
|
9a71443005 | ||
|
|
ee165249bc | ||
|
|
7c7d767195 | ||
|
|
af8cfb79e5 | ||
|
|
27c8bf3021 | ||
|
|
b0a09b31bb | ||
|
|
13ebfb1a49 | ||
|
|
f863b30951 | ||
|
|
1ce27a5e6b | ||
|
|
933d622860 | ||
|
|
5d67223236 | ||
|
|
825d2c948c | ||
|
|
29390b439a | ||
|
|
799c17eb90 | ||
|
|
5fb84866e0 | ||
|
|
4965344ef5 | ||
|
|
46ca53ad0d | ||
|
|
61ff1b3584 | ||
|
|
7c0c5de4bd | ||
|
|
8d134b8df8 | ||
|
|
a9bacc1b6b | ||
|
|
e4ff3dac86 | ||
|
|
9443b18076 | ||
|
|
bb4e81aa19 | ||
|
|
a9a9f6782a | ||
|
|
2fa6c3c918 | ||
|
|
4bd84eb523 | ||
|
|
e2073dcd30 | ||
|
|
fd72669c7e | ||
|
|
81c144697b | ||
|
|
aecf180dfe | ||
|
|
10fa4a4f20 | ||
|
|
534732564b | ||
|
|
6a314bc9cd | ||
|
|
1d4356b97e | ||
|
|
a11566012d | ||
|
|
8d929027c8 | ||
|
|
1d1ed012d8 | ||
|
|
b092b7a937 | ||
|
|
d133fa6dc1 | ||
|
|
afa7de969e | ||
|
|
41b6a89ffd | ||
|
|
9aa82ec5bf | ||
|
|
b17a2e9f96 | ||
|
|
af4e9ceeed | ||
|
|
9c62c41f5f | ||
|
|
184c9d21bb | ||
|
|
9744d8de99 | ||
|
|
f3e6ecb2c3 |
No files matched your search
@@ -13,7 +13,6 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build_plus_test:
|
||||
@@ -64,7 +63,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -20,7 +20,6 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
glibc_fault_test:
|
||||
@@ -71,7 +70,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_FEX_LINUX_TESTS=True -DENABLE_GLIBC_ALLOCATOR_HOOK_FAULT=True -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -13,7 +13,6 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
hostrunner_tests:
|
||||
@@ -64,7 +63,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -13,7 +13,6 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
instcountci_tests:
|
||||
@@ -74,7 +73,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -10,7 +10,6 @@ on:
|
||||
|
||||
env:
|
||||
BUILD_TYPE: Debug
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
mingw_build:
|
||||
@@ -74,7 +73,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -13,7 +13,6 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
vixl_simulator:
|
||||
@@ -65,7 +64,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -43,6 +43,14 @@ if (NOT CONTAINS_MINGW EQUAL -1)
|
||||
set (ENABLE_JEMALLOC FALSE)
|
||||
endif()
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
message (STATUS "Clang version ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
set (CLANG_MINIMUM_VERSION 12.0)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_LESS ${CLANG_MINIMUM_VERSION})
|
||||
message (FATAL_ERROR "Clang version too old for FEX. Need at least ${CLANG_MINIMUM_VERSION} but has ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
@@ -110,6 +118,12 @@ else()
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
option(ENABLE_X86_HOST_DEBUG "Enables compiling on x86_64 host" FALSE)
|
||||
if (NOT ENABLE_X86_HOST_DEBUG)
|
||||
message(FATAL_ERROR
|
||||
" FEX-Emu doesn't support compiling for x86-64 hosts!"
|
||||
" This is /only/ a supported configuration for FEX CI and nothing else!")
|
||||
endif()
|
||||
set(_M_X86_64 1)
|
||||
add_definitions(-D_M_X86_64=1)
|
||||
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
@@ -354,6 +368,15 @@ if (BUILD_TESTS)
|
||||
include(CTest)
|
||||
enable_testing()
|
||||
message(STATUS "Unit tests are enabled")
|
||||
|
||||
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
|
||||
if (TEST_JOB_COUNT)
|
||||
message(STATUS "Running tests with ${TEST_JOB_COUNT} jobs")
|
||||
endif()
|
||||
if (CMAKE_VERSION VERSION_LESS "3.29")
|
||||
execute_process(COMMAND "nproc" OUTPUT_STRIP_TRAILING_WHITESPACE OUTPUT_VARIABLE TEST_JOB_COUNT)
|
||||
endif()
|
||||
set(TEST_JOB_FLAG "-j${TEST_JOB_COUNT}")
|
||||
endif()
|
||||
|
||||
add_subdirectory(FEXHeaderUtils/)
|
||||
|
||||
@@ -3780,8 +3780,7 @@ public:
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_AA_FMT(MemSrc.MetaType.ExtendedType.Shift == false, "Can't shift byte");
|
||||
LOGMAN_MSG_A_FMT("Nope"); // XXX: Implement
|
||||
// strb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
strb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
}
|
||||
else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strb(rt, MemSrc.rn);
|
||||
@@ -3811,8 +3810,7 @@ public:
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_AA_FMT(MemSrc.MetaType.ExtendedType.Shift == false, "Can't shift byte");
|
||||
LOGMAN_MSG_A_FMT("Nope"); // XXX: Implement
|
||||
// ldrb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
ldrb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
}
|
||||
else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrb(rt, MemSrc.rn);
|
||||
@@ -3841,9 +3839,7 @@ public:
|
||||
void strh(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_AA_FMT(MemSrc.MetaType.ExtendedType.Shift == false, "Can't shift byte");
|
||||
LOGMAN_MSG_A_FMT("Nope"); // XXX: Implement
|
||||
// strh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
strh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
}
|
||||
else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strh(rt, MemSrc.rn);
|
||||
@@ -3872,9 +3868,7 @@ public:
|
||||
void ldrh(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_AA_FMT(MemSrc.MetaType.ExtendedType.Shift == false, "Can't shift byte");
|
||||
LOGMAN_MSG_A_FMT("Nope"); // XXX: Implement
|
||||
// ldrh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
ldrh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
}
|
||||
else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrh(rt, MemSrc.rn);
|
||||
|
||||
@@ -3321,7 +3321,18 @@ public:
|
||||
|
||||
// SVE Memory - Contiguous Store with Immediate Offset
|
||||
// SVE contiguous non-temporal store (scalar plus immediate)
|
||||
// XXX:
|
||||
void stnt1b(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
|
||||
SVEContiguousNontemporalStore(0b00, zt, pg, rn, Imm);
|
||||
}
|
||||
void stnt1h(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
|
||||
SVEContiguousNontemporalStore(0b01, zt, pg, rn, Imm);
|
||||
}
|
||||
void stnt1w(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
|
||||
SVEContiguousNontemporalStore(0b10, zt, pg, rn, Imm);
|
||||
}
|
||||
void stnt1d(ZRegister zt, PRegister pg, Register rn, int32_t Imm = 0) {
|
||||
SVEContiguousNontemporalStore(0b11, zt, pg, rn, Imm);
|
||||
}
|
||||
|
||||
// SVE store multiple structures (scalar plus immediate)
|
||||
void st2b(ZRegister zt1, ZRegister zt2, PRegister pg, Register rn, int32_t Imm = 0) {
|
||||
@@ -4481,6 +4492,22 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// SVE contiguous non-temporal store (scalar plus immediate)
|
||||
void SVEContiguousNontemporalStore(uint32_t msz, ZRegister zt, PRegister pg, Register rn, int32_t imm) {
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
LOGMAN_THROW_AA_FMT(imm >= -8 && imm <= 7,
|
||||
"Invalid loadstore offset ({}). Must be between [-8, 7]", imm);
|
||||
|
||||
const auto imm4 = static_cast<uint32_t>(imm) & 0xF;
|
||||
uint32_t Instr = 0b1110'0100'0001'0000'1110'0000'0000'0000;
|
||||
Instr |= msz << 23;
|
||||
Instr |= imm4 << 16;
|
||||
Instr |= pg.Idx() << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= zt.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVEContiguousLoadImm(bool is_store, uint32_t dtype, int32_t imm, PRegister pg, Register rn, ZRegister zt) {
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
LOGMAN_THROW_AA_FMT(imm >= -8 && imm <= 7,
|
||||
|
||||
+4
-72
@@ -20,8 +20,7 @@ the coding style of LLVM. It can also be installed as a pre-commit git hook to
|
||||
check the coding style before submitting it. The canonical source of this script
|
||||
is in the LLVM source tree under llvm/utils/git.
|
||||
|
||||
For C/C++ code it uses clang-format and for Python code it uses darker (which
|
||||
in turn invokes black).
|
||||
For C/C++ code it uses clang-format.
|
||||
|
||||
You can learn more about the LLVM coding style on llvm.org:
|
||||
https://llvm.org/docs/CodingStandards.html
|
||||
@@ -31,8 +30,8 @@ directory:
|
||||
|
||||
ln -s $(pwd)/llvm/utils/git/code-format-helper.py .git/hooks/pre-commit
|
||||
|
||||
You can control the exact path to clang-format or darker with the following
|
||||
environment variables: $CLANG_FORMAT_PATH and $DARKER_FORMAT_PATH.
|
||||
You can control the exact path to clang-format with the following
|
||||
environment variable: $CLANG_FORMAT_PATH.
|
||||
"""
|
||||
|
||||
|
||||
@@ -245,74 +244,7 @@ class ClangFormatHelper(FormatHelper):
|
||||
else:
|
||||
return None
|
||||
|
||||
|
||||
class DarkerFormatHelper(FormatHelper):
|
||||
name = "darker"
|
||||
friendly_name = "Python code formatter"
|
||||
|
||||
@property
|
||||
def instructions(self) -> str:
|
||||
return " ".join(self.darker_cmd)
|
||||
|
||||
def filter_changed_files(self, changed_files: List[str]) -> List[str]:
|
||||
filtered_files = []
|
||||
for path in changed_files:
|
||||
name, ext = os.path.splitext(path)
|
||||
if ext == ".py":
|
||||
filtered_files.append(path)
|
||||
|
||||
return filtered_files
|
||||
|
||||
@property
|
||||
def darker_fmt_path(self) -> str:
|
||||
if "DARKER_FORMAT_PATH" in os.environ:
|
||||
return os.environ["DARKER_FORMAT_PATH"]
|
||||
return "darker"
|
||||
|
||||
def has_tool(self) -> bool:
|
||||
cmd = [self.darker_fmt_path, "--version"]
|
||||
proc = None
|
||||
try:
|
||||
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||
except:
|
||||
return False
|
||||
return proc.returncode == 0
|
||||
|
||||
def format_run(self, changed_files: List[str], args: FormatArgs) -> Optional[str]:
|
||||
py_files = self.filter_changed_files(changed_files)
|
||||
if not py_files:
|
||||
return None
|
||||
darker_cmd = [
|
||||
self.darker_fmt_path,
|
||||
"--check",
|
||||
"--diff",
|
||||
]
|
||||
if args.start_rev and args.end_rev:
|
||||
darker_cmd += ["-r", f"{args.start_rev}...{args.end_rev}"]
|
||||
darker_cmd += py_files
|
||||
if args.verbose:
|
||||
print(f"Running: {' '.join(darker_cmd)}")
|
||||
self.darker_cmd = darker_cmd
|
||||
proc = subprocess.run(
|
||||
darker_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE
|
||||
)
|
||||
if args.verbose:
|
||||
sys.stdout.write(proc.stderr.decode("utf-8"))
|
||||
|
||||
if proc.returncode != 0:
|
||||
# formatting needed, or the command otherwise failed
|
||||
if args.verbose:
|
||||
print(f"error: {self.name} exited with code {proc.returncode}")
|
||||
# Print the diff in the log so that it is viewable there
|
||||
print(proc.stdout.decode("utf-8"))
|
||||
return proc.stdout.decode("utf-8")
|
||||
else:
|
||||
sys.stdout.write(proc.stdout.decode("utf-8"))
|
||||
return None
|
||||
|
||||
|
||||
ALL_FORMATTERS = (DarkerFormatHelper(), ClangFormatHelper())
|
||||
|
||||
ALL_FORMATTERS = [ClangFormatHelper()]
|
||||
|
||||
def hook_main():
|
||||
# fill out args
|
||||
|
||||
Vendored
+1
-1
Submodule External/vixl updated: 7725aec177...a90f5d5020.
@@ -552,14 +552,20 @@ def print_ir_arg_printer():
|
||||
output_file.write("\t*out << \" \";\n")
|
||||
|
||||
SSAArgNum = 0
|
||||
FirstArg = True
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
|
||||
# No point printing temporaries that we can't recover
|
||||
if arg.Temporary:
|
||||
# Temporary that we can't recover
|
||||
output_file.write("\t*out << \"{}:Tmp:{}\";\n".format(arg.Type, arg.Name))
|
||||
elif arg.IsSSA:
|
||||
continue
|
||||
|
||||
if FirstArg:
|
||||
FirstArg = False
|
||||
else:
|
||||
output_file.write('\t*out << ", ";\n')
|
||||
|
||||
if arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("\tPrintArg(out, IR, Op->Header.Args[{}], RAData);\n".format(SSAArgNum))
|
||||
SSAArgNum = SSAArgNum + 1
|
||||
@@ -567,9 +573,6 @@ def print_ir_arg_printer():
|
||||
# User defined op that is stored
|
||||
output_file.write("\tPrintArg(out, IR, Op->{});\n".format(arg.Name))
|
||||
|
||||
if not LastArg:
|
||||
output_file.write("\t*out << \", \";\n")
|
||||
|
||||
output_file.write("break;\n")
|
||||
output_file.write("}\n")
|
||||
|
||||
|
||||
@@ -95,6 +95,7 @@ set (SRCS
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
Interface/Core/ObjectCache/ObjectCacheService.cpp
|
||||
Interface/Core/OpcodeDispatcher/AVX_128.cpp
|
||||
Interface/Core/OpcodeDispatcher/Crypto.cpp
|
||||
Interface/Core/OpcodeDispatcher/Flags.cpp
|
||||
Interface/Core/OpcodeDispatcher/Vector.cpp
|
||||
@@ -367,16 +368,6 @@ function(AddLibrary Name Type)
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
if (MINGW_BUILD)
|
||||
# Mingw build isn't building a linux shared library, so it can't have a SONAME.
|
||||
set_target_properties(${Name} PROPERTIES NO_SONAME ON)
|
||||
# Change the suffixes otherwise cmake continues using .a and .so
|
||||
if (${Type} STREQUAL SHARED)
|
||||
set_target_properties(${Name} PROPERTIES SUFFIX ".dll")
|
||||
elseif(${Type} STREQUAL STATIC)
|
||||
set_target_properties(${Name} PROPERTIES SUFFIX ".lib")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
endfunction()
|
||||
|
||||
@@ -50,8 +50,6 @@
|
||||
"DISABLESVE": "disablesve",
|
||||
"ENABLEAVX": "enableavx",
|
||||
"DISABLEAVX": "disableavx",
|
||||
"ENABLEAVX2": "enableavx2",
|
||||
"DISABLEAVX2": "disableavx2",
|
||||
"ENABLEAFP": "enableafp",
|
||||
"DISABLEAFP": "disableafp",
|
||||
"ENABLELRCPC": "enablelrcpc",
|
||||
@@ -86,7 +84,6 @@
|
||||
"\toff: Default CPU features queried from CPU features",
|
||||
"\t{enable,disable}sve: Will force enable or disable sve even if the host doesn't support it",
|
||||
"\t{enable,disable}avx: Will force enable or disable avx even if the host doesn't support it",
|
||||
"\t{enable,disable}avx2: Will force enable or disable avx2 even if the host doesn't support it",
|
||||
"\t{enable,disable}afp: Will force enable or disable afp even if the host doesn't support it",
|
||||
"\t{enable,disable}lrcpc: Will force enable or disable lrcpc even if the host doesn't support it",
|
||||
"\t{enable,disable}lrcpc2: Will force enable or disable lrcpc2 even if the host doesn't support it",
|
||||
|
||||
@@ -104,6 +104,9 @@ public:
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) override;
|
||||
|
||||
void ReconstructXMMRegisters(const FEXCore::Core::InternalThreadState* Thread, __uint128_t* XMM_Low, __uint128_t* YMM_High) override;
|
||||
void SetXMMRegistersFromState(FEXCore::Core::InternalThreadState* Thread, const __uint128_t* XMM_Low, const __uint128_t* YMM_High) override;
|
||||
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
|
||||
@@ -24,10 +24,6 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation on Linux in the future to gain one more register.
|
||||
|
||||
namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
@@ -628,7 +624,7 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
|
||||
@@ -713,13 +709,15 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
// It's not a concern if they get trounced by something else.
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE) {
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE128) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
@@ -783,8 +781,8 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushVectorRegisters(ARMEmitter::Register TmpReg, bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs) {
|
||||
if (SVERegs) {
|
||||
void Arm64Emitter::PushVectorRegisters(ARMEmitter::Register TmpReg, bool SVE256Regs, std::span<const ARMEmitter::VRegister> VRegs) {
|
||||
if (SVE256Regs) {
|
||||
size_t i = 0;
|
||||
|
||||
for (; i < (VRegs.size() % 4); i += 2) {
|
||||
@@ -834,8 +832,8 @@ void Arm64Emitter::PushGeneralRegisters(ARMEmitter::Register TmpReg, std::span<c
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopVectorRegisters(bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs) {
|
||||
if (SVERegs) {
|
||||
void Arm64Emitter::PopVectorRegisters(bool SVE256Regs, std::span<const ARMEmitter::VRegister> VRegs) {
|
||||
if (SVE256Regs) {
|
||||
size_t i = 0;
|
||||
for (; i < (VRegs.size() % 4); i += 2) {
|
||||
const auto Reg1 = VRegs[i];
|
||||
@@ -884,9 +882,9 @@ void Arm64Emitter::PopGeneralRegisters(std::span<const ARMEmitter::Register> Reg
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE256 ? 32 : 16;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
@@ -898,7 +896,7 @@ void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 2 == 0, "Needs to have multiple of 2 FPRs for RA");
|
||||
|
||||
// Push the vector registers
|
||||
PushVectorRegisters(TmpReg, CanUseSVE, GeneralFPRegisters);
|
||||
PushVectorRegisters(TmpReg, CanUseSVE256, GeneralFPRegisters);
|
||||
|
||||
// Push the general registers.
|
||||
PushGeneralRegisters(TmpReg, ConfiguredDynamicRegisterBase);
|
||||
@@ -909,10 +907,10 @@ void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
|
||||
// Pop vectors first
|
||||
PopVectorRegisters(CanUseSVE, GeneralFPRegisters);
|
||||
PopVectorRegisters(CanUseSVE256, GeneralFPRegisters);
|
||||
|
||||
// Pop GPRs second
|
||||
PopGeneralRegisters(ConfiguredDynamicRegisterBase);
|
||||
@@ -923,8 +921,8 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
const auto FPRRegSize = CanUseSVE256 ? 32 : 16;
|
||||
|
||||
std::span<const ARMEmitter::Register> DynamicGPRs {};
|
||||
std::span<const ARMEmitter::VRegister> DynamicFPRs {};
|
||||
@@ -936,7 +934,7 @@ void Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool
|
||||
PreserveSRAMask = x64::PreserveAll_SRAMask;
|
||||
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRMask;
|
||||
|
||||
if (CanUseSVE) {
|
||||
if (CanUseSVE256) {
|
||||
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
|
||||
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
|
||||
}
|
||||
@@ -946,7 +944,7 @@ void Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool
|
||||
PreserveSRAMask = x32::PreserveAll_SRAMask;
|
||||
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRMask;
|
||||
|
||||
if (CanUseSVE) {
|
||||
if (CanUseSVE256) {
|
||||
DynamicFPRs = x32::PreserveAll_DynamicFPRSVE;
|
||||
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRSVEMask;
|
||||
}
|
||||
@@ -965,14 +963,14 @@ void Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
// Push the vector registers.
|
||||
PushVectorRegisters(TmpReg, CanUseSVE, DynamicFPRs);
|
||||
PushVectorRegisters(TmpReg, CanUseSVE256, DynamicFPRs);
|
||||
|
||||
// Push the general registers.
|
||||
PushGeneralRegisters(TmpReg, DynamicGPRs);
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
|
||||
std::span<const ARMEmitter::Register> DynamicGPRs {};
|
||||
std::span<const ARMEmitter::VRegister> DynamicFPRs {};
|
||||
@@ -985,7 +983,7 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
PreserveSRAMask = x64::PreserveAll_SRAMask;
|
||||
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRMask;
|
||||
|
||||
if (CanUseSVE) {
|
||||
if (CanUseSVE256) {
|
||||
DynamicFPRs = x64::PreserveAll_DynamicFPRSVE;
|
||||
PreserveSRAFPRMask = x64::PreserveAll_SRAFPRSVEMask;
|
||||
}
|
||||
@@ -995,7 +993,7 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
PreserveSRAMask = x32::PreserveAll_SRAMask;
|
||||
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRMask;
|
||||
|
||||
if (CanUseSVE) {
|
||||
if (CanUseSVE256) {
|
||||
DynamicFPRs = x32::PreserveAll_DynamicFPRSVE;
|
||||
PreserveSRAFPRMask = x32::PreserveAll_SRAFPRSVEMask;
|
||||
}
|
||||
@@ -1005,7 +1003,7 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
FillStaticRegs(true, PreserveSRAMask, PreserveSRAFPRMask);
|
||||
|
||||
// Pop the vector registers.
|
||||
PopVectorRegisters(CanUseSVE, DynamicFPRs);
|
||||
PopVectorRegisters(CanUseSVE256, DynamicFPRs);
|
||||
|
||||
// Pop the general registers.
|
||||
PopGeneralRegisters(DynamicGPRs);
|
||||
|
||||
@@ -19,6 +19,10 @@ namespace CPU {
|
||||
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
|
||||
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_PSUBADDPS_INVERT
|
||||
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_PSUBADDPS_INVERT_UPPER
|
||||
{0x0000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_PSUBADDPD_INVERT
|
||||
{0x0000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_PSUBADDPD_INVERT_UPPER
|
||||
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
|
||||
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
|
||||
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
|
||||
|
||||
@@ -36,7 +36,6 @@ namespace CodeSerialize {
|
||||
namespace CPU {
|
||||
struct CPUBackendFeatures {
|
||||
bool SupportsFlags = false;
|
||||
bool SupportsSaturatingRoundingShifts = false;
|
||||
bool SupportsVTBL2 = false;
|
||||
};
|
||||
|
||||
|
||||
@@ -347,8 +347,7 @@ void CPUIDEmu::SetupHostHybridFlag() {}
|
||||
|
||||
|
||||
void CPUIDEmu::SetupFeatures() {
|
||||
// TODO: Enable once AVX is supported.
|
||||
if (false && CTX->HostFeatures.SupportsAVX) {
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
|
||||
@@ -417,7 +416,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
(1 << 9) | // SSSE3
|
||||
(0 << 10) | // L1 context ID
|
||||
(0 << 11) | // Silicon debug
|
||||
(0 << 12) | // FMA3
|
||||
(SupportsAVX() << 12) | // FMA3
|
||||
(1 << 13) | // CMPXCHG16B
|
||||
(0 << 14) | // xTPR update control
|
||||
(0 << 15) | // Perfmon and debug capability
|
||||
@@ -434,7 +433,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
(SupportsAVX() << 26) | // XSAVE
|
||||
(SupportsAVX() << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(SupportsAVX() << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(Hypervisor << 31);
|
||||
|
||||
@@ -601,6 +600,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
// This is due to LRCPC performance on Cortex being abysmal.
|
||||
// Only enable EnhancedREPMOVS if SoftwareTSO isn't required OR if MemcpySetTSO is not enabled.
|
||||
const uint32_t SupportsEnhancedREPMOVS = CTX->SoftwareTSORequired() == false || MemcpySetTSOEnabled() == false;
|
||||
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
|
||||
|
||||
// Number of subfunctions
|
||||
Res.eax = 0x0;
|
||||
@@ -609,7 +609,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 2) | // SGX
|
||||
(SupportsAVX() << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(0 << 5) | // AVX2 support
|
||||
(SupportsAVX() << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(SupportsAVX() << 8) | // BMI2
|
||||
@@ -637,38 +637,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
Res.ecx = (1 << 0) | // PREFETCHWT1
|
||||
(0 << 1) | // AVX512VBMI
|
||||
(0 << 2) | // Usermode instruction prevention
|
||||
(0 << 3) | // Protection keys for user mode pages
|
||||
(0 << 4) | // OS protection keys
|
||||
(0 << 5) | // waitpkg
|
||||
(0 << 6) | // AVX512_VBMI2
|
||||
(0 << 7) | // CET shadow stack
|
||||
(0 << 8) | // GFNI
|
||||
(0 << 9) | // VAES
|
||||
(0 << 10) | // VPCLMULQDQ
|
||||
(0 << 11) | // AVX512_VNNI
|
||||
(0 << 12) | // AVX512_BITALG
|
||||
(0 << 13) | // Intel Total Memory Encryption
|
||||
(0 << 14) | // AVX512_VPOPCNTDQ
|
||||
(0 << 15) | // Reserved
|
||||
(0 << 16) | // 5 Level page tables
|
||||
(0 << 17) | // MPX MAWAU
|
||||
(0 << 18) | // MPX MAWAU
|
||||
(0 << 19) | // MPX MAWAU
|
||||
(0 << 20) | // MPX MAWAU
|
||||
(0 << 21) | // MPX MAWAU
|
||||
(1 << 22) | // RDPID Read Processor ID
|
||||
(0 << 23) | // Reserved
|
||||
(0 << 24) | // Reserved
|
||||
(0 << 25) | // CLDEMOTE
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // MOVDIRI
|
||||
(0 << 28) | // MOVDIR64B
|
||||
(0 << 29) | // Reserved
|
||||
(0 << 30) | // SGX Launch configuration
|
||||
(0 << 31); // Reserved
|
||||
Res.ecx = (1 << 0) | // PREFETCHWT1
|
||||
(0 << 1) | // AVX512VBMI
|
||||
(0 << 2) | // Usermode instruction prevention
|
||||
(0 << 3) | // Protection keys for user mode pages
|
||||
(0 << 4) | // OS protection keys
|
||||
(0 << 5) | // waitpkg
|
||||
(0 << 6) | // AVX512_VBMI2
|
||||
(0 << 7) | // CET shadow stack
|
||||
(0 << 8) | // GFNI
|
||||
(CTX->HostFeatures.SupportsAES256 << 9) | // VAES
|
||||
(SupportsVPCLMULQDQ << 10) | // VPCLMULQDQ
|
||||
(0 << 11) | // AVX512_VNNI
|
||||
(0 << 12) | // AVX512_BITALG
|
||||
(0 << 13) | // Intel Total Memory Encryption
|
||||
(0 << 14) | // AVX512_VPOPCNTDQ
|
||||
(0 << 15) | // Reserved
|
||||
(0 << 16) | // 5 Level page tables
|
||||
(0 << 17) | // MPX MAWAU
|
||||
(0 << 18) | // MPX MAWAU
|
||||
(0 << 19) | // MPX MAWAU
|
||||
(0 << 20) | // MPX MAWAU
|
||||
(0 << 21) | // MPX MAWAU
|
||||
(1 << 22) | // RDPID Read Processor ID
|
||||
(0 << 23) | // Reserved
|
||||
(0 << 24) | // Reserved
|
||||
(0 << 25) | // CLDEMOTE
|
||||
(0 << 26) | // Reserved
|
||||
(0 << 27) | // MOVDIRI
|
||||
(0 << 28) | // MOVDIR64B
|
||||
(0 << 29) | // Reserved
|
||||
(0 << 30) | // SGX Launch configuration
|
||||
(0 << 31); // Reserved
|
||||
|
||||
Res.edx = (0 << 0) | // Reserved
|
||||
(0 << 1) | // Reserved
|
||||
@@ -887,7 +887,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // Reserved
|
||||
(0 << 21) | // XOP-TBM
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
@@ -895,7 +895,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // Reserved
|
||||
(0 << 29) | // MONITORX
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
|
||||
@@ -216,6 +216,55 @@ uint32_t ContextImpl::ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadSt
|
||||
return EFLAGS;
|
||||
}
|
||||
|
||||
void ContextImpl::ReconstructXMMRegisters(const FEXCore::Core::InternalThreadState* Thread, __uint128_t* XMM_Low, __uint128_t* YMM_High) {
|
||||
const size_t MaximumRegisters = Config.Is64BitMode ? FEXCore::Core::CPUState::NUM_XMMS : 8;
|
||||
|
||||
if (YMM_High != nullptr && HostFeatures.SupportsAVX) {
|
||||
const bool SupportsConvergedRegisters = HostFeatures.SupportsSVE256;
|
||||
|
||||
if (SupportsConvergedRegisters) {
|
||||
///< Output wants to de-interleave
|
||||
for (size_t i = 0; i < MaximumRegisters; ++i) {
|
||||
memcpy(&XMM_Low[i], &Thread->CurrentFrame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
memcpy(&YMM_High[i], &Thread->CurrentFrame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
///< Matches what FEX wants with non-converged registers
|
||||
for (size_t i = 0; i < MaximumRegisters; ++i) {
|
||||
memcpy(&XMM_Low[i], &Thread->CurrentFrame->State.xmm.sse.data[i][0], sizeof(__uint128_t));
|
||||
memcpy(&YMM_High[i], &Thread->CurrentFrame->State.avx_high[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Only support SSE, no AVX here, even if requested.
|
||||
memcpy(XMM_Low, Thread->CurrentFrame->State.xmm.sse.data, MaximumRegisters * sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::SetXMMRegistersFromState(FEXCore::Core::InternalThreadState* Thread, const __uint128_t* XMM_Low, const __uint128_t* YMM_High) {
|
||||
const size_t MaximumRegisters = Config.Is64BitMode ? FEXCore::Core::CPUState::NUM_XMMS : 8;
|
||||
if (YMM_High != nullptr && HostFeatures.SupportsAVX) {
|
||||
const bool SupportsConvergedRegisters = HostFeatures.SupportsSVE256;
|
||||
|
||||
if (SupportsConvergedRegisters) {
|
||||
///< Output wants to de-interleave
|
||||
for (size_t i = 0; i < MaximumRegisters; ++i) {
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &XMM_Low[i], sizeof(__uint128_t));
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][2], &YMM_High[i], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
///< Matches what FEX wants with non-converged registers
|
||||
for (size_t i = 0; i < MaximumRegisters; ++i) {
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &XMM_Low[i], sizeof(__uint128_t));
|
||||
memcpy(&Thread->CurrentFrame->State.avx_high[i][0], &YMM_High[i], sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Only support SSE, no AVX here, even if requested.
|
||||
memcpy(Thread->CurrentFrame->State.xmm.sse.data, XMM_Low, MaximumRegisters * sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
@@ -583,7 +632,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->_ExitFunction(
|
||||
Thread->OpDispatcher->ExitFunction(
|
||||
Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
@@ -614,7 +663,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
const bool NeedsBlockEnd =
|
||||
@@ -631,7 +680,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->_ExitFunction(
|
||||
Thread->OpDispatcher->ExitFunction(
|
||||
Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -110,7 +110,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
add(ARMEmitter::Size::i64Bit, StaticRegisters[X86State::REG_RSP], ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, TMP1, 0);
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE) {
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE128) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
|
||||
|
||||
@@ -221,14 +221,16 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
{
|
||||
// If we have a VSIB byte (as opposed to SIB), then the index register is a vector.
|
||||
const bool IsIndexVector = (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
uint8_t InvalidSIBIndex = 0b100; ///< SIB Index where there is no register encoding.
|
||||
if (IsIndexVector) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_VSIB_BYTE;
|
||||
InvalidSIBIndex = ~0; ///< No Invalid SIB Index with Index Vectors.
|
||||
}
|
||||
|
||||
const uint8_t IndexREX = (DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X) != 0 ? 1 : 0;
|
||||
const uint8_t BaseREX = (DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B) != 0 ? 1 : 0;
|
||||
|
||||
Operand->Data.SIB.Index = MapModRMToReg(IndexREX, SIB.index, false, false, IsIndexVector, false, 0b100);
|
||||
Operand->Data.SIB.Index = MapModRMToReg(IndexREX, SIB.index, false, false, IsIndexVector, false, InvalidSIBIndex);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(BaseREX, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
}
|
||||
|
||||
@@ -659,6 +661,9 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
if (CTX->Config.Is64BitMode && (Byte1 & 0b00100000) == 0) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_B;
|
||||
}
|
||||
if (options.w) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_OPTION_AVX_W;
|
||||
}
|
||||
if (!(map_select >= 1 && map_select <= 3)) {
|
||||
LogMan::Msg::EFmt("We don't understand a map_select of: {}", map_select);
|
||||
return false;
|
||||
@@ -941,14 +946,12 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// auto RIPOffset = LoadSource(Op, Op->Src[0], Op->Flags);
|
||||
// auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
|
||||
// Target offset is PC + InstSize + Literal
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Literal();
|
||||
break;
|
||||
}
|
||||
case 0xE9:
|
||||
case 0xEB: // Both are unconditional JMP instructions
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Literal();
|
||||
Conditional = false;
|
||||
break;
|
||||
case 0xE8: // Call - Immediate target, We don't want to inline calls
|
||||
@@ -1000,8 +1003,7 @@ bool Decoder::BranchTargetCanContinue(bool FinalInstruction) const {
|
||||
|
||||
if (DecodeInst->OP == 0xE8) { // Call - immediate target
|
||||
const uint64_t NextRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Data.Literal.Value;
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Literal();
|
||||
|
||||
if (GPRSize == 4) {
|
||||
// If we are running a 32bit guest then wrap around addresses that go above 32bit
|
||||
|
||||
@@ -44,6 +44,13 @@ static uint32_t GetFPCR() {
|
||||
static void SetFPCR(uint64_t Value) {
|
||||
__asm("msr FPCR, %[Value]" ::[Value] "r"(Value));
|
||||
}
|
||||
|
||||
static uint32_t GetMIDR() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], MIDR_EL1" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
#else
|
||||
static uint32_t GetDCZID() {
|
||||
// Return unsupported
|
||||
@@ -51,7 +58,7 @@ static uint32_t GetDCZID() {
|
||||
}
|
||||
#endif
|
||||
|
||||
static void OverrideFeatures(HostFeatures* Features) {
|
||||
static void OverrideFeatures(HostFeatures* Features, uint64_t ForceSVEWidth) {
|
||||
// Override features if the user has specifically called for it.
|
||||
FEX_CONFIG_OPT(HostFeatures, HOSTFEATURES);
|
||||
if (!HostFeatures()) {
|
||||
@@ -75,8 +82,7 @@ static void OverrideFeatures(HostFeatures* Features) {
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
|
||||
|
||||
ENABLE_DISABLE_OPTION(SupportsAVX, AVX, AVX);
|
||||
ENABLE_DISABLE_OPTION(SupportsAVX2, AVX2, AVX2);
|
||||
ENABLE_DISABLE_OPTION(SupportsSVE, SVE, SVE);
|
||||
ENABLE_DISABLE_OPTION(SupportsSVE128, SVE, SVE);
|
||||
ENABLE_DISABLE_OPTION(SupportsAFP, AFP, AFP);
|
||||
ENABLE_DISABLE_OPTION(SupportsRCPC, LRCPC, LRCPC);
|
||||
ENABLE_DISABLE_OPTION(SupportsTSOImm9, LRCPC2, LRCPC2);
|
||||
@@ -100,12 +106,17 @@ static void OverrideFeatures(HostFeatures* Features) {
|
||||
Features->SupportsCRC = true;
|
||||
Features->SupportsSHA = true;
|
||||
Features->SupportsPMULL_128Bit = true;
|
||||
Features->SupportsAES256 = true;
|
||||
} else if (DisableCrypto) {
|
||||
Features->SupportsAES = false;
|
||||
Features->SupportsCRC = false;
|
||||
Features->SupportsSHA = false;
|
||||
Features->SupportsPMULL_128Bit = false;
|
||||
Features->SupportsAES256 = false;
|
||||
}
|
||||
|
||||
///< Only force enable SVE256 if SVE is already enabled and ForceSVEWidth is set to >= 256.
|
||||
Features->SupportsSVE256 = ForceSVEWidth && ForceSVEWidth >= 256;
|
||||
}
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
@@ -122,6 +133,9 @@ HostFeatures::HostFeatures() {
|
||||
auto Features = vixl::CPUFeatures::InferFromIDRegisters();
|
||||
#endif
|
||||
|
||||
FEX_CONFIG_OPT(ForceSVEWidth, FORCESVEWIDTH);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsSHA = Features.Has(vixl::CPUFeatures::Feature::kSHA1) && Features.Has(vixl::CPUFeatures::Feature::kSHA2);
|
||||
@@ -141,16 +155,19 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode enable SVE with 256-bit wide registers.
|
||||
SupportsSVE = true;
|
||||
SupportsAVX = true;
|
||||
SupportsSVE128 = ForceSVEWidth() ? ForceSVEWidth() >= 128 : true;
|
||||
SupportsSVE256 = ForceSVEWidth() ? ForceSVEWidth() >= 256 : true;
|
||||
#else
|
||||
SupportsSVE = Features.Has(vixl::CPUFeatures::Feature::kSVE);
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) && vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
SupportsSVE128 = Features.Has(vixl::CPUFeatures::Feature::kSVE2);
|
||||
SupportsSVE256 = Features.Has(vixl::CPUFeatures::Feature::kSVE2) && vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
#endif
|
||||
// TODO: AVX2 is currently unsupported. Disable until the remaining features are implemented.
|
||||
SupportsAVX2 = false;
|
||||
SupportsAVX = true;
|
||||
|
||||
SupportsAES256 = SupportsAVX && SupportsAES;
|
||||
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
SupportsCLWB = true;
|
||||
@@ -185,6 +202,24 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
// Set FPCR back to original just in case anything changed
|
||||
SetFPCR(OriginalFPCR);
|
||||
|
||||
if (SupportsRAND) {
|
||||
const auto MIDR = GetMIDR();
|
||||
constexpr uint32_t Implementer_QCOM = 0x51;
|
||||
constexpr uint32_t PartNum_Oryon1 = 0x001;
|
||||
const uint32_t MIDR_Implementer = (MIDR >> 24) & 0xFF;
|
||||
const uint32_t MIDR_PartNum = (MIDR >> 4) & 0xFFF;
|
||||
if (MIDR_Implementer == Implementer_QCOM && MIDR_PartNum == PartNum_Oryon1) {
|
||||
// Work around an errata in Qualcomm's Oryon.
|
||||
// While this CPU implements the RAND extension:
|
||||
// - The RNDR register works.
|
||||
// - The RNDRRS register will never read a random number. (Always return failure)
|
||||
// This is contrary to x86 RNG behaviour where it allows spurious failure with RDSEED, but guarantees eventual success.
|
||||
// This manifested itself on Linux when an x86 processor failed to guarantee forward progress and boot of services would infinite
|
||||
// loop. Just disable this extension if this CPU is detected.
|
||||
SupportsRAND = false;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -219,12 +254,12 @@ HostFeatures::HostFeatures() {
|
||||
Supports3DNow = X86Features.has(Xbyak::util::Cpu::t3DN) && X86Features.has(Xbyak::util::Cpu::tE3DN);
|
||||
SupportsSSE4A = X86Features.has(Xbyak::util::Cpu::tSSE4a);
|
||||
SupportsAVX = true;
|
||||
SupportsAVX2 = true;
|
||||
SupportsSHA = X86Features.has(Xbyak::util::Cpu::tSHA);
|
||||
SupportsBMI1 = X86Features.has(Xbyak::util::Cpu::tBMI1);
|
||||
SupportsBMI2 = X86Features.has(Xbyak::util::Cpu::tBMI2);
|
||||
SupportsCLWB = X86Features.has(Xbyak::util::Cpu::tCLWB);
|
||||
SupportsPMULL_128Bit = X86Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
|
||||
SupportsAES256 = SupportsAES && X86Features.has(Xbyak::util::Cpu::tVAES);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
@@ -242,6 +277,17 @@ HostFeatures::HostFeatures() {
|
||||
#endif
|
||||
#endif
|
||||
SupportsPreserveAllABI = FEXCORE_HAS_PRESERVE_ALL_ATTR;
|
||||
OverrideFeatures(this);
|
||||
|
||||
if (!Is64BitMode()) {
|
||||
///< Always disable AVX and AVX2 in 32-bit mode.
|
||||
// When AVX256 is enabled, signal frames start using significantly more stack space.
|
||||
// - 16bytes * 16 registers = 256 bytes for XMM registers.
|
||||
// - 32bytes * 16 registers = 512 bytes for YMM registers.
|
||||
// There are known game failures on real x86 hardware where a 32-bit game is running up against the wall on stack space on non-AVX
|
||||
// hardware and then explodes when run on AVX hardware. This is to guard against that.
|
||||
SupportsAVX = false;
|
||||
}
|
||||
|
||||
OverrideFeatures(this, ForceSVEWidth());
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -355,6 +355,39 @@ DEF_OP(Vector_FToF) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFCVTL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VFCVTL2>();
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
fcvtl2(SubEmitSize, Dst.D(), Vector.D());
|
||||
}
|
||||
|
||||
DEF_OP(VFCVTN2) {
|
||||
const auto Op = IROp->C<IR::IROp_VFCVTN2>();
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
|
||||
auto Lower = VectorLower;
|
||||
if (Dst != VectorLower) {
|
||||
mov(VTMP1.Q(), VectorLower.Q());
|
||||
Lower = VTMP1;
|
||||
}
|
||||
|
||||
fcvtn2(SubEmitSize, Lower.Q(), VectorUpper.Q());
|
||||
|
||||
if (Dst != VectorLower) {
|
||||
mov(Dst.Q(), Lower.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -521,8 +521,9 @@ void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::NodeID Node) {}
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsAVX}
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsAVX256 {ctx->HostFeatures.SupportsAVX && ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP}
|
||||
, CTX {ctx} {
|
||||
@@ -723,8 +724,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
// LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
@@ -889,7 +888,6 @@ fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl*
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsFlags = true,
|
||||
.SupportsSaturatingRoundingShifts = true,
|
||||
.SupportsVTBL2 = true,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -72,6 +72,7 @@ private:
|
||||
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
const bool HostSupportsAVX256 {};
|
||||
const bool HostSupportsRPRES {};
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
|
||||
@@ -99,7 +99,7 @@ DEF_OP(LoadRegister) {
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regSize = HostSupportsAVX256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
LOGMAN_THROW_A_FMT(OpSize == regSize, "expected sized");
|
||||
|
||||
@@ -107,7 +107,7 @@ DEF_OP(LoadRegister) {
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
if (HostSupportsSVE256) {
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, host.Z(), PRED_TMP_32B.Merging(), guest.Z());
|
||||
} else {
|
||||
mov(host.Q(), guest.Q());
|
||||
@@ -137,7 +137,7 @@ DEF_OP(StoreRegister) {
|
||||
mov(ARMEmitter::Size::i64Bit, reg, Src);
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regSize = HostSupportsAVX256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "reg out of range");
|
||||
LOGMAN_THROW_A_FMT(OpSize == regSize, "expected sized");
|
||||
|
||||
@@ -145,7 +145,7 @@ DEF_OP(StoreRegister) {
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
if (guest.Idx() != host.Idx()) {
|
||||
if (HostSupportsSVE256) {
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, guest.Z(), PRED_TMP_32B.Merging(), host.Z());
|
||||
} else {
|
||||
mov(guest.Q(), host.Q());
|
||||
@@ -752,12 +752,14 @@ DEF_OP(LoadMemTSO) {
|
||||
}
|
||||
|
||||
DEF_OP(VLoadVectorMasked) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE support in order to use VLoadVectorMasked");
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VLoadVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VLoadVectorMasked with 256-bit operation");
|
||||
}
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -766,39 +768,95 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
// Check if the sign bit is set for the given element size.
|
||||
cmplt(SubRegSize, CMPPredicate, GoverningPredicate.Zeroing(), MaskReg.Z(), 0);
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: {
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ld1h<ARMEmitter::SubRegSize::i16Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ld1w<ARMEmitter::SubRegSize::i32Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ld1d(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
// Check if the sign bit is set for the given element size.
|
||||
cmplt(SubRegSize, CMPPredicate, GoverningPredicate.Zeroing(), MaskReg.Z(), 0);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: {
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ld1h<ARMEmitter::SubRegSize::i16Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ld1w<ARMEmitter::SubRegSize::i32Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ld1d(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
} else {
|
||||
const auto PerformMove = [this](size_t ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1: umov<ARMEmitter::SubRegSize::i8Bit>(Dst, Vector, index); break;
|
||||
case 2: umov<ARMEmitter::SubRegSize::i16Bit>(Dst, Vector, index); break;
|
||||
case 4: umov<ARMEmitter::SubRegSize::i32Bit>(Dst, Vector, index); break;
|
||||
case 8: umov<ARMEmitter::SubRegSize::i64Bit>(Dst, Vector, index); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", ElementSize); break;
|
||||
}
|
||||
};
|
||||
|
||||
// Prepare yourself adventurer. For a masked load without instructions that implement it.
|
||||
LOGMAN_THROW_A_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Only supports 128-bit without SVE256");
|
||||
size_t NumElements = IROp->Size / IROp->ElementSize;
|
||||
|
||||
// Use VTMP1 as the temporary destination
|
||||
auto TempDst = VTMP1;
|
||||
auto WorkingReg = TMP1;
|
||||
auto TempMemReg = MemReg;
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, TempDst.Q(), 0);
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "Complex addressing requested and not supported!");
|
||||
|
||||
const uint64_t ElementSizeInBits = IROp->ElementSize * 8;
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
// Extract the mask element.
|
||||
PerformMove(IROp->ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::SingleUseForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: ld1<ARMEmitter::SubRegSize::i8Bit>(TempDst.Q(), i, TempMemReg); break;
|
||||
case 2: ld1<ARMEmitter::SubRegSize::i16Bit>(TempDst.Q(), i, TempMemReg); break;
|
||||
case 4: ld1<ARMEmitter::SubRegSize::i32Bit>(TempDst.Q(), i, TempMemReg); break;
|
||||
case 8: ld1<ARMEmitter::SubRegSize::i64Bit>(TempDst.Q(), i, TempMemReg); break;
|
||||
case 16: ldr(TempDst.Q(), TempMemReg, 0); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
|
||||
if ((i + 1) != NumElements) {
|
||||
// Handle register rename to save a move.
|
||||
auto WorkingReg = TempMemReg;
|
||||
TempMemReg = TMP2;
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, IROp->ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
// Move result.
|
||||
mov(Dst.Q(), TempDst.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VStoreVectorMasked) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE support in order to use VStoreVectorMasked");
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VStoreVectorMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreVectorMasked with 256-bit operation");
|
||||
}
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -807,29 +865,280 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto RegData = GetVReg(Op->Data.ID());
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemDst = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
const auto MemDst = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
// Check if the sign bit is set for the given element size.
|
||||
cmplt(SubRegSize, CMPPredicate, GoverningPredicate.Zeroing(), MaskReg.Z(), 0);
|
||||
// Check if the sign bit is set for the given element size.
|
||||
cmplt(SubRegSize, CMPPredicate, GoverningPredicate.Zeroing(), MaskReg.Z(), 0);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: {
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: {
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
st1h<ARMEmitter::SubRegSize::i16Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
st1w<ARMEmitter::SubRegSize::i32Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
st1d(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
} else {
|
||||
const auto PerformMove = [this](size_t ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1: umov<ARMEmitter::SubRegSize::i8Bit>(Dst, Vector, index); break;
|
||||
case 2: umov<ARMEmitter::SubRegSize::i16Bit>(Dst, Vector, index); break;
|
||||
case 4: umov<ARMEmitter::SubRegSize::i32Bit>(Dst, Vector, index); break;
|
||||
case 8: umov<ARMEmitter::SubRegSize::i64Bit>(Dst, Vector, index); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", ElementSize); break;
|
||||
}
|
||||
};
|
||||
|
||||
// Prepare yourself adventurer. For a masked store without instructions that implement it.
|
||||
LOGMAN_THROW_A_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Only supports 128-bit without SVE256");
|
||||
size_t NumElements = IROp->Size / IROp->ElementSize;
|
||||
|
||||
// Use VTMP1 as the temporary destination
|
||||
auto WorkingReg = TMP1;
|
||||
auto TempMemReg = MemReg;
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "Complex addressing requested and not supported!");
|
||||
|
||||
const uint64_t ElementSizeInBits = IROp->ElementSize * 8;
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
// Extract the mask element.
|
||||
PerformMove(IROp->ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::SingleUseForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: st1<ARMEmitter::SubRegSize::i8Bit>(RegData.Q(), i, TempMemReg); break;
|
||||
case 2: st1<ARMEmitter::SubRegSize::i16Bit>(RegData.Q(), i, TempMemReg); break;
|
||||
case 4: st1<ARMEmitter::SubRegSize::i32Bit>(RegData.Q(), i, TempMemReg); break;
|
||||
case 8: st1<ARMEmitter::SubRegSize::i64Bit>(RegData.Q(), i, TempMemReg); break;
|
||||
case 16: str(RegData.Q(), TempMemReg, 0); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
|
||||
if ((i + 1) != NumElements) {
|
||||
// Handle register rename to save a move.
|
||||
auto WorkingReg = TempMemReg;
|
||||
TempMemReg = TMP2;
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, IROp->ElementSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
case 2: {
|
||||
st1h<ARMEmitter::SubRegSize::i16Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
|
||||
DEF_OP(VLoadVectorGatherMasked) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadVectorGatherMasked>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto VectorIndexSize = Op->VectorIndexElementSize;
|
||||
const auto OffsetScale = Op->OffsetScale;
|
||||
const auto DataElementOffsetStart = Op->DataElementOffsetStart;
|
||||
const auto IndexElementOffsetStart = Op->IndexElementOffsetStart;
|
||||
|
||||
///< This IR operation handles discontiguous masked gather loadstore instructions. Some things to note about its behaviour.
|
||||
/// - VSIB behaviour is mostly entirely exposed in the IR operation directly.
|
||||
/// - Displacement is the only value missing as that can be added directly to AddrBase.
|
||||
/// - VectorIndex{Low,High} contains the index offsets for each element getting loaded.
|
||||
/// - These element sizes are decoupled from the resulting element size. These can be 32-bit or 64-bit.
|
||||
/// - When the element size is 32-bit then the value is zero-extended to the full 64-bit address calculation
|
||||
/// - When loading a 128-bit result with 64-bit VectorIndex Elements, this requires the use of both VectorIndexLow and VectorIndexHigh
|
||||
/// to get enough pointers.
|
||||
/// - When VectorIndexElementSize and OffsetScale matches Arm64 SVE behaviour then the operation becomes more optimal
|
||||
/// - When the behaviour doesn't match then it gets decomposed to ASIMD style masked load.
|
||||
/// - AddrBase also doesn't need to exist
|
||||
/// - If the instruction is using 64-bit vector indexing or 32-bit addresses where the top-bit isn't set then this is valid!
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreVectorMasked with 256-bit operation");
|
||||
}
|
||||
case 4: {
|
||||
st1w<ARMEmitter::SubRegSize::i32Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
st1d(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
std::optional<ARMEmitter::Register> BaseAddr = !Op->AddrBase.IsInvalid() ? std::make_optional(GetReg(Op->AddrBase.ID())) : std::nullopt;
|
||||
const auto VectorIndexLow = GetVReg(Op->VectorIndexLow.ID());
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh =
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh.ID())) : std::nullopt;
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
const bool SupportsSVELoad = (HostSupportsSVE128 || HostSupportsSVE256) && (OffsetScale == 1 || OffsetScale == VectorIndexSize) &&
|
||||
(VectorIndexSize == IROp->ElementSize);
|
||||
|
||||
const auto PerformSMove = [this](size_t ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1: smov<ARMEmitter::SubRegSize::i8Bit>(Dst.X(), Vector, index); break;
|
||||
case 2: smov<ARMEmitter::SubRegSize::i16Bit>(Dst.X(), Vector, index); break;
|
||||
case 4: smov<ARMEmitter::SubRegSize::i32Bit>(Dst.X(), Vector, index); break;
|
||||
case 8: umov<ARMEmitter::SubRegSize::i64Bit>(Dst.X(), Vector, index); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", ElementSize); break;
|
||||
}
|
||||
};
|
||||
|
||||
const auto PerformMove = [this](size_t ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1: umov<ARMEmitter::SubRegSize::i8Bit>(Dst, Vector, index); break;
|
||||
case 2: umov<ARMEmitter::SubRegSize::i16Bit>(Dst, Vector, index); break;
|
||||
case 4: umov<ARMEmitter::SubRegSize::i32Bit>(Dst, Vector, index); break;
|
||||
case 8: umov<ARMEmitter::SubRegSize::i64Bit>(Dst, Vector, index); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", ElementSize); break;
|
||||
}
|
||||
};
|
||||
|
||||
if (SupportsSVELoad) {
|
||||
ARMEmitter::SVEModType ModType = ARMEmitter::SVEModType::MOD_NONE;
|
||||
uint8_t SVEScale = FEXCore::ilog2(OffsetScale);
|
||||
if (VectorIndexSize == 4) {
|
||||
ModType = ARMEmitter::SVEModType::MOD_SXTW;
|
||||
} else if (VectorIndexSize == 8 && OffsetScale != 1) {
|
||||
ModType = ARMEmitter::SVEModType::MOD_LSL;
|
||||
}
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
|
||||
// Check if the sign bit is set for the given element size.
|
||||
cmplt(SubRegSize, CMPPredicate, GoverningPredicate.Zeroing(), MaskReg.Z(), 0);
|
||||
auto TempDst = VTMP1;
|
||||
|
||||
// No need to load a temporary register in the case that we weren't provided a base address and there is no scaling.
|
||||
ARMEmitter::SVEMemOperand MemDst {ARMEmitter::SVEMemOperand(VectorIndexLow.Z(), 0)};
|
||||
if (BaseAddr.has_value() || OffsetScale != 1) {
|
||||
ARMEmitter::Register AddrReg = TMP1;
|
||||
if (BaseAddr.has_value()) {
|
||||
AddrReg = GetReg(Op->AddrBase.ID());
|
||||
} else {
|
||||
///< OpcodeDispatcher didn't provide a Base address while SVE requires one.
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0);
|
||||
}
|
||||
MemDst = ARMEmitter::SVEMemOperand(AddrReg.X(), VectorIndexLow.Z(), ModType, SVEScale);
|
||||
}
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: {
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(TempDst.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ld1h<ARMEmitter::SubRegSize::i16Bit>(TempDst.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ld1w<ARMEmitter::SubRegSize::i32Bit>(TempDst.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ld1d(TempDst.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
|
||||
///< Merge elements based on predicate.
|
||||
sel(SubRegSize, Dst.Z(), CMPPredicate, TempDst.Z(), IncomingDst.Z());
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit, "Can't emulate this gather load in the backend! Programming error!");
|
||||
|
||||
// FEX needs to use a temporary destination vector register in a couple of instances.
|
||||
// When Dst overlaps MaskReg, VectorIndexLow, or VectorIndexHigh
|
||||
// Due to x86 gather instruction limitations, it is highly likely that a destination temporary isn't required.
|
||||
const bool NeedsDestTmp = Dst == MaskReg || Dst == VectorIndexLow || (VectorIndexHigh.has_value() && Dst == *VectorIndexHigh);
|
||||
|
||||
// If the incoming destination isn't the destination then we need to move.
|
||||
const bool NeedsIncomingDestMove = Dst != IncomingDst || NeedsDestTmp;
|
||||
|
||||
///< Adventurers beware, emulated ASIMD style gather masked load operation.
|
||||
// Number of elements to load is calculated by the number of index elements available.
|
||||
size_t NumAddrElements = (VectorIndexHigh.has_value() ? 32 : 16) / VectorIndexSize;
|
||||
// The number of elements is clamped by the resulting register size.
|
||||
size_t NumDataElements = std::min<size_t>(IROp->Size / IROp->ElementSize, NumAddrElements);
|
||||
|
||||
size_t IndexElementsSizeBytes = NumAddrElements * VectorIndexSize;
|
||||
if (IndexElementsSizeBytes > 16) {
|
||||
// We must have a high register in this case.
|
||||
LOGMAN_THROW_A_FMT(VectorIndexHigh.has_value(), "Need High vector index register!");
|
||||
}
|
||||
|
||||
auto ResultReg = Dst;
|
||||
if (NeedsDestTmp) {
|
||||
// Use VTMP1 as the temporary destination
|
||||
ResultReg = VTMP1;
|
||||
}
|
||||
auto WorkingReg = TMP1;
|
||||
auto TempMemReg = TMP2;
|
||||
const uint64_t ElementSizeInBits = IROp->ElementSize * 8;
|
||||
|
||||
if (NeedsIncomingDestMove) {
|
||||
mov(ResultReg.Q(), IncomingDst.Q());
|
||||
}
|
||||
|
||||
for (size_t i = DataElementOffsetStart, IndexElement = IndexElementOffsetStart; i < NumDataElements; ++i, ++IndexElement) {
|
||||
ARMEmitter::SingleUseForwardLabel Skip {};
|
||||
// Extract mask element
|
||||
PerformMove(IROp->ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// Skip if the mask's sign bit isn't set
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
|
||||
// Extract Index Element
|
||||
if ((IndexElement * VectorIndexSize) >= 16) {
|
||||
// Fetch from the high index register.
|
||||
PerformSMove(VectorIndexSize, WorkingReg, *VectorIndexHigh, IndexElement - (16 / VectorIndexSize));
|
||||
} else {
|
||||
// Fetch from the low index register.
|
||||
PerformSMove(VectorIndexSize, WorkingReg, VectorIndexLow, IndexElement);
|
||||
}
|
||||
|
||||
// Calculate memory position for this gather load
|
||||
if (BaseAddr.has_value()) {
|
||||
if (VectorIndexSize == 4) {
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
} else {
|
||||
///< In this case we have no base address, All addresses come from the vector register itself
|
||||
if (VectorIndexSize == 4) {
|
||||
// Sign extend and shift in to the 64-bit register
|
||||
sbfiz(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
|
||||
} else {
|
||||
lsl(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
}
|
||||
|
||||
// Now that the address is calculated. Do the load.
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: ld1<ARMEmitter::SubRegSize::i8Bit>(ResultReg.Q(), i, TempMemReg); break;
|
||||
case 2: ld1<ARMEmitter::SubRegSize::i16Bit>(ResultReg.Q(), i, TempMemReg); break;
|
||||
case 4: ld1<ARMEmitter::SubRegSize::i32Bit>(ResultReg.Q(), i, TempMemReg); break;
|
||||
case 8: ld1<ARMEmitter::SubRegSize::i64Bit>(ResultReg.Q(), i, TempMemReg); break;
|
||||
case 16: ldr(ResultReg.Q(), TempMemReg, 0); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
}
|
||||
|
||||
if (NeedsDestTmp) {
|
||||
// Move result.
|
||||
mov(Dst.Q(), ResultReg.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1883,5 +2192,47 @@ DEF_OP(Prefetch) {
|
||||
prfm(PrefetchType[LUT], MemSrc);
|
||||
}
|
||||
|
||||
DEF_OP(VStoreNonTemporal) {
|
||||
const auto Op = IROp->C<IR::IROp_VStoreNonTemporal>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use VStoreNonTemporal with 256-bit operation");
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
const auto OffsetScaled = Offset / 32;
|
||||
stnt1b(Value.Z(), GoverningPredicate, MemReg, OffsetScaled);
|
||||
} else if (Is128Bit && HostSupportsSVE128) {
|
||||
const auto GoverningPredicate = PRED_TMP_16B.Zeroing();
|
||||
const auto OffsetScaled = Offset / 16;
|
||||
stnt1b(Value.Z(), GoverningPredicate, MemReg, OffsetScaled);
|
||||
} else {
|
||||
// Treat the non-temporal store as a regular vector store in this case for compatibility
|
||||
str(Value.Q(), MemReg, Offset);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VStoreNonTemporalPair) {
|
||||
const auto Op = IROp->C<IR::IROp_VStoreNonTemporalPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "This IR operation only operates at 128-bit wide");
|
||||
|
||||
const auto ValueLow = GetVReg(Op->ValueLow.ID());
|
||||
const auto ValueHigh = GetVReg(Op->ValueHigh.ID());
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
stnp(ValueLow.Q(), ValueHigh.Q(), MemReg, Offset);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -120,6 +120,39 @@ DEF_OP(SetRoundingMode) {
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(PushRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_PushRoundingMode>();
|
||||
auto Dest = GetReg(Node);
|
||||
|
||||
// Save the old rounding mode
|
||||
mrs(Dest, ARMEmitter::SystemRegister::FPCR);
|
||||
|
||||
// vixl simulator doesn't support anything beyond ties-to-even rounding
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
return;
|
||||
}
|
||||
|
||||
// Insert the rounding flags, reversing the mode bits as above
|
||||
if (Op->RoundMode == 3) {
|
||||
orr(ARMEmitter::Size::i64Bit, TMP1, Dest, 3 << 22);
|
||||
} else if (Op->RoundMode == 0) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(3 << 22));
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(Op->RoundMode << 22));
|
||||
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, (Op->RoundMode == 2 ? 1 : 2) << 22);
|
||||
}
|
||||
|
||||
// Now save the new FPCR
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(PopRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_PopRoundingMode>();
|
||||
msr(ARMEmitter::SystemRegister::FPCR, GetReg(Op->FPCR.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
|
||||
@@ -194,7 +194,7 @@ DEF_UNOP(VNeg, neg, false)
|
||||
DEF_UNOP(VFNeg, fneg, false)
|
||||
|
||||
DEF_BITOP(VAnd, and_)
|
||||
DEF_BITOP(VBic, bic)
|
||||
DEF_BITOP(VAndn, bic)
|
||||
DEF_BITOP(VOr, orr)
|
||||
DEF_BITOP(VXor, eor)
|
||||
|
||||
@@ -224,16 +224,18 @@ DEF_FBINOP_SCALAR_INSERT(VFSubScalarInsert, fsub)
|
||||
DEF_FBINOP_SCALAR_INSERT(VFMulScalarInsert, fmul)
|
||||
DEF_FBINOP_SCALAR_INSERT(VFDivScalarInsert, fdiv)
|
||||
|
||||
|
||||
// VFScalarOperation performs the operation described through ScalarEmit between Vector1 and Vector2,
|
||||
// storing it into Dst. This is a scalar operation, so the only lowest element of each vector is used for the operation.
|
||||
// The result is stored into the destination. The untouched bits of the destination come from Vector1, unless it's a 256 vector
|
||||
// and ZeroUpperBits is true, in which case the upper bits are zero.
|
||||
void Arm64JITCore::VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2) {
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (!Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(ZeroUpperBits == false, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(Is256Bit || !ZeroUpperBits, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
|
||||
// The upper bits of the destination comes from Vector1.
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
@@ -261,8 +263,8 @@ void Arm64JITCore::VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool Z
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
} else if (Dst != Vector2) {
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
} else if (Dst != Vector2) { // Dst different from both Vector1 and Vector2
|
||||
if (Is256Bit && !ZeroUpperBits) {
|
||||
mov(Dst.Z(), Vector1.Z());
|
||||
} else {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
@@ -279,36 +281,30 @@ void Arm64JITCore::VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool Z
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Destination intersects Vector2, can't do anything optimal in this case.
|
||||
// Do the scalar operation first and then move and insert.
|
||||
} else { // Dst same as Vector2
|
||||
|
||||
ScalarEmit(VTMP1, Vector1, Vector2);
|
||||
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
mov(Dst.Z(), Vector1.Z());
|
||||
} else {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
} else {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Similarly to VFScalarOperation it performs the operation described through ScalarEmit operating on Vector2.
|
||||
// However the result of the scalar operation is inserted into Vector1 and moved to Destination.
|
||||
// The untouched bits of the destination come from Vector1, unless it's a 256 vector
|
||||
// and ZeroUpperBits is true, in which case the upper bits are zero.
|
||||
void Arm64JITCore::VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1,
|
||||
std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2) {
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (!Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(ZeroUpperBits == false, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
|
||||
}
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
LOGMAN_THROW_A_FMT(Is256Bit || !ZeroUpperBits, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
@@ -327,7 +323,7 @@ void Arm64JITCore::VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, b
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
|
||||
if (HostSupportsAFP) {
|
||||
if (HostSupportsAFP) { // or Dst (here Dst == Vector1)
|
||||
// If the host CPU supports AFP then scalar does an insert without modifying upper bits.
|
||||
ScalarEmit(Dst, Vector2);
|
||||
} else {
|
||||
@@ -366,14 +362,10 @@ void Arm64JITCore::VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, b
|
||||
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
mov(Dst.Z(), Vector1.Z());
|
||||
} else {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
} else {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
@@ -457,12 +449,17 @@ DEF_OP(VFRSqrtScalarInsert) {
|
||||
|
||||
fmov(SubRegSize.Scalar, VTMP1.Q(), 1.0f);
|
||||
fsqrt(SubRegSize.Scalar, VTMP2, Src);
|
||||
fdiv(SubRegSize.Scalar, Dst, VTMP1, VTMP2);
|
||||
if (HostSupportsAFP) {
|
||||
fdiv(SubRegSize.Scalar, VTMP1, VTMP1, VTMP2);
|
||||
ins(SubRegSize.Vector, Dst, 0, VTMP1, 0);
|
||||
} else {
|
||||
fdiv(SubRegSize.Scalar, Dst, VTMP1, VTMP2);
|
||||
}
|
||||
};
|
||||
|
||||
auto ScalarEmitRPRES = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
frsqrte(SubRegSize.Scalar, Dst.S(), Src.S());
|
||||
frsqrte(SubRegSize.Scalar, Dst.D(), Src.D());
|
||||
};
|
||||
|
||||
std::array<ScalarUnaryOpCaller, 2> Handlers = {
|
||||
@@ -590,7 +587,28 @@ DEF_OP(VSToFVectorInsert) {
|
||||
// Claim the element size is 8-bytes.
|
||||
// Might be scalar 8-byte (cvtsi2ss xmm0, rax)
|
||||
// Might be vector i32v2 (cvtpi2ps xmm0, mm0)
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize * (HasTwoElements ? 2 : 1), Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
if (!HasTwoElements) {
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
return;
|
||||
}
|
||||
|
||||
// Dealing with the odd case of this being actually a vector operation rather than scalar.
|
||||
const auto Is256Bit = IROp->Size == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
constexpr auto Predicate = ARMEmitter::PReg::p0;
|
||||
|
||||
ScalarEmit(VTMP1, Vector2);
|
||||
if (!Op->ZeroUpperBits && Is256Bit) {
|
||||
if (Dst != Vector1) {
|
||||
mov(Dst.Z(), Vector1.Z());
|
||||
}
|
||||
ptrue(ARMEmitter::SubRegSize::i64Bit, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
} else {
|
||||
if (Dst != Vector1) {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSToFGPRInsert) {
|
||||
@@ -679,11 +697,11 @@ DEF_OP(VFCMPScalarInsert) {
|
||||
auto ScalarEmitEQ = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
switch (SubRegSize.Scalar) {
|
||||
case ARMEmitter::ScalarRegSize::i16Bit: {
|
||||
fcmeq(Dst.H(), Src1.H(), Src2.H());
|
||||
fcmeq(Dst.H(), Src2.H(), Src1.H());
|
||||
break;
|
||||
}
|
||||
case ARMEmitter::ScalarRegSize::i32Bit:
|
||||
case ARMEmitter::ScalarRegSize::i64Bit: fcmeq(SubRegSize.Scalar, Dst, Src1, Src2); break;
|
||||
case ARMEmitter::ScalarRegSize::i64Bit: fcmeq(SubRegSize.Scalar, Dst, Src2, Src1); break;
|
||||
default: break;
|
||||
}
|
||||
};
|
||||
@@ -748,11 +766,11 @@ DEF_OP(VFCMPScalarInsert) {
|
||||
[this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
switch (SubRegSize.Scalar) {
|
||||
case ARMEmitter::ScalarRegSize::i16Bit: {
|
||||
fcmeq(VTMP1.H(), Src1.H(), Src2.H());
|
||||
fcmeq(VTMP1.H(), Src2.H(), Src1.H());
|
||||
break;
|
||||
}
|
||||
case ARMEmitter::ScalarRegSize::i32Bit:
|
||||
case ARMEmitter::ScalarRegSize::i64Bit: fcmeq(SubRegSize.Scalar, VTMP1, Src1, Src2); break;
|
||||
case ARMEmitter::ScalarRegSize::i64Bit: fcmeq(SubRegSize.Scalar, VTMP1, Src2, Src1); break;
|
||||
default: break;
|
||||
}
|
||||
// If the destination is a temporary then it is going to do an insert after the operation.
|
||||
@@ -1750,6 +1768,7 @@ DEF_OP(VBSL) {
|
||||
const auto Op = IROp->C<IR::IROp_VBSL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorFalse = GetVReg(Op->VectorFalse.ID());
|
||||
@@ -1770,6 +1789,11 @@ DEF_OP(VBSL) {
|
||||
bsl(VTMP1.Z(), VTMP1.Z(), VectorFalse.Z(), VectorMask.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
} else if (!HostSupportsSVE256 && HostSupportsSVE128 && Is128Bit && Dst != VectorFalse && Dst != VectorTrue && Dst != VectorMask) {
|
||||
// Needs to move but SVE movprfx+bsl is slightly more efficient than ASIMD mov+bsl on CPUs that support
|
||||
// movprfx fusion and NOT zero-cycle vector register moves.
|
||||
movprfx(Dst.Z(), VectorTrue.Z());
|
||||
bsl(Dst.Z(), Dst.Z(), VectorFalse.Z(), VectorMask.Z());
|
||||
} else {
|
||||
if (VectorMask == Dst) {
|
||||
// Can use BSL without any moves.
|
||||
@@ -3960,5 +3984,257 @@ DEF_OP(VFCADD) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFMLA) {
|
||||
///< Dest = (Vector1 * Vector2) + Addend
|
||||
// Matches:
|
||||
// - SVE - FMLA
|
||||
// - ASIMD - FMLA
|
||||
// - Scalar - FMADD
|
||||
const auto Op = IROp->C<IR::IROp_VFMLA>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != VectorAddend) {
|
||||
DestTmp = VTMP1;
|
||||
mov(DestTmp.Z(), VectorAddend.Z());
|
||||
}
|
||||
|
||||
fmla(SubRegSize, DestTmp.Z(), Mask, Vector1.Z(), Vector2.Z());
|
||||
if (Dst != VectorAddend) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (IROp->ElementSize == OpSize) {
|
||||
if (IROp->ElementSize == 2) {
|
||||
fmadd(Dst.H(), Vector1.H(), Vector2.H(), VectorAddend.H());
|
||||
} else if (IROp->ElementSize == 4) {
|
||||
fmadd(Dst.S(), Vector1.S(), Vector2.S(), VectorAddend.S());
|
||||
} else if (IROp->ElementSize == 8) {
|
||||
fmadd(Dst.D(), Vector1.D(), Vector2.D(), VectorAddend.D());
|
||||
}
|
||||
return;
|
||||
}
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != VectorAddend) {
|
||||
DestTmp = VTMP1;
|
||||
mov(DestTmp.Q(), VectorAddend.Q());
|
||||
}
|
||||
if (OpSize == 16) {
|
||||
fmla(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
fmla(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
|
||||
}
|
||||
|
||||
if (Dst != VectorAddend) {
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFMLS) {
|
||||
///< Dest = (Vector1 * Vector2) - Addend
|
||||
// Matches:
|
||||
// - SVE - FNMLS
|
||||
// - ASIMD - FMLA (With negated addend)
|
||||
// - Scalar - FNMSUB
|
||||
const auto Op = IROp->C<IR::IROp_VFMLS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != VectorAddend) {
|
||||
DestTmp = VTMP1;
|
||||
mov(DestTmp.Z(), VectorAddend.Z());
|
||||
}
|
||||
|
||||
fnmls(SubRegSize, DestTmp.Z(), Mask, Vector1.Z(), Vector2.Z());
|
||||
if (Dst != VectorAddend) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else if (HostSupportsSVE128 && Is128Bit) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != VectorAddend) {
|
||||
DestTmp = VTMP1;
|
||||
mov(DestTmp.Z(), VectorAddend.Z());
|
||||
}
|
||||
|
||||
fnmls(SubRegSize, DestTmp.Z(), Mask, Vector1.Z(), Vector2.Z());
|
||||
if (Dst != VectorAddend) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (IROp->ElementSize == OpSize) {
|
||||
if (IROp->ElementSize == 2) {
|
||||
fnmsub(Dst.H(), Vector1.H(), Vector2.H(), VectorAddend.H());
|
||||
} else if (IROp->ElementSize == 4) {
|
||||
fnmsub(Dst.S(), Vector1.S(), Vector2.S(), VectorAddend.S());
|
||||
} else if (IROp->ElementSize == 8) {
|
||||
fnmsub(Dst.D(), Vector1.D(), Vector2.D(), VectorAddend.D());
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Addend needs to get negated to match correct behaviour here.
|
||||
ARMEmitter::VRegister DestTmp = VTMP1;
|
||||
if (Is128Bit) {
|
||||
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
|
||||
fmla(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
} else {
|
||||
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
|
||||
fmla(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
|
||||
mov(Dst.D(), DestTmp.D());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFNMLA) {
|
||||
///< Dest = (-Vector1 * Vector2) + Addend
|
||||
// Matches:
|
||||
// - SVE - FMLS
|
||||
// - ASIMD - FMLS
|
||||
// - Scalar - FMSUB
|
||||
const auto Op = IROp->C<IR::IROp_VFMLA>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != VectorAddend) {
|
||||
DestTmp = VTMP1;
|
||||
mov(DestTmp.Z(), VectorAddend.Z());
|
||||
}
|
||||
|
||||
fmls(SubRegSize, DestTmp.Z(), Mask, Vector1.Z(), Vector2.Z());
|
||||
if (Dst != VectorAddend) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (IROp->ElementSize == OpSize) {
|
||||
if (IROp->ElementSize == 2) {
|
||||
fmsub(Dst.H(), Vector1.H(), Vector2.H(), VectorAddend.H());
|
||||
} else if (IROp->ElementSize == 4) {
|
||||
fmsub(Dst.S(), Vector1.S(), Vector2.S(), VectorAddend.S());
|
||||
} else if (IROp->ElementSize == 8) {
|
||||
fmsub(Dst.D(), Vector1.D(), Vector2.D(), VectorAddend.D());
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != VectorAddend) {
|
||||
DestTmp = VTMP1;
|
||||
mov(DestTmp.Q(), VectorAddend.Q());
|
||||
}
|
||||
if (OpSize == 16) {
|
||||
fmls(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
fmls(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
|
||||
}
|
||||
|
||||
if (Dst != VectorAddend) {
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFNMLS) {
|
||||
///< Dest = (-Vector1 * Vector2) - Addend
|
||||
// Matches:
|
||||
// - SVE - FNMLA
|
||||
// - ASIMD - FMLS (With Negated addend)
|
||||
// - Scalar - FNMADD
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VFMLS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != VectorAddend) {
|
||||
DestTmp = VTMP1;
|
||||
mov(DestTmp.Z(), VectorAddend.Z());
|
||||
}
|
||||
|
||||
fnmla(SubRegSize, DestTmp.Z(), Mask, Vector1.Z(), Vector2.Z());
|
||||
if (Dst != VectorAddend) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else if (HostSupportsSVE128 && Is128Bit) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != VectorAddend) {
|
||||
DestTmp = VTMP1;
|
||||
mov(DestTmp.Z(), VectorAddend.Z());
|
||||
}
|
||||
|
||||
fnmla(SubRegSize, DestTmp.Z(), Mask, Vector1.Z(), Vector2.Z());
|
||||
if (Dst != VectorAddend) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (IROp->ElementSize == OpSize) {
|
||||
if (IROp->ElementSize == 2) {
|
||||
fnmadd(Dst.H(), Vector1.H(), Vector2.H(), VectorAddend.H());
|
||||
} else if (IROp->ElementSize == 4) {
|
||||
fnmadd(Dst.S(), Vector1.S(), Vector2.S(), VectorAddend.S());
|
||||
} else if (IROp->ElementSize == 8) {
|
||||
fnmadd(Dst.D(), Vector1.D(), Vector2.D(), VectorAddend.D());
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Addend needs to get negated to match correct behaviour here.
|
||||
ARMEmitter::VRegister DestTmp = VTMP1;
|
||||
if (Is128Bit) {
|
||||
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
|
||||
fmls(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
} else {
|
||||
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
|
||||
fmls(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
|
||||
mov(Dst.D(), DestTmp.D());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -113,6 +113,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, RIPAfterInst, 8);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
auto SyscallOp = _Syscall(Arguments[0], Arguments[1], Arguments[2], Arguments[3], Arguments[4], Arguments[5], Arguments[6], DefaultSyscallFlags);
|
||||
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_HANGOVER &&
|
||||
@@ -125,24 +126,20 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_BLOCK_END) {
|
||||
// RIP could have been updated after coming back from the Syscall.
|
||||
NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip));
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(NewRIP);
|
||||
ExitFunction(NewRIP);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
uint8_t* sha256 = (uint8_t*)(Op->PC + 2);
|
||||
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
// x86-64 ABI puts the function argument in RDI
|
||||
_Thunk(LoadGPRRegister(X86State::REG_RDI), *reinterpret_cast<SHA256Sum*>(sha256));
|
||||
Thunk(LoadGPRRegister(X86State::REG_RDI), *reinterpret_cast<SHA256Sum*>(sha256));
|
||||
} else {
|
||||
// x86 fastcall ABI puts the function argument in ECX
|
||||
_Thunk(LoadGPRRegister(X86State::REG_RCX), *reinterpret_cast<SHA256Sum*>(sha256));
|
||||
Thunk(LoadGPRRegister(X86State::REG_RCX), *reinterpret_cast<SHA256Sum*>(sha256));
|
||||
}
|
||||
|
||||
auto Constant = _Constant(GPRSize);
|
||||
@@ -152,10 +149,9 @@ void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
ExitFunction(NewRIP);
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
|
||||
@@ -186,11 +182,9 @@ void OpDispatchBuilder::RETOp(OpcodeArgs) {
|
||||
// ABI Optimization: Flags don't survive calls or rets
|
||||
if (CTX->Config.ABILocalFlags) {
|
||||
_InvalidateFlags(~0UL); // all flags
|
||||
InvalidatePF_AF();
|
||||
// Deferred flags are invalidated now
|
||||
InvalidateDeferredFlags();
|
||||
} else {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
|
||||
auto Constant = _Constant(GPRSize);
|
||||
@@ -207,10 +201,9 @@ void OpDispatchBuilder::RETOp(OpcodeArgs) {
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
ExitFunction(NewRIP);
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
|
||||
@@ -231,9 +224,6 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
auto Constant = _Constant(GPRSize);
|
||||
@@ -271,8 +261,7 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RSP, SP);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(NewRIP);
|
||||
ExitFunction(NewRIP);
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
|
||||
@@ -281,9 +270,8 @@ void OpDispatchBuilder::CallbackReturnOp(OpcodeArgs) {
|
||||
// Store the new RIP
|
||||
_CallbackReturn();
|
||||
auto NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip));
|
||||
CalculateDeferredFlags();
|
||||
// This ExitFunction won't actually get hit but needs to exist
|
||||
_ExitFunction(NewRIP);
|
||||
ExitFunction(NewRIP);
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
|
||||
@@ -407,6 +395,7 @@ void OpDispatchBuilder::PUSHOp(OpcodeArgs) {
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHREGOp(OpcodeArgs) {
|
||||
@@ -419,6 +408,7 @@ void OpDispatchBuilder::PUSHREGOp(OpcodeArgs) {
|
||||
auto NewSP = _Push(GPRSize, Size, Src, OldSP);
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHAOp(OpcodeArgs) {
|
||||
@@ -468,6 +458,7 @@ void OpDispatchBuilder::PUSHAOp(OpcodeArgs) {
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP, 4);
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
|
||||
template<uint32_t SegmentReg>
|
||||
@@ -664,18 +655,18 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) {
|
||||
// ABI Optimization: Flags don't survive calls or rets
|
||||
if (CTX->Config.ABILocalFlags) {
|
||||
_InvalidateFlags(~0UL); // all flags
|
||||
InvalidatePF_AF();
|
||||
// Deferred flags are invalidated now
|
||||
InvalidateDeferredFlags();
|
||||
} else {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
|
||||
auto ConstantPC = GetRelocatedPC(Op);
|
||||
// Call instruction only uses up to 32-bit signed displacement
|
||||
int64_t TargetOffset = Op->Src[0].Literal();
|
||||
uint64_t InstRIP = Op->PC + Op->InstSize;
|
||||
uint64_t TargetRIP = InstRIP + TargetOffset;
|
||||
|
||||
Ref JMPPCOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref NewRIP = _Add(IR::SizeToOpSize(GPRSize), ConstantPC, JMPPCOffset);
|
||||
Ref NewRIP = _Add(IR::SizeToOpSize(GPRSize), ConstantPC, _Constant(TargetOffset));
|
||||
|
||||
// Push the return address.
|
||||
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
|
||||
@@ -685,22 +676,16 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
|
||||
const uint64_t NextRIP = Op->PC + Op->InstSize;
|
||||
LOGMAN_THROW_A_FMT(Op->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
const uint64_t TargetRIP = Op->PC + Op->InstSize + Op->Src[0].Data.Literal.Value;
|
||||
|
||||
CalculateDeferredFlags();
|
||||
if (NextRIP != TargetRIP) {
|
||||
// Store the RIP
|
||||
_ExitFunction(NewRIP); // If we get here then leave the function now
|
||||
ExitFunction(NewRIP); // If we get here then leave the function now
|
||||
} else {
|
||||
NeedsBlockEnd = true;
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
BlockSetRIP = true;
|
||||
|
||||
const uint8_t Size = GetSrcSize(Op);
|
||||
@@ -716,8 +701,7 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
|
||||
// Store the RIP
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(JMPPCOffset); // If we get here then leave the function now
|
||||
ExitFunction(JMPPCOffset); // If we get here then leave the function now
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue) {
|
||||
@@ -843,8 +827,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
BlockSetRIP = true;
|
||||
|
||||
// Jump instruction only uses up to 32-bit signed displacement
|
||||
LOGMAN_THROW_A_FMT(Op->Src[0].IsLiteral(), "Src1 needs to be literal here");
|
||||
int64_t TargetOffset = Op->Src[0].Data.Literal.Value;
|
||||
int64_t TargetOffset = Op->Src[0].Literal();
|
||||
uint64_t InstRIP = Op->PC + Op->InstSize;
|
||||
uint64_t Target = InstRIP + TargetOffset;
|
||||
|
||||
@@ -892,7 +875,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
auto NewRIP = GetRelocatedPC(Op, TargetOffset);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
ExitFunction(NewRIP);
|
||||
}
|
||||
|
||||
// Failure to take branch
|
||||
@@ -910,7 +893,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
auto RIPTargetConst = GetRelocatedPC(Op);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPTargetConst);
|
||||
ExitFunction(RIPTargetConst);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -928,9 +911,7 @@ void OpDispatchBuilder::CondJUMPRCXOp(OpcodeArgs) {
|
||||
TakeBranch = _Constant(1);
|
||||
DoNotTakeBranch = _Constant(0);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Src[0].IsLiteral(), "Src1 needs to be literal here");
|
||||
|
||||
uint64_t Target = Op->PC + Op->InstSize + Op->Src[0].Data.Literal.Value;
|
||||
uint64_t Target = Op->PC + Op->InstSize + Op->Src[0].Literal();
|
||||
|
||||
Ref CondReg = LoadGPRRegister(X86State::REG_RCX, JcxGPRSize);
|
||||
|
||||
@@ -955,7 +936,7 @@ void OpDispatchBuilder::CondJUMPRCXOp(OpcodeArgs) {
|
||||
auto NewRIP = GetRelocatedPC(Op, Op->Src[0].Data.Literal.Value);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
ExitFunction(NewRIP);
|
||||
}
|
||||
|
||||
// Failure to take branch
|
||||
@@ -973,7 +954,7 @@ void OpDispatchBuilder::CondJUMPRCXOp(OpcodeArgs) {
|
||||
auto RIPTargetConst = GetRelocatedPC(Op);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPTargetConst);
|
||||
ExitFunction(RIPTargetConst);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -995,9 +976,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
OpSize = OpSize::i32Bit;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
|
||||
uint64_t Target = Op->PC + Op->InstSize + Op->Src[1].Data.Literal.Value;
|
||||
uint64_t Target = Op->PC + Op->InstSize + Op->Src[1].Literal();
|
||||
|
||||
Ref CondReg = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
CondReg = _Sub(OpSize, CondReg, _Constant(SrcSize * 8, 1));
|
||||
@@ -1031,7 +1010,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
auto NewRIP = GetRelocatedPC(Op, Op->Src[1].Data.Literal.Value);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
ExitFunction(NewRIP);
|
||||
}
|
||||
|
||||
// Failure to take branch
|
||||
@@ -1049,7 +1028,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
auto RIPTargetConst = GetRelocatedPC(Op);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPTargetConst);
|
||||
ExitFunction(RIPTargetConst);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1061,8 +1040,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
BlockSetRIP = true;
|
||||
|
||||
// Jump instruction only uses up to 32-bit signed displacement
|
||||
LOGMAN_THROW_A_FMT(Op->Src[0].IsLiteral(), "Src1 needs to be literal here");
|
||||
int64_t TargetOffset = Op->Src[0].Data.Literal.Value;
|
||||
int64_t TargetOffset = Op->Src[0].Literal();
|
||||
uint64_t InstRIP = Op->PC + Op->InstSize;
|
||||
uint64_t TargetRIP = InstRIP + TargetOffset;
|
||||
|
||||
@@ -1094,7 +1072,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
SetJumpTarget(Jump_, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
_ExitFunction(GetRelocatedPC(Op, TargetOffset));
|
||||
ExitFunction(GetRelocatedPC(Op, TargetOffset));
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -1105,7 +1083,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
auto NewRIP = _Add(OpSize::i64Bit, _Constant(TargetOffset), RIPTargetConst);
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
ExitFunction(NewRIP);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1118,10 +1096,9 @@ void OpDispatchBuilder::JUMPAbsoluteOp(OpcodeArgs) {
|
||||
// This uses ModRM to determine its location
|
||||
// No way to use this effectively in multiblock
|
||||
auto RIPOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPOffset);
|
||||
ExitFunction(RIPOffset);
|
||||
}
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
@@ -1143,7 +1120,7 @@ void OpDispatchBuilder::TESTOp(OpcodeArgs) {
|
||||
|
||||
HandleNZ00Write();
|
||||
CalculatePF(_AndWithFlags(IR::SizeToOpSize(Size), Dest, Src));
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
InvalidateAF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MOVSXDOp(OpcodeArgs) {
|
||||
@@ -1338,7 +1315,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
|
||||
case FEXCore::X86State::REG_RCX: // CS
|
||||
case FEXCore::X86State::REG_R9: // CS
|
||||
// CPL3 can't write to this
|
||||
_Break(FEXCore::IR::BreakDefinition {
|
||||
Break(FEXCore::IR::BreakDefinition {
|
||||
.ErrorRegister = 0,
|
||||
.Signal = SIGILL,
|
||||
.TrapNumber = 0,
|
||||
@@ -1472,8 +1449,7 @@ uint32_t OpDispatchBuilder::LoadConstantShift(X86Tables::DecodedOp Op, bool Is1B
|
||||
const uint32_t Size = GetSrcBitSize(Op);
|
||||
uint64_t Mask = Size == 64 ? 0x3F : 0x1F;
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
return Op->Src[1].Data.Literal.Value & Mask;
|
||||
return Op->Src[1].Literal() & Mask;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1907,10 +1883,9 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RORX(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src[1] needs to be literal here");
|
||||
const auto Amount = Op->Src[1].Data.Literal.Value;
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto SrcSizeBits = SrcSize * 8;
|
||||
const auto Amount = Op->Src[1].Literal() & (SrcSizeBits - 1);
|
||||
const auto GPRSize = CTX->GetGPRSize();
|
||||
|
||||
const auto DoRotation = Amount != 0 && Amount < SrcSizeBits;
|
||||
@@ -2947,7 +2922,7 @@ void OpDispatchBuilder::AAMOp(OpcodeArgs) {
|
||||
|
||||
SetNZ_ZeroCV(1, Res);
|
||||
CalculatePF(Res);
|
||||
_InvalidateFlags(1u << X86State::RFLAG_AF_RAW_LOC);
|
||||
InvalidateAF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AADOp(OpcodeArgs) {
|
||||
@@ -2962,7 +2937,7 @@ void OpDispatchBuilder::AADOp(OpcodeArgs) {
|
||||
|
||||
SetNZ_ZeroCV(1, Result);
|
||||
CalculatePF(Result);
|
||||
_InvalidateFlags(1u << X86State::RFLAG_AF_RAW_LOC);
|
||||
InvalidateAF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XLATOp(OpcodeArgs) {
|
||||
@@ -3005,9 +2980,7 @@ void OpDispatchBuilder::WriteSegmentReg(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::EnterOp(OpcodeArgs) {
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Src[0].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t Value = Op->Src[0].Data.Literal.Value;
|
||||
const uint64_t Value = Op->Src[0].Literal();
|
||||
|
||||
const uint16_t AllocSpace = Value & 0xFFFF;
|
||||
const uint8_t Level = (Value >> 16) & 0x1F;
|
||||
@@ -3436,7 +3409,6 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
auto Src1 = _LoadRegister(Core::CPUState::PF_AS_GREG, GPRClass, CTX->GetGPRSize());
|
||||
auto Src2 = _LoadRegister(Core::CPUState::AF_AS_GREG, GPRClass, CTX->GetGPRSize());
|
||||
GenerateFlags_SUB(Op, Src2, Src1);
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
auto Jump_ = Jump();
|
||||
|
||||
@@ -4023,7 +3995,6 @@ void OpDispatchBuilder::BeginFunction(uint64_t RIP, const fextl::vector<FEXCore:
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::Finalize() {
|
||||
// Calculate flags early.
|
||||
// This usually doesn't emit any IR but in the case of hitting the block instruction limit it will
|
||||
CalculateDeferredFlags();
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
@@ -4042,8 +4013,7 @@ void OpDispatchBuilder::Finalize() {
|
||||
|
||||
// We haven't emitted. Dump out to the dispatcher
|
||||
SetCurrentCodeBlock(Handler.second.BlockEntry);
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(_EntrypointOffset(IR::SizeToOpSize(GPRSize), Handler.first - Entry));
|
||||
ExitFunction(_EntrypointOffset(IR::SizeToOpSize(GPRSize), Handler.first - Entry));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4319,64 +4289,32 @@ AddressMode OpDispatchBuilder::SelectAddressMode(AddressMode A, bool AtomicTSO,
|
||||
};
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
uint8_t OpSize, uint32_t Flags, const LoadSourceOptions& Options) {
|
||||
LOGMAN_THROW_A_FMT(
|
||||
Operand.IsGPR() || Operand.IsLiteral() || Operand.IsGPRDirect() || Operand.IsGPRIndirect() || Operand.IsRIPRelative() || Operand.IsSIB(),
|
||||
"Unsupported Src type");
|
||||
|
||||
auto [Align, LoadData, ForceLoad, AccessType, AllowUpperGarbage] = Options;
|
||||
|
||||
AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
MemoryAccessType AccessType, bool IsLoad) {
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
bool LoadableType = false;
|
||||
|
||||
AddressMode A {};
|
||||
A.AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
A.NonTSO = AccessType == MemoryAccessType::NONTSO || AccessType == MemoryAccessType::STREAM;
|
||||
|
||||
if (Operand.IsLiteral()) {
|
||||
A.Offset = Operand.Data.Literal.Value;
|
||||
uint64_t width = Operand.Data.Literal.Size * 8;
|
||||
A.Offset = Operand.Literal();
|
||||
|
||||
if (Operand.Data.Literal.Size != 8) {
|
||||
if (Operand.Data.Literal.Size != 8 && IsLoad) {
|
||||
// zero extend
|
||||
uint64_t width = Operand.Data.Literal.Size * 8;
|
||||
A.Offset &= ((1ULL << width) - 1);
|
||||
}
|
||||
} else if (Operand.IsGPR()) {
|
||||
const auto gpr = Operand.Data.GPR.GPR;
|
||||
const auto highIndex = Operand.Data.GPR.HighBits ? 1 : 0;
|
||||
|
||||
if (gpr >= FEXCore::X86State::REG_MM_0) {
|
||||
A.Base = _LoadContext(OpSize, FPRClass, offsetof(FEXCore::Core::CPUState, mm[gpr - FEXCore::X86State::REG_MM_0]));
|
||||
} else if (gpr >= FEXCore::X86State::REG_XMM_0) {
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
|
||||
// Load the full register size if it is a XMM register source.
|
||||
A.Base = LoadXMMRegister(gprIndex);
|
||||
|
||||
// Now extract the subregister if it was a partial load /smaller/ than SSE size
|
||||
// TODO: Instead of doing the VMov implicitly on load, hunt down all use cases that require partial loads and do it after load.
|
||||
// We don't have information here to know if the operation needs zero upper bits or can contain data.
|
||||
if (!AllowUpperGarbage && OpSize < Core::CPUState::XMM_SSE_REG_SIZE) {
|
||||
A.Base = _VMov(OpSize, A.Base);
|
||||
}
|
||||
} else {
|
||||
A.Base = LoadGPRRegister(gpr, OpSize, highIndex ? 8 : 0, AllowUpperGarbage);
|
||||
}
|
||||
// Not an address, let the caller deal with it
|
||||
A.AddrSize = GPRSize;
|
||||
} else if (Operand.IsGPRDirect()) {
|
||||
A.Base = LoadGPRRegister(Operand.Data.GPR.GPR, GPRSize);
|
||||
|
||||
LoadableType = true;
|
||||
if (Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::DEFAULT) {
|
||||
AccessType = MemoryAccessType::NONTSO;
|
||||
}
|
||||
A.NonTSO |= IsNonTSOReg(AccessType, Operand.Data.GPR.GPR);
|
||||
} else if (Operand.IsGPRIndirect()) {
|
||||
A.Base = LoadGPRRegister(Operand.Data.GPRIndirect.GPR, GPRSize);
|
||||
A.Offset = Operand.Data.GPRIndirect.Displacement;
|
||||
|
||||
LoadableType = true;
|
||||
if (Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::DEFAULT) {
|
||||
AccessType = MemoryAccessType::NONTSO;
|
||||
}
|
||||
A.NonTSO |= IsNonTSOReg(AccessType, Operand.Data.GPRIndirect.GPR);
|
||||
} else if (Operand.IsRIPRelative()) {
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
A.Base = GetRelocatedPC(Op, Operand.Data.RIPLiteral.Value.s);
|
||||
@@ -4384,10 +4322,8 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
// 32bit this isn't RIP relative but instead absolute
|
||||
A.Offset = Operand.Data.RIPLiteral.Value.u;
|
||||
}
|
||||
|
||||
LoadableType = true;
|
||||
} else if (Operand.IsSIB()) {
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
const bool IsVSIB = IsLoad && ((Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0);
|
||||
|
||||
if (Operand.Data.SIB.Base != FEXCore::X86State::REG_INVALID) {
|
||||
A.Base = LoadGPRRegister(Operand.Data.SIB.Base, GPRSize);
|
||||
@@ -4408,23 +4344,47 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
}
|
||||
|
||||
A.Offset = Operand.Data.SIB.Offset;
|
||||
|
||||
if ((Operand.Data.SIB.Base == FEXCore::X86State::REG_RSP || Operand.Data.SIB.Index == FEXCore::X86State::REG_RSP) &&
|
||||
AccessType == MemoryAccessType::DEFAULT) {
|
||||
AccessType = MemoryAccessType::NONTSO;
|
||||
}
|
||||
|
||||
LoadableType = true;
|
||||
A.NonTSO |= IsNonTSOReg(AccessType, Operand.Data.SIB.Base) || IsNonTSOReg(AccessType, Operand.Data.SIB.Index);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unknown Src Type: {}\n", Operand.Type);
|
||||
}
|
||||
|
||||
if ((LoadableType && LoadData) || ForceLoad) {
|
||||
return A;
|
||||
}
|
||||
|
||||
|
||||
Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
uint8_t OpSize, uint32_t Flags, const LoadSourceOptions& Options) {
|
||||
auto [Align, LoadData, ForceLoad, AccessType, AllowUpperGarbage] = Options;
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
|
||||
|
||||
if (Operand.IsGPR()) {
|
||||
const auto gpr = Operand.Data.GPR.GPR;
|
||||
const auto highIndex = Operand.Data.GPR.HighBits ? 1 : 0;
|
||||
|
||||
if (gpr >= FEXCore::X86State::REG_MM_0) {
|
||||
A.Base = _LoadContext(OpSize, FPRClass, offsetof(FEXCore::Core::CPUState, mm[gpr - FEXCore::X86State::REG_MM_0]));
|
||||
} else if (gpr >= FEXCore::X86State::REG_XMM_0) {
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
|
||||
// Load the full register size if it is a XMM register source.
|
||||
A.Base = LoadXMMRegister(gprIndex);
|
||||
|
||||
// Now extract the subregister if it was a partial load /smaller/ than SSE size
|
||||
// TODO: Instead of doing the VMov implicitly on load, hunt down all use cases that require partial loads and do it after load.
|
||||
// We don't have information here to know if the operation needs zero upper bits or can contain data.
|
||||
if (!AllowUpperGarbage && OpSize < Core::CPUState::XMM_SSE_REG_SIZE) {
|
||||
A.Base = _VMov(OpSize, A.Base);
|
||||
}
|
||||
} else {
|
||||
A.Base = LoadGPRRegister(gpr, OpSize, highIndex ? 8 : 0, AllowUpperGarbage);
|
||||
}
|
||||
}
|
||||
|
||||
if ((IsOperandMem(Operand, true) && LoadData) || ForceLoad) {
|
||||
A = AddSegmentToAddress(A, Flags);
|
||||
|
||||
bool ForceNonTSO = AccessType == MemoryAccessType::NONTSO || AccessType == MemoryAccessType::STREAM;
|
||||
return _LoadMemAutoTSO(Class, OpSize, A, Align == -1 ? OpSize : Align, ForceNonTSO);
|
||||
return _LoadMemAutoTSO(Class, OpSize, A, Align == -1 ? OpSize : Align);
|
||||
} else {
|
||||
return LoadEffectiveAddress(A, AllowUpperGarbage);
|
||||
}
|
||||
@@ -4455,8 +4415,7 @@ Ref OpDispatchBuilder::LoadGPRRegister(uint32_t GPR, int8_t Size, uint8_t Offset
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadXMMRegister(uint32_t XMM) {
|
||||
const auto VectorSize = CTX->HostFeatures.SupportsAVX ? 32 : 16;
|
||||
return _LoadRegister(XMM, FPRClass, VectorSize);
|
||||
return _LoadRegister(XMM, FPRClass, GetGuestVectorLength());
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::StoreGPRRegister(uint32_t GPR, const Ref Src, int8_t Size, uint8_t Offset) {
|
||||
@@ -4476,8 +4435,7 @@ void OpDispatchBuilder::StoreGPRRegister(uint32_t GPR, const Ref Src, int8_t Siz
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::StoreXMMRegister(uint32_t XMM, const Ref Src) {
|
||||
const auto VectorSize = CTX->HostFeatures.SupportsAVX ? 32 : 16;
|
||||
_StoreRegister(Src, XMM, FPRClass, VectorSize);
|
||||
_StoreRegister(Src, XMM, FPRClass, GetGuestVectorLength());
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
@@ -4489,28 +4447,17 @@ Ref OpDispatchBuilder::LoadSource(RegisterClassType Class, const X86Tables::Deco
|
||||
void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op,
|
||||
const FEXCore::X86Tables::DecodedOperand& Operand, const Ref Src, uint8_t OpSize,
|
||||
int8_t Align, MemoryAccessType AccessType) {
|
||||
LOGMAN_THROW_A_FMT(
|
||||
Operand.IsGPR() || Operand.IsLiteral() || Operand.IsGPRDirect() || Operand.IsGPRIndirect() || Operand.IsRIPRelative() || Operand.IsSIB(),
|
||||
"Unsupported Dest type");
|
||||
if (Operand.IsGPR()) {
|
||||
// 8Bit and 16bit destination types store their result without effecting the upper bits
|
||||
// 32bit ops ZEXT the result to 64bit
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
// 8Bit and 16bit destination types store their result without effecting the upper bits
|
||||
// 32bit ops ZEXT the result to 64bit
|
||||
bool MemStore = false;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
AddressMode A {};
|
||||
A.AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
|
||||
if (Operand.IsLiteral()) {
|
||||
A.Offset = Operand.Data.Literal.Value;
|
||||
MemStore = true; // Literals are ONLY hardcoded memory destinations
|
||||
} else if (Operand.IsGPR()) {
|
||||
const auto gpr = Operand.Data.GPR.GPR;
|
||||
if (gpr >= FEXCore::X86State::REG_MM_0) {
|
||||
_StoreContext(OpSize, Class, Src, offsetof(FEXCore::Core::CPUState, mm[gpr - FEXCore::X86State::REG_MM_0]));
|
||||
} else if (gpr >= FEXCore::X86State::REG_XMM_0) {
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
const auto VectorSize = CTX->HostFeatures.SupportsAVX ? 32 : 16;
|
||||
const auto VectorSize = (CTX->HostFeatures.SupportsSVE256 && CTX->HostFeatures.SupportsAVX) ? 32 : 16;
|
||||
|
||||
auto Result = Src;
|
||||
if (OpSize != VectorSize) {
|
||||
@@ -4550,63 +4497,21 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if (Operand.IsGPRDirect()) {
|
||||
A.Base = LoadGPRRegister(Operand.Data.GPR.GPR, GPRSize);
|
||||
|
||||
MemStore = true;
|
||||
if (Operand.Data.GPR.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::DEFAULT) {
|
||||
AccessType = MemoryAccessType::NONTSO;
|
||||
}
|
||||
} else if (Operand.IsGPRIndirect()) {
|
||||
A.Base = LoadGPRRegister(Operand.Data.GPRIndirect.GPR, GPRSize);
|
||||
A.Offset = Operand.Data.GPRIndirect.Displacement;
|
||||
|
||||
MemStore = true;
|
||||
if (Operand.Data.GPRIndirect.GPR == FEXCore::X86State::REG_RSP && AccessType == MemoryAccessType::DEFAULT) {
|
||||
AccessType = MemoryAccessType::NONTSO;
|
||||
}
|
||||
} else if (Operand.IsRIPRelative()) {
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
A.Base = GetRelocatedPC(Op, Operand.Data.RIPLiteral.Value.s);
|
||||
} else {
|
||||
// 32bit this isn't RIP relative but instead absolute
|
||||
A.Offset = Operand.Data.RIPLiteral.Value.u;
|
||||
}
|
||||
MemStore = true;
|
||||
} else if (Operand.IsSIB()) {
|
||||
if (Operand.Data.SIB.Base != FEXCore::X86State::REG_INVALID) {
|
||||
A.Base = LoadGPRRegister(Operand.Data.SIB.Base, GPRSize);
|
||||
}
|
||||
|
||||
if (Operand.Data.SIB.Index != FEXCore::X86State::REG_INVALID) {
|
||||
A.Index = LoadGPRRegister(Operand.Data.SIB.Index, GPRSize);
|
||||
}
|
||||
|
||||
A.IndexScale = Operand.Data.SIB.Scale;
|
||||
A.Offset = Operand.Data.SIB.Offset;
|
||||
|
||||
if ((Operand.Data.SIB.Base == FEXCore::X86State::REG_RSP || Operand.Data.SIB.Index == FEXCore::X86State::REG_RSP) &&
|
||||
AccessType == MemoryAccessType::DEFAULT) {
|
||||
AccessType = MemoryAccessType::NONTSO;
|
||||
}
|
||||
|
||||
MemStore = true;
|
||||
return;
|
||||
}
|
||||
|
||||
if (MemStore) {
|
||||
A = AddSegmentToAddress(A, Op->Flags);
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
|
||||
A = AddSegmentToAddress(A, Op->Flags);
|
||||
|
||||
if (OpSize == 10) {
|
||||
Ref MemStoreDst = LoadEffectiveAddress(A);
|
||||
if (OpSize == 10) {
|
||||
Ref MemStoreDst = LoadEffectiveAddress(A);
|
||||
|
||||
// For X87 extended doubles, split before storing
|
||||
_StoreMem(FPRClass, 8, MemStoreDst, Src, Align);
|
||||
auto Upper = _VExtractToGPR(16, 8, Src, 1);
|
||||
_StoreMem(GPRClass, 2, Upper, MemStoreDst, _Constant(8), std::min<uint8_t>(Align, 8), MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
bool ForceNonTSO = AccessType == MemoryAccessType::NONTSO || AccessType == MemoryAccessType::STREAM;
|
||||
_StoreMemAutoTSO(Class, OpSize, A, Src, Align == -1 ? OpSize : Align, ForceNonTSO);
|
||||
}
|
||||
// For X87 extended doubles, split before storing
|
||||
_StoreMem(FPRClass, 8, MemStoreDst, Src, Align);
|
||||
auto Upper = _VExtractToGPR(16, 8, Src, 1);
|
||||
_StoreMem(GPRClass, 2, Upper, MemStoreDst, _Constant(8), std::min<uint8_t>(Align, 8), MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
_StoreMemAutoTSO(Class, OpSize, A, Src, Align == -1 ? OpSize : Align);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4625,6 +4530,16 @@ OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx)
|
||||
, CTX {ctx} {
|
||||
ResetWorkingList();
|
||||
InstallHostSpecificOpcodeHandlers();
|
||||
|
||||
if (CTX->HostFeatures.SupportsAVX && CTX->HostFeatures.SupportsSVE256) {
|
||||
SaveAVXStateFunc = &OpDispatchBuilder::SaveAVXState;
|
||||
RestoreAVXStateFunc = &OpDispatchBuilder::RestoreAVXState;
|
||||
DefaultAVXStateFunc = &OpDispatchBuilder::DefaultAVXState;
|
||||
} else if (CTX->HostFeatures.SupportsAVX) {
|
||||
SaveAVXStateFunc = &OpDispatchBuilder::AVX128_SaveAVXState;
|
||||
RestoreAVXStateFunc = &OpDispatchBuilder::AVX128_RestoreAVXState;
|
||||
DefaultAVXStateFunc = &OpDispatchBuilder::AVX128_DefaultAVXState;
|
||||
}
|
||||
}
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator& Allocator)
|
||||
: IREmitter {Allocator}
|
||||
@@ -4709,7 +4624,7 @@ void OpDispatchBuilder::ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCor
|
||||
case FEXCore::IR::IROps::OP_ANDWITHFLAGS: {
|
||||
HandleNZ00Write();
|
||||
CalculatePF(Result);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
InvalidateAF();
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
@@ -4818,7 +4733,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
|
||||
auto NewRIP = GetRelocatedPC(Op);
|
||||
_StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
_Break(Reason);
|
||||
Break(Reason);
|
||||
|
||||
// Make sure to start a new block after ending this one
|
||||
auto JumpTarget = CreateNewCodeBlockAfter(FalseBlock);
|
||||
@@ -4827,7 +4742,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
StartNewBlock();
|
||||
} else {
|
||||
BlockSetRIP = true;
|
||||
_Break(Reason);
|
||||
Break(Reason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4972,15 +4887,12 @@ void OpDispatchBuilder::RDRANDOp(OpcodeArgs) {
|
||||
|
||||
|
||||
void OpDispatchBuilder::BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDefinition) {
|
||||
// Ensure flags are calculated on invalid op.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
// We don't actually support this instruction
|
||||
// Multiblock may hit it though
|
||||
_StoreContext(GPRSize, GPRClass, GetRelocatedPC(Op, -Op->InstSize), offsetof(FEXCore::Core::CPUState, rip));
|
||||
_Break(BreakDefinition);
|
||||
Break(BreakDefinition);
|
||||
|
||||
BlockSetRIP = true;
|
||||
|
||||
@@ -5164,8 +5076,8 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b00, 0x56), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VOR, 16>},
|
||||
{OPD(1, 0b01, 0x56), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VOR, 16>},
|
||||
|
||||
{OPD(1, 0b00, 0x57), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VXOR, 16>},
|
||||
{OPD(1, 0b01, 0x57), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VXOR, 16>},
|
||||
{OPD(1, 0b00, 0x57), 1, &OpDispatchBuilder::AVXVectorXOROp},
|
||||
{OPD(1, 0b01, 0x57), 1, &OpDispatchBuilder::AVXVectorXOROp},
|
||||
|
||||
{OPD(1, 0b00, 0x58), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 4>},
|
||||
{OPD(1, 0b01, 0x58), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 8>},
|
||||
@@ -5297,7 +5209,7 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b01, 0xEC), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VSQADD, 1>},
|
||||
{OPD(1, 0b01, 0xED), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VSQADD, 2>},
|
||||
{OPD(1, 0b01, 0xEE), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VSMAX, 2>},
|
||||
{OPD(1, 0b01, 0xEF), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VXOR, 16>},
|
||||
{OPD(1, 0b01, 0xEF), 1, &OpDispatchBuilder::AVXVectorXOROp},
|
||||
|
||||
{OPD(1, 0b11, 0xF0), 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{OPD(1, 0b01, 0xF1), 1, &OpDispatchBuilder::VPSLLOp<2>},
|
||||
@@ -5335,6 +5247,7 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(2, 0b01, 0x0E), 1, &OpDispatchBuilder::VTESTPOp<4>},
|
||||
{OPD(2, 0b01, 0x0F), 1, &OpDispatchBuilder::VTESTPOp<8>},
|
||||
|
||||
{OPD(2, 0b01, 0x13), 1, &OpDispatchBuilder::VCVTPH2PSOp},
|
||||
{OPD(2, 0b01, 0x16), 1, &OpDispatchBuilder::VPERMDOp},
|
||||
{OPD(2, 0b01, 0x17), 1, &OpDispatchBuilder::PTestOp},
|
||||
{OPD(2, 0b01, 0x18), 1, &OpDispatchBuilder::VBROADCASTOp<4>},
|
||||
@@ -5394,6 +5307,47 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(2, 0b01, 0x8C), 1, &OpDispatchBuilder::VPMASKMOVOp<false>},
|
||||
{OPD(2, 0b01, 0x8E), 1, &OpDispatchBuilder::VPMASKMOVOp<true>},
|
||||
|
||||
{OPD(2, 0b01, 0x90), 1, &OpDispatchBuilder::VPGATHER<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x91), 1, &OpDispatchBuilder::VPGATHER<OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x92), 1, &OpDispatchBuilder::VPGATHER<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x93), 1, &OpDispatchBuilder::VPGATHER<OpSize::i64Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x96), 1, &OpDispatchBuilder::VFMADDSUB<1, 3, 2>},
|
||||
{OPD(2, 0b01, 0x97), 1, &OpDispatchBuilder::VFMSUBADD<1, 3, 2>},
|
||||
|
||||
{OPD(2, 0b01, 0x98), 1, &OpDispatchBuilder::VFMADD<false, 1, 3, 2>},
|
||||
{OPD(2, 0b01, 0x99), 1, &OpDispatchBuilder::VFMADD<true, 1, 3, 2>},
|
||||
{OPD(2, 0b01, 0x9A), 1, &OpDispatchBuilder::VFMSUB<false, 1, 3, 2>},
|
||||
{OPD(2, 0b01, 0x9B), 1, &OpDispatchBuilder::VFMSUB<true, 1, 3, 2>},
|
||||
{OPD(2, 0b01, 0x9C), 1, &OpDispatchBuilder::VFNMADD<false, 1, 3, 2>},
|
||||
{OPD(2, 0b01, 0x9D), 1, &OpDispatchBuilder::VFNMADD<true, 1, 3, 2>},
|
||||
{OPD(2, 0b01, 0x9E), 1, &OpDispatchBuilder::VFNMSUB<false, 1, 3, 2>},
|
||||
{OPD(2, 0b01, 0x9F), 1, &OpDispatchBuilder::VFNMSUB<true, 1, 3, 2>},
|
||||
|
||||
{OPD(2, 0b01, 0xA8), 1, &OpDispatchBuilder::VFMADD<false, 2, 1, 3>},
|
||||
{OPD(2, 0b01, 0xA9), 1, &OpDispatchBuilder::VFMADD<true, 2, 1, 3>},
|
||||
{OPD(2, 0b01, 0xAA), 1, &OpDispatchBuilder::VFMSUB<false, 2, 1, 3>},
|
||||
{OPD(2, 0b01, 0xAB), 1, &OpDispatchBuilder::VFMSUB<true, 2, 1, 3>},
|
||||
{OPD(2, 0b01, 0xAC), 1, &OpDispatchBuilder::VFNMADD<false, 2, 1, 3>},
|
||||
{OPD(2, 0b01, 0xAD), 1, &OpDispatchBuilder::VFNMADD<true, 2, 1, 3>},
|
||||
{OPD(2, 0b01, 0xAE), 1, &OpDispatchBuilder::VFNMSUB<false, 2, 1, 3>},
|
||||
{OPD(2, 0b01, 0xAF), 1, &OpDispatchBuilder::VFNMSUB<true, 2, 1, 3>},
|
||||
|
||||
{OPD(2, 0b01, 0xB8), 1, &OpDispatchBuilder::VFMADD<false, 2, 3, 1>},
|
||||
{OPD(2, 0b01, 0xB9), 1, &OpDispatchBuilder::VFMADD<true, 2, 3, 1>},
|
||||
{OPD(2, 0b01, 0xBA), 1, &OpDispatchBuilder::VFMSUB<false, 2, 3, 1>},
|
||||
{OPD(2, 0b01, 0xBB), 1, &OpDispatchBuilder::VFMSUB<true, 2, 3, 1>},
|
||||
{OPD(2, 0b01, 0xBC), 1, &OpDispatchBuilder::VFNMADD<false, 2, 3, 1>},
|
||||
{OPD(2, 0b01, 0xBD), 1, &OpDispatchBuilder::VFNMADD<true, 2, 3, 1>},
|
||||
{OPD(2, 0b01, 0xBE), 1, &OpDispatchBuilder::VFNMSUB<false, 2, 3, 1>},
|
||||
{OPD(2, 0b01, 0xBF), 1, &OpDispatchBuilder::VFNMSUB<true, 2, 3, 1>},
|
||||
|
||||
{OPD(2, 0b01, 0xA6), 1, &OpDispatchBuilder::VFMADDSUB<2, 1, 3>},
|
||||
{OPD(2, 0b01, 0xA7), 1, &OpDispatchBuilder::VFMSUBADD<2, 1, 3>},
|
||||
|
||||
{OPD(2, 0b01, 0xB6), 1, &OpDispatchBuilder::VFMADDSUB<2, 3, 1>},
|
||||
{OPD(2, 0b01, 0xB7), 1, &OpDispatchBuilder::VFMSUBADD<2, 3, 1>},
|
||||
|
||||
{OPD(2, 0b01, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
|
||||
{OPD(2, 0b01, 0xDC), 1, &OpDispatchBuilder::VAESEncOp},
|
||||
{OPD(2, 0b01, 0xDD), 1, &OpDispatchBuilder::VAESEncLastOp},
|
||||
@@ -5422,6 +5376,7 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
|
||||
{OPD(3, 0b01, 0x18), 1, &OpDispatchBuilder::VINSERTOp},
|
||||
{OPD(3, 0b01, 0x19), 1, &OpDispatchBuilder::VEXTRACT128Op},
|
||||
{OPD(3, 0b01, 0x1D), 1, &OpDispatchBuilder::VCVTPS2PHOp},
|
||||
{OPD(3, 0b01, 0x20), 1, &OpDispatchBuilder::VPINSRBOp},
|
||||
{OPD(3, 0b01, 0x21), 1, &OpDispatchBuilder::VINSERTPSOp},
|
||||
{OPD(3, 0b01, 0x22), 1, &OpDispatchBuilder::VPINSRDQOp},
|
||||
@@ -5498,14 +5453,18 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
InstallToTable(FEXCore::X86Tables::SecondInstGroupOps, SecondaryExtensionOp_RDRAND);
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
if (CTX->HostFeatures.SupportsAVX && CTX->HostFeatures.SupportsSVE256) {
|
||||
InstallToTable(FEXCore::X86Tables::VEXTableOps, AVXTable);
|
||||
InstallToTable(FEXCore::X86Tables::VEXTableGroupOps, VEXTableGroupOps);
|
||||
if (CTX->HostFeatures.SupportsPMULL_128Bit) {
|
||||
InstallToTable(FEXCore::X86Tables::VEXTableOps, VEX_PCLMUL);
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsAVX) {
|
||||
InstallAVX128Handlers();
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsPMULL_128Bit) {
|
||||
InstallToTable(FEXCore::X86Tables::H0F3ATableOps, H0F3A_PCLMUL);
|
||||
InstallToTable(FEXCore::X86Tables::VEXTableOps, VEX_PCLMUL);
|
||||
}
|
||||
Initialized = true;
|
||||
}
|
||||
@@ -5678,9 +5637,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x52, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4>},
|
||||
{0x54, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VAND, 16>},
|
||||
{0x55, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VBIC, 8>},
|
||||
{0x55, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VANDN, 8>},
|
||||
{0x56, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VOR, 16>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VXOR, 16>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFADD, 4>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFMUL, 4>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4>},
|
||||
@@ -5724,7 +5683,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xDC, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VUQADD, 1>},
|
||||
{0xDD, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VUQADD, 2>},
|
||||
{0xDE, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VUMAX, 1>},
|
||||
{0xDF, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VBIC, 8>},
|
||||
{0xDF, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VANDN, 8>},
|
||||
{0xE0, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VURAVG, 1>},
|
||||
{0xE1, 1, &OpDispatchBuilder::PSRAOp<2>},
|
||||
{0xE2, 1, &OpDispatchBuilder::PSRAOp<4>},
|
||||
@@ -5739,7 +5698,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xEC, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSQADD, 1>},
|
||||
{0xED, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSQADD, 2>},
|
||||
{0xEE, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSMAX, 2>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VXOR, 8>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
|
||||
{0xF1, 1, &OpDispatchBuilder::PSLL<2>},
|
||||
{0xF2, 1, &OpDispatchBuilder::PSLL<4>},
|
||||
@@ -5967,9 +5926,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x50, 1, &OpDispatchBuilder::MOVMSKOp<8>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8>},
|
||||
{0x54, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VAND, 16>},
|
||||
{0x55, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VBIC, 8>},
|
||||
{0x55, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VANDN, 8>},
|
||||
{0x56, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VOR, 16>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VXOR, 16>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFADD, 8>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFMUL, 8>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8>},
|
||||
@@ -6024,7 +5983,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xDC, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VUQADD, 1>},
|
||||
{0xDD, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VUQADD, 2>},
|
||||
{0xDE, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VUMAX, 1>},
|
||||
{0xDF, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VBIC, 8>},
|
||||
{0xDF, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VANDN, 8>},
|
||||
{0xE0, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VURAVG, 1>},
|
||||
{0xE1, 1, &OpDispatchBuilder::PSRAOp<2>},
|
||||
{0xE2, 1, &OpDispatchBuilder::PSRAOp<4>},
|
||||
@@ -6040,7 +5999,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xEC, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSQADD, 1>},
|
||||
{0xED, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSQADD, 2>},
|
||||
{0xEE, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VSMAX, 2>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VXOR, 16>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
|
||||
{0xF1, 1, &OpDispatchBuilder::PSLL<2>},
|
||||
{0xF2, 1, &OpDispatchBuilder::PSLL<4>},
|
||||
|
||||
@@ -79,6 +79,7 @@ struct AddressMode {
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
uint8_t AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
class OpDispatchBuilder final : public IREmitter {
|
||||
@@ -134,6 +135,7 @@ public:
|
||||
// Changes get stored out by CalculateDeferredFlags.
|
||||
CachedNZCV = nullptr;
|
||||
PossiblySetNZCVBits = ~0U;
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// New block needs to reset segment telemetry.
|
||||
SegmentsNeedReadCheck = ~0U;
|
||||
@@ -171,6 +173,18 @@ public:
|
||||
auto Placeholder = _InlineConstant(0);
|
||||
return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true);
|
||||
}
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP) {
|
||||
CalculateDeferredFlags();
|
||||
return _ExitFunction(NewRIP);
|
||||
}
|
||||
IRPair<IROp_Break> Break(BreakDefinition Reason) {
|
||||
CalculateDeferredFlags();
|
||||
return _Break(Reason);
|
||||
}
|
||||
IRPair<IROp_Thunk> Thunk(Ref ArgPtr, SHA256Sum ThunkNameHash) {
|
||||
CalculateDeferredFlags();
|
||||
return _Thunk(ArgPtr, ThunkNameHash);
|
||||
}
|
||||
|
||||
bool FinishOp(uint64_t NextRIP, bool LastOp) {
|
||||
// If we are switching to a new block and this current block has yet to set a RIP
|
||||
@@ -184,9 +198,6 @@ public:
|
||||
// cmp qword [rdi-8], 0
|
||||
// jne .label
|
||||
if (LastOp && !BlockSetRIP) {
|
||||
// Calculate flags first
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto it = JumpTargets.find(NextRIP);
|
||||
if (it == JumpTargets.end()) {
|
||||
|
||||
@@ -194,7 +205,7 @@ public:
|
||||
// If we don't have a jump target to a new block then we have to leave
|
||||
// Set the RIP to the next instruction and leave
|
||||
auto RelocatedNextRIP = _EntrypointOffset(IR::SizeToOpSize(GPRSize), NextRIP - Entry);
|
||||
_ExitFunction(RelocatedNextRIP);
|
||||
ExitFunction(RelocatedNextRIP);
|
||||
} else if (it != JumpTargets.end()) {
|
||||
Jump(it->second.BlockEntry);
|
||||
return true;
|
||||
@@ -442,6 +453,8 @@ public:
|
||||
void MOVSSOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorALUOp(OpcodeArgs);
|
||||
void VectorXOROp(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorALUROp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
@@ -533,6 +546,7 @@ public:
|
||||
// AVX Ops
|
||||
template<IROps IROp, size_t ElementSize>
|
||||
void AVXVectorALUOp(OpcodeArgs);
|
||||
void AVXVectorXOROp(OpcodeArgs);
|
||||
template<IROps IROp, size_t ElementSize>
|
||||
void AVXVectorUnaryOp(OpcodeArgs);
|
||||
|
||||
@@ -572,6 +586,8 @@ public:
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
RoundType TranslateRoundType(uint8_t Mode);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void InsertScalarRound(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
@@ -598,6 +614,7 @@ public:
|
||||
|
||||
void VANDNOp(OpcodeArgs);
|
||||
|
||||
Ref VBLENDOpImpl(uint32_t VecSize, uint32_t ElementSize, Ref Src1, Ref Src2, Ref ZeroRegister, uint64_t Selector);
|
||||
void VBLENDPDOp(OpcodeArgs);
|
||||
void VPBLENDDOp(OpcodeArgs);
|
||||
void VPBLENDWOp(OpcodeArgs);
|
||||
@@ -649,12 +666,18 @@ public:
|
||||
void VPCMPISTRIOp(OpcodeArgs);
|
||||
void VPCMPISTRMOp(OpcodeArgs);
|
||||
|
||||
void VCVTPH2PSOp(OpcodeArgs);
|
||||
void VCVTPS2PHOp(OpcodeArgs);
|
||||
|
||||
Ref VPERMDIndices(OpSize DstSize, Ref Indices, Ref IndexMask, Ref Repeating3210);
|
||||
void VPERM2Op(OpcodeArgs);
|
||||
void VPERMDOp(OpcodeArgs);
|
||||
void VPERMQOp(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VPERMILImmOp(OpcodeArgs);
|
||||
|
||||
Ref VPERMILRegOpImpl(OpSize DstSize, size_t ElementSize, Ref Src, Ref Indices);
|
||||
template<size_t ElementSize>
|
||||
void VPERMILRegOp(OpcodeArgs);
|
||||
|
||||
@@ -937,6 +960,34 @@ public:
|
||||
void AESDecLastOp(OpcodeArgs);
|
||||
void AESKeyGenAssist(OpcodeArgs);
|
||||
|
||||
void VFMAImpl(OpcodeArgs, IROps IROp, bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
void VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
|
||||
template<bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void VFMADD(OpcodeArgs);
|
||||
template<bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void VFMSUB(OpcodeArgs);
|
||||
template<bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void VFNMADD(OpcodeArgs);
|
||||
template<bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void VFNMSUB(OpcodeArgs);
|
||||
|
||||
template<uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void VFMADDSUB(OpcodeArgs);
|
||||
template<uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void VFMSUBADD(OpcodeArgs);
|
||||
|
||||
struct RefVSIB {
|
||||
Ref Low, High;
|
||||
Ref BaseAddr;
|
||||
int32_t Displacement;
|
||||
uint8_t Scale;
|
||||
};
|
||||
|
||||
RefVSIB LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags);
|
||||
template<size_t AddrElementSize>
|
||||
void VPGATHER(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, size_t DstElementSize, bool Signed>
|
||||
void ExtendVectorElements(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
@@ -947,6 +998,7 @@ public:
|
||||
|
||||
template<size_t ElementSize>
|
||||
void VectorVariableBlend(OpcodeArgs);
|
||||
void PTestOpImpl(OpSize Size, Ref Dest, Ref Src);
|
||||
void PTestOp(OpcodeArgs);
|
||||
void PHMINPOSUWOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
@@ -962,6 +1014,264 @@ public:
|
||||
void UnimplementedOp(OpcodeArgs);
|
||||
void PermissionRestrictedOp(OpcodeArgs);
|
||||
|
||||
// AVX 128-bit operations
|
||||
Ref AVX128_LoadXMMRegister(uint32_t XMM, bool High);
|
||||
void AVX128_StoreXMMRegister(uint32_t XMM, const Ref Src, bool High);
|
||||
|
||||
struct RefPair {
|
||||
Ref Low, High;
|
||||
};
|
||||
|
||||
RefPair AVX128_Zext(Ref R) {
|
||||
RefPair Pair;
|
||||
Pair.Low = R;
|
||||
Pair.High = LoadZeroVector(OpSize::i128Bit);
|
||||
return Pair;
|
||||
}
|
||||
|
||||
RefPair AVX128_LoadSource_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
bool NeedsHigh, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
|
||||
RefVSIB AVX128_LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags, bool NeedsHigh);
|
||||
void AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand, const RefPair Src,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void InstallAVX128Handlers();
|
||||
void AVX128_VMOVScalarImpl(OpcodeArgs, size_t ElementSize);
|
||||
void AVX128_VectorALUImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVX128_VectorUnaryImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVX128_VectorUnaryImpl(OpcodeArgs, size_t SrcSize, size_t ElementSize, std::function<Ref(size_t ElementSize, Ref Src)> Helper);
|
||||
void AVX128_VectorBinaryImpl(OpcodeArgs, size_t SrcSize, size_t ElementSize, std::function<Ref(size_t ElementSize, Ref Src1, Ref Src2)> Helper);
|
||||
void AVX128_VectorShiftWideImpl(OpcodeArgs, size_t ElementSize, IROps IROp);
|
||||
void AVX128_VectorShiftImmImpl(OpcodeArgs, size_t ElementSize, IROps IROp);
|
||||
void AVX128_VectorTrinaryImpl(OpcodeArgs, size_t SrcSize, size_t ElementSize, Ref Src3,
|
||||
std::function<Ref(size_t ElementSize, Ref Src1, Ref Src2, Ref Src3)> Helper);
|
||||
|
||||
enum class ShiftDirection { RIGHT, LEFT };
|
||||
void AVX128_ShiftDoubleImm(OpcodeArgs, ShiftDirection Dir);
|
||||
|
||||
void AVX128_VMOVAPS(OpcodeArgs);
|
||||
void AVX128_VMOVSD(OpcodeArgs);
|
||||
void AVX128_VMOVSS(OpcodeArgs);
|
||||
|
||||
template<IROps IROp, size_t ElementSize>
|
||||
void AVX128_VectorALU(OpcodeArgs);
|
||||
void AVX128_VectorXOR(OpcodeArgs);
|
||||
template<IROps IROp, size_t ElementSize>
|
||||
void AVX128_VectorUnary(OpcodeArgs);
|
||||
|
||||
void AVX128_VZERO(OpcodeArgs);
|
||||
void AVX128_MOVVectorNT(OpcodeArgs);
|
||||
void AVX128_MOVQ(OpcodeArgs);
|
||||
void AVX128_VMOVLP(OpcodeArgs);
|
||||
void AVX128_VMOVHP(OpcodeArgs);
|
||||
void AVX128_VMOVDDUP(OpcodeArgs);
|
||||
void AVX128_VMOVSLDUP(OpcodeArgs);
|
||||
void AVX128_VMOVSHDUP(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VBROADCAST(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPUNPCKL(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPUNPCKH(OpcodeArgs);
|
||||
void AVX128_MOVVectorUnaligned(OpcodeArgs);
|
||||
template<size_t DstElementSize>
|
||||
void AVX128_InsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool HostRoundingMode>
|
||||
void AVX128_CVTFPR_To_GPR(OpcodeArgs);
|
||||
void AVX128_VANDN(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPACKSS(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPACKUS(OpcodeArgs);
|
||||
Ref AVX128_PSIGNImpl(size_t ElementSize, Ref Src1, Ref Src2);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPSIGN(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_UCOMISx(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void AVX128_VectorScalarInsertALU(OpcodeArgs);
|
||||
Ref AVX128_VFCMPImpl(size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VFCMP(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_InsertScalarFCMP(OpcodeArgs);
|
||||
void AVX128_MOVBetweenGPR_FPR(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_PExtr(OpcodeArgs);
|
||||
template<size_t ElementSize, size_t DstElementSize, bool Signed>
|
||||
void AVX128_ExtendVectorElements(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_MOVMSK(OpcodeArgs);
|
||||
void AVX128_MOVMSKB(OpcodeArgs);
|
||||
void AVX128_PINSRImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
void AVX128_VPINSRB(OpcodeArgs);
|
||||
void AVX128_VPINSRW(OpcodeArgs);
|
||||
void AVX128_VPINSRDQ(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPSRA(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPSLL(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPSRL(OpcodeArgs);
|
||||
|
||||
void AVX128_VariableShiftImpl(OpcodeArgs, IROps IROp);
|
||||
void AVX128_VPSLLV(OpcodeArgs);
|
||||
void AVX128_VPSRAVD(OpcodeArgs);
|
||||
void AVX128_VPSRLV(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPSRLI(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPSLLI(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPSRAI(OpcodeArgs);
|
||||
|
||||
void AVX128_VPSRLDQ(OpcodeArgs);
|
||||
void AVX128_VPSLLDQ(OpcodeArgs);
|
||||
|
||||
void AVX128_VINSERT(OpcodeArgs);
|
||||
void AVX128_VINSERTPS(OpcodeArgs);
|
||||
|
||||
Ref AVX128_PHSUBImpl(Ref Src1, Ref Src2, size_t ElementSize);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPHSUB(OpcodeArgs);
|
||||
|
||||
void AVX128_VPHSUBSW(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VADDSUBP(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool Signed>
|
||||
void AVX128_VPMULL(OpcodeArgs);
|
||||
|
||||
void AVX128_VPMULHRSW(OpcodeArgs);
|
||||
|
||||
template<bool Signed>
|
||||
void AVX128_VPMULHW(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void AVX128_Vector_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void AVX128_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
|
||||
template<size_t SrcElementSize, bool Widen>
|
||||
void AVX128_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
|
||||
void AVX128_VEXTRACT128(OpcodeArgs);
|
||||
void AVX128_VAESImc(OpcodeArgs);
|
||||
void AVX128_VAESEnc(OpcodeArgs);
|
||||
void AVX128_VAESEncLast(OpcodeArgs);
|
||||
void AVX128_VAESDec(OpcodeArgs);
|
||||
void AVX128_VAESDecLast(OpcodeArgs);
|
||||
void AVX128_VAESKeyGenAssist(OpcodeArgs);
|
||||
|
||||
void AVX128_VPCMPESTRI(OpcodeArgs);
|
||||
void AVX128_VPCMPESTRM(OpcodeArgs);
|
||||
void AVX128_VPCMPISTRI(OpcodeArgs);
|
||||
void AVX128_VPCMPISTRM(OpcodeArgs);
|
||||
|
||||
void AVX128_PHMINPOSUW(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VectorRound(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_InsertScalarRound(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VDPP(OpcodeArgs);
|
||||
void AVX128_VPERMQ(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool Low>
|
||||
void AVX128_VPSHUF(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VSHUF(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPERMILImm(OpcodeArgs);
|
||||
|
||||
template<IROps IROp, size_t ElementSize>
|
||||
void AVX128_VHADDP(OpcodeArgs);
|
||||
|
||||
void AVX128_VPHADDSW(OpcodeArgs);
|
||||
|
||||
void AVX128_VPMADDUBSW(OpcodeArgs);
|
||||
void AVX128_VPMADDWD(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VBLEND(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VHSUBP(OpcodeArgs);
|
||||
|
||||
void AVX128_VPSHUFB(OpcodeArgs);
|
||||
void AVX128_VPSADBW(OpcodeArgs);
|
||||
|
||||
void AVX128_VMPSADBW(OpcodeArgs);
|
||||
void AVX128_VPALIGNR(OpcodeArgs);
|
||||
|
||||
void AVX128_VMASKMOVImpl(OpcodeArgs, size_t ElementSize, size_t DstSize, bool IsStore, const X86Tables::DecodedOperand& MaskOp,
|
||||
const X86Tables::DecodedOperand& DataOp);
|
||||
|
||||
template<bool IsStore>
|
||||
void AVX128_VPMASKMOV(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool IsStore>
|
||||
void AVX128_VMASKMOV(OpcodeArgs);
|
||||
|
||||
void AVX128_MASKMOV(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VectorVariableBlend(OpcodeArgs);
|
||||
|
||||
void AVX128_SaveAVXState(Ref MemBase);
|
||||
void AVX128_RestoreAVXState(Ref MemBase);
|
||||
void AVX128_DefaultAVXState();
|
||||
|
||||
void AVX128_VPERM2(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VTESTP(OpcodeArgs);
|
||||
void AVX128_PTest(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void AVX128_VPERMILReg(OpcodeArgs);
|
||||
|
||||
void AVX128_VPERMD(OpcodeArgs);
|
||||
|
||||
void AVX128_VPCLMULQDQ(OpcodeArgs);
|
||||
|
||||
void AVX128_VFMAImpl(OpcodeArgs, IROps IROp, bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
void AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
|
||||
template<bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void AVX128_VFMADD(OpcodeArgs);
|
||||
template<bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void AVX128_VFMSUB(OpcodeArgs);
|
||||
template<bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void AVX128_VFNMADD(OpcodeArgs);
|
||||
template<bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void AVX128_VFNMSUB(OpcodeArgs);
|
||||
|
||||
template<uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void AVX128_VFMADDSUB(OpcodeArgs);
|
||||
template<uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx>
|
||||
void AVX128_VFMSUBADD(OpcodeArgs);
|
||||
|
||||
template<size_t AddrElementSize>
|
||||
RefPair AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
|
||||
|
||||
template<size_t AddrElementSize>
|
||||
void AVX128_VPGATHER(OpcodeArgs);
|
||||
|
||||
void AVX128_VCVTPH2PS(OpcodeArgs);
|
||||
void AVX128_VCVTPS2PH(OpcodeArgs);
|
||||
|
||||
// End of AVX 128-bit implementation
|
||||
void InvalidOp(OpcodeArgs);
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, Ref Src);
|
||||
@@ -996,6 +1306,17 @@ protected:
|
||||
if (CTX->HostFeatures.SupportsAFP) {
|
||||
return;
|
||||
}
|
||||
break;
|
||||
|
||||
case OP_VLOADVECTORMASKED:
|
||||
case OP_VLOADVECTORGATHERMASKED:
|
||||
case OP_VSTOREVECTORMASKED:
|
||||
/* On ASIMD platforms, the emulation happens to preserve NZCV, unlike the
|
||||
* more optimal SVE implementation that clobbers.
|
||||
*/
|
||||
if (!CTX->HostFeatures.SupportsSVE128 && !CTX->HostFeatures.SupportsSVE256) {
|
||||
return;
|
||||
}
|
||||
|
||||
break;
|
||||
default: break;
|
||||
@@ -1042,11 +1363,17 @@ private:
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
using SaveStoreAVXStatePtr = void (OpDispatchBuilder::*)(Ref MemBase);
|
||||
using DefaultAVXStatePtr = void (OpDispatchBuilder::*)();
|
||||
SaveStoreAVXStatePtr SaveAVXStateFunc {&OpDispatchBuilder::SaveAVXState};
|
||||
SaveStoreAVXStatePtr RestoreAVXStateFunc {&OpDispatchBuilder::RestoreAVXState};
|
||||
DefaultAVXStatePtr DefaultAVXStateFunc {&OpDispatchBuilder::DefaultAVXState};
|
||||
|
||||
void ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
|
||||
|
||||
// Opcode helpers for generalizing behavior across VEX and non-VEX variants.
|
||||
|
||||
Ref ADDSUBPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
Ref ADDSUBPOpImpl(OpSize Size, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
void AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
@@ -1060,54 +1387,49 @@ private:
|
||||
|
||||
Ref CVTGPR_To_FPRImpl(OpcodeArgs, size_t DstElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
Ref DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, size_t ElementSize);
|
||||
Ref DPPOpImpl(size_t DstSize, Ref Src1, Ref Src2, uint8_t Mask, size_t ElementSize);
|
||||
|
||||
Ref VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
Ref ExtendVectorElementsImpl(OpcodeArgs, size_t ElementSize, size_t DstElementSize, bool Signed);
|
||||
|
||||
Ref HSUBPOpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref HSUBPOpImpl(OpSize Size, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
Ref InsertPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
Ref MPSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& ImmOp);
|
||||
|
||||
Ref PACKSSOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PACKUSOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
Ref MPSADBWOpImpl(size_t SrcSize, Ref Src1, Ref Src2, uint8_t Select);
|
||||
|
||||
Ref PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, bool IsAVX);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
|
||||
|
||||
Ref PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2);
|
||||
Ref PHADDSOpImpl(OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PHMINPOSUWOpImpl(OpcodeArgs);
|
||||
|
||||
Ref PHSUBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, size_t ElementSize);
|
||||
Ref PHSUBOpImpl(OpSize Size, Ref Src1, Ref Src2, size_t ElementSize);
|
||||
|
||||
Ref PHSUBSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref PHSUBSOpImpl(OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PINSROpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
Ref PMADDWDOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2);
|
||||
Ref PMADDWDOpImpl(size_t Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PMADDUBSWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref PMADDUBSWOpImpl(size_t Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PMULHRSWOpImpl(OpcodeArgs, Ref Src1, Ref Src2);
|
||||
Ref PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PMULLOpImpl(OpcodeArgs, size_t ElementSize, bool Signed, Ref Src1, Ref Src2);
|
||||
Ref PMULLOpImpl(OpSize Size, size_t ElementSize, bool Signed, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref PSADBWOpImpl(size_t Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PSHUFBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2);
|
||||
Ref GeneratePSHUFBMask(uint8_t SrcSize);
|
||||
Ref PSHUFBOpImpl(uint8_t SrcSize, Ref Src1, Ref Src2, Ref MaskVector);
|
||||
|
||||
Ref PSIGNImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
@@ -1119,8 +1441,7 @@ private:
|
||||
|
||||
Ref PSRLDOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec);
|
||||
|
||||
Ref SHUFOpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
Ref SHUFOpImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, Ref Src1, Ref Src2, uint8_t Shuffle);
|
||||
|
||||
void VMASKMOVOpImpl(OpcodeArgs, size_t ElementSize, size_t DataSize, bool IsStore, const X86Tables::DecodedOperand& MaskOp,
|
||||
const X86Tables::DecodedOperand& DataOp);
|
||||
@@ -1130,7 +1451,7 @@ private:
|
||||
|
||||
Ref VFCMPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
|
||||
void VTESTOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
void VTESTOpImpl(OpSize SrcSize, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
void VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
@@ -1161,10 +1482,9 @@ private:
|
||||
Ref InsertScalarRoundImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, uint64_t Mode, bool ZeroUpperBits);
|
||||
|
||||
Ref InsertScalarFCMPOpImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, uint8_t CompType, bool ZeroUpperBits);
|
||||
Ref InsertScalarFCMPOpImpl(OpSize Size, uint8_t OpDstSize, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType, bool ZeroUpperBits);
|
||||
|
||||
Ref VectorRoundImpl(OpcodeArgs, size_t ElementSize, Ref Src, uint64_t Mode);
|
||||
Ref VectorRoundImpl(OpSize Size, size_t ElementSize, Ref Src, uint64_t Mode);
|
||||
|
||||
Ref Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op);
|
||||
@@ -1210,6 +1530,17 @@ private:
|
||||
Ref LoadEffectiveAddress(AddressMode A, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, unsigned AccessSize);
|
||||
|
||||
bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
|
||||
// Literals are immediates as sources but memory addresses as destinations.
|
||||
return !(Load && Operand.IsLiteral()) && !Operand.IsGPR();
|
||||
}
|
||||
|
||||
bool IsNonTSOReg(MemoryAccessType Access, uint8_t Reg) {
|
||||
return Access == MemoryAccessType::DEFAULT && Reg == X86State::REG_RSP;
|
||||
}
|
||||
|
||||
AddressMode DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, MemoryAccessType AccessType, bool IsLoad);
|
||||
|
||||
Ref LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {});
|
||||
Ref LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
@@ -1240,7 +1571,7 @@ private:
|
||||
}
|
||||
|
||||
constexpr OpSize GetGuestVectorLength() const {
|
||||
return CTX->HostFeatures.SupportsAVX ? OpSize::i256Bit : OpSize::i128Bit;
|
||||
return (CTX->HostFeatures.SupportsSVE256 && CTX->HostFeatures.SupportsAVX) ? OpSize::i256Bit : OpSize::i128Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -1432,6 +1763,14 @@ private:
|
||||
|
||||
void ZeroPF_AF();
|
||||
|
||||
void InvalidateAF() {
|
||||
_InvalidateFlags((1u << X86State::RFLAG_AF_RAW_LOC));
|
||||
}
|
||||
|
||||
void InvalidatePF_AF() {
|
||||
_InvalidateFlags((1u << X86State::RFLAG_PF_RAW_LOC) | (1u << X86State::RFLAG_AF_RAW_LOC));
|
||||
}
|
||||
|
||||
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_SF_RAW_LOC: return Invert ? CondClassType {COND_PL} : CondClassType {COND_MI};
|
||||
@@ -1492,10 +1831,14 @@ private:
|
||||
return _AddShift(OpSize::i64Bit, X, _LoadDF(), ShiftType::LSL, Shift);
|
||||
}
|
||||
|
||||
// Set SSE comparison flags based on the result set by Arm FCMP. This converts
|
||||
// NZCV from the Arm representation to an eXternal representation that's
|
||||
// totally not a euphemism for x86 or anything, nuh-uh.
|
||||
void ConvertNZCVToSSE() {
|
||||
// Compares two floats and sets flags for a COMISS instruction
|
||||
void Comiss(size_t ElementSize, Ref Src1, Ref Src2, bool InvalidateAF = false) {
|
||||
// First, set flags according to Arm FCMP.
|
||||
HandleNZCVWrite();
|
||||
_FCmp(ElementSize, Src1, Src2);
|
||||
|
||||
// Now set COMISS flags by converts NZCV from the Arm representation to an
|
||||
// eXternal representation that's totally not a euphemism for x86, nuh-uh.
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
|
||||
@@ -1533,10 +1876,15 @@ private:
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Or(OpSize::i32Bit, Z, V));
|
||||
|
||||
// Note that we store PF inverted.
|
||||
// TODO: We could maybe optimize this xor out for non-flagm platforms with
|
||||
// bfi/bfxil?
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Xor(OpSize::i32Bit, V, _Constant(1)));
|
||||
}
|
||||
|
||||
if (!InvalidateAF) {
|
||||
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
|
||||
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
|
||||
// byte to zero will indeed zero AF as intended.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(0));
|
||||
}
|
||||
}
|
||||
|
||||
// Set x87 comparison flags based on the result set by Arm FCMP. Clobbers
|
||||
@@ -1640,6 +1988,14 @@ private:
|
||||
return Constant;
|
||||
}
|
||||
|
||||
Ref LoadUncachedZeroVector(uint8_t Size) {
|
||||
return _LoadNamedVectorConstant(Size, IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
}
|
||||
|
||||
Ref LoadZeroVector(uint8_t Size) {
|
||||
return LoadAndCacheNamedVectorConstant(Size, IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
}
|
||||
|
||||
// Reset the named vector constants cache array.
|
||||
// These are only cached per block.
|
||||
void ClearCachedNamedConstants() {
|
||||
@@ -2098,8 +2454,8 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, uint8_t Align = 1, bool ForceNonTSO = false) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !ForceNonTSO;
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, uint8_t Align = 1) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
@@ -2109,8 +2465,8 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, Ref Value, uint8_t Align = 1, bool ForceNonTSO = false) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !ForceNonTSO;
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, Ref Value, uint8_t Align = 1) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -65,7 +65,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
|
||||
// Do all the work without it.
|
||||
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(OpSize::i32Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
@@ -90,8 +90,6 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here to indicate function and constants");
|
||||
|
||||
using FnType = Ref (*)(OpDispatchBuilder&, Ref, Ref, Ref);
|
||||
|
||||
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
@@ -121,7 +119,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
f3,
|
||||
};
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Data.Literal.Value & 0b11;
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(32, k_array[Imm8]);
|
||||
|
||||
@@ -312,8 +310,7 @@ void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
Ref Result = _VAESEnc(16, Dest, Src, ZeroRegister);
|
||||
Ref Result = _VAESEnc(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -326,8 +323,7 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -335,8 +331,7 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
Ref Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
|
||||
Ref Result = _VAESEncLast(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -349,8 +344,7 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -358,8 +352,7 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
Ref Result = _VAESDec(16, Dest, Src, ZeroRegister);
|
||||
Ref Result = _VAESDec(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -372,8 +365,7 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, ZeroRegister);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -381,8 +373,7 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
Ref Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
|
||||
Ref Result = _VAESDecLast(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -395,20 +386,17 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t RCON = Op->Src[1].Data.Literal.Value;
|
||||
const uint64_t RCON = Op->Src[1].Literal();
|
||||
|
||||
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(16, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
return _VAESKeyGenAssist(Src, KeyGenSwizzle, ZeroRegister, RCON);
|
||||
return _VAESKeyGenAssist(Src, KeyGenSwizzle, LoadZeroVector(16), RCON);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
@@ -417,26 +405,22 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Data.Literal.Value);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
|
||||
|
||||
auto Res = _PCLMUL(16, Dest, Src, Selector);
|
||||
auto Res = _PCLMUL(16, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
|
||||
|
||||
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector);
|
||||
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
|
||||
@@ -459,65 +459,42 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, Ref Res, Ref High) {
|
||||
HandleNZCVWrite();
|
||||
InvalidatePF_AF();
|
||||
|
||||
// PF/AF/ZF/SF
|
||||
// Undefined
|
||||
{
|
||||
_InvalidateFlags(1 << X86State::RFLAG_PF_RAW_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
}
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// If the value can fit then the top bits will be zero
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
|
||||
_SubNZCV(OpSize::i64Bit, High, SignBit);
|
||||
|
||||
// CF/OF
|
||||
{
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// If the value can fit then the top bits will be zero
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
|
||||
_SubNZCV(OpSize::i64Bit, High, SignBit);
|
||||
|
||||
// If High = SignBit, then sets to nZcv. Else sets to nzCV. Since SF/ZF
|
||||
// undefined, this does what we need.
|
||||
auto Zero = _Constant(0);
|
||||
_CondAddNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
}
|
||||
// If High = SignBit, then sets to nZcv. Else sets to nzCV. Since SF/ZF
|
||||
// undefined, this does what we need.
|
||||
auto Zero = _Constant(0);
|
||||
_CondAddNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
HandleNZCVWrite();
|
||||
InvalidatePF_AF();
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
OpSize Size = IR::SizeToOpSize(GetOpSize(High));
|
||||
|
||||
// AF/SF/PF/ZF
|
||||
// Undefined
|
||||
{
|
||||
_InvalidateFlags(1 << X86State::RFLAG_PF_RAW_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
}
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// The result register will be all zero if it can't fit due to how multiplication behaves
|
||||
_SubNZCV(Size, High, Zero);
|
||||
|
||||
// CF/OF
|
||||
{
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// The result register will be all zero if it can't fit due to how multiplication behaves
|
||||
_SubNZCV(Size, High, Zero);
|
||||
|
||||
// If High = 0, then sets to nZcv. Else sets to nzCV. Since SF/ZF undefined,
|
||||
// this does what we need.
|
||||
_CondAddNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
}
|
||||
// If High = 0, then sets to nZcv. Else sets to nzCV. Since SF/ZF undefined,
|
||||
// this does what we need.
|
||||
_CondAddNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x3 /* nzCV */);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
InvalidateAF();
|
||||
|
||||
if (SrcSize >= 4) {
|
||||
HandleNZ00Write();
|
||||
CalculatePF(_AndWithFlags(IR::SizeToOpSize(SrcSize), Res, Res));
|
||||
} else {
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
CalculatePF(Res);
|
||||
}
|
||||
CalculatePF(Res);
|
||||
|
||||
// SF/ZF/CF/OF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
@@ -541,10 +518,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref U
|
||||
}
|
||||
|
||||
CalculatePF(UnmaskedRes);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
InvalidateAF();
|
||||
|
||||
// OF
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
@@ -571,10 +545,7 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
InvalidateAF();
|
||||
|
||||
// OF
|
||||
// Only defined when Shift is 1 else undefined. Only is set if the top bit was set to 1 when
|
||||
@@ -594,10 +565,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
InvalidateAF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
@@ -643,8 +611,7 @@ void OpDispatchBuilder::CalculateFlags_BEXTR(Ref Src) {
|
||||
// ZF is set properly. CF and OF are defined as being set to zero. SF, PF, and
|
||||
// AF are undefined.
|
||||
SetNZ_ZeroCV(GetOpSize(Src), Src);
|
||||
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
InvalidatePF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, Ref Result) {
|
||||
@@ -654,17 +621,14 @@ void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, Ref Result) {
|
||||
//
|
||||
// ZF/SF/OF set as usual.
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
InvalidatePF_AF();
|
||||
|
||||
auto CFOp = GetRFLAG(X86State::RFLAG_ZF_RAW_LOC, true /* Invert */);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
InvalidatePF_AF();
|
||||
|
||||
// CF set according to the Src
|
||||
auto Zero = _Constant(0);
|
||||
@@ -684,9 +648,7 @@ void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, Ref Result, Ref Src
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
InvalidatePF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_POPCOUNT(Ref Result) {
|
||||
@@ -698,9 +660,7 @@ void OpDispatchBuilder::CalculateFlags_POPCOUNT(Ref Result) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
InvalidatePF_AF();
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -745,6 +745,9 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
auto b = _LoadContextIndexed(arg, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
// Set C1 to Zero
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(b, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(a, arg, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -752,19 +755,20 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
Ref arg;
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
// Implicit arg
|
||||
auto offset = _Constant(Op->OP & 7);
|
||||
arg = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, offset), mask);
|
||||
Ref arg = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, offset), mask);
|
||||
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
// Write to ST[TOP]
|
||||
// Write to ST[i]
|
||||
_StoreContextIndexed(a, arg, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
// Set Tag for ST[i]
|
||||
SetX87ValidTag(arg, true);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
// if we are popping then we must first mark this location as empty
|
||||
SetX87ValidTag(top, false);
|
||||
|
||||
@@ -623,13 +623,7 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
} else {
|
||||
// Invalidate deferred flags early
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
_FCmp(8, a, b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToSSE();
|
||||
Comiss(8, a, b, true /* InvalidateAF */);
|
||||
}
|
||||
|
||||
if constexpr (poptwice) {
|
||||
|
||||
@@ -27,8 +27,8 @@ std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
|
||||
{OPD(1, 0b10, 0x11), 1, X86InstInfo{"VMOVSS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x11), 1, X86InstInfo{"VMOVSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(1, 0b00, 0x12), 1, X86InstInfo{"VMOVLPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x12), 1, X86InstInfo{"VMOVLPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b00, 0x12), 1, X86InstInfo{"VMOVLPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b01, 0x12), 1, X86InstInfo{"VMOVLPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(1, 0b10, 0x12), 1, X86InstInfo{"VMOVSLDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(1, 0b11, 0x12), 1, X86InstInfo{"VMOVDDUP", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
@@ -282,7 +282,7 @@ std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
|
||||
{OPD(2, 0b01, 0x0E), 1, X86InstInfo{"VTESTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x0F), 1, X86InstInfo{"VTESTPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x13), 1, X86InstInfo{"VCVTPH2PS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x13), 1, X86InstInfo{"VCVTPH2PS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x16), 1, X86InstInfo{"VPERMPS", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x17), 1, X86InstInfo{"VPTEST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
@@ -343,46 +343,46 @@ std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
|
||||
{OPD(2, 0b01, 0x8C), 1, X86InstInfo{"VPMASKMOV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x8E), 1, X86InstInfo{"VPMASKMOV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x90), 1, X86InstInfo{"VPGATHERDD/Q", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x91), 1, X86InstInfo{"VPGATHERQD/Q", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x92), 1, X86InstInfo{"VGATHERDPS/D", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x93), 1, X86InstInfo{"VGATHERQPS/D", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x90), 1, X86InstInfo{"VPGATHERDD/Q", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x91), 1, X86InstInfo{"VPGATHERQD/Q", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x92), 1, X86InstInfo{"VGATHERDPS/D", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x93), 1, X86InstInfo{"VGATHERQPS/D", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x96), 1, X86InstInfo{"VFMADDSUB132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x97), 1, X86InstInfo{"VFMSUBADD132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x96), 1, X86InstInfo{"VFMADDSUB132", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x97), 1, X86InstInfo{"VFMSUBADD132", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x98), 1, X86InstInfo{"VFMADD132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x99), 1, X86InstInfo{"VFMADD132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9A), 1, X86InstInfo{"VFMSUB132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9B), 1, X86InstInfo{"VFMSUB132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9C), 1, X86InstInfo{"VFNMADD132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9D), 1, X86InstInfo{"VFNMADD132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9E), 1, X86InstInfo{"VFNMSUB132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9F), 1, X86InstInfo{"VFNMSUB132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x98), 1, X86InstInfo{"VFMADD132", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x99), 1, X86InstInfo{"VFMADD132_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9A), 1, X86InstInfo{"VFMSUB132", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9B), 1, X86InstInfo{"VFMSUB132_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9C), 1, X86InstInfo{"VFNMADD132", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9D), 1, X86InstInfo{"VFNMADD132_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9E), 1, X86InstInfo{"VFNMSUB132", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x9F), 1, X86InstInfo{"VFNMSUB132_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0xA8), 1, X86InstInfo{"VFMADD213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xA9), 1, X86InstInfo{"VFMADD213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAA), 1, X86InstInfo{"VFMSUB213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAB), 1, X86InstInfo{"VFMSUB213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAC), 1, X86InstInfo{"VFNMADD213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAD), 1, X86InstInfo{"VFNMADD213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAE), 1, X86InstInfo{"VFNMSUB213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAF), 1, X86InstInfo{"VFNMSUB213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xA8), 1, X86InstInfo{"VFMADD213", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xA9), 1, X86InstInfo{"VFMADD213_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAA), 1, X86InstInfo{"VFMSUB213", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAB), 1, X86InstInfo{"VFMSUB213_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAC), 1, X86InstInfo{"VFNMADD213", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAD), 1, X86InstInfo{"VFNMADD213_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAE), 1, X86InstInfo{"VFNMSUB213", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xAF), 1, X86InstInfo{"VFNMSUB213_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0xB8), 1, X86InstInfo{"VFMADD231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xB9), 1, X86InstInfo{"VFMADD231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBA), 1, X86InstInfo{"VFMSUB231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBB), 1, X86InstInfo{"VFMSUB231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBC), 1, X86InstInfo{"VFNMADD231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBD), 1, X86InstInfo{"VFNMADD231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBE), 1, X86InstInfo{"VFNMSUB231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBF), 1, X86InstInfo{"VFNMSUB231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xB8), 1, X86InstInfo{"VFMADD231", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xB9), 1, X86InstInfo{"VFMADD231_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBA), 1, X86InstInfo{"VFMSUB231", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBB), 1, X86InstInfo{"VFMSUB231_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBC), 1, X86InstInfo{"VFNMADD231", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBD), 1, X86InstInfo{"VFNMADD231_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBE), 1, X86InstInfo{"VFNMSUB231", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xBF), 1, X86InstInfo{"VFNMSUB231_S", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0xA6), 1, X86InstInfo{"VFMADDSUB213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xA7), 1, X86InstInfo{"VFMSUBADD213", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xA6), 1, X86InstInfo{"VFMADDSUB213", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xA7), 1, X86InstInfo{"VFMSUBADD213", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0xB6), 1, X86InstInfo{"VFMADDSUB231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xB7), 1, X86InstInfo{"VFMSUBADD231", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xB6), 1, X86InstInfo{"VFMADDSUB231", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xB7), 1, X86InstInfo{"VFMSUBADD231", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0xDB), 1, X86InstInfo{"VAESIMC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xDC), 1, X86InstInfo{"VAESENC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -433,7 +433,7 @@ std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
|
||||
|
||||
{OPD(3, 0b01, 0x18), 1, X86InstInfo{"VINSERTF128", TYPE_INST, GenFlagsSameSize(SIZE_256BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x19), 1, X86InstInfo{"VEXTRACTF128", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_256BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x1D), 1, X86InstInfo{"VCVTPS2PH", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x1D), 1, X86InstInfo{"VCVTPS2PH", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x20), 1, X86InstInfo{"VPINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x21), 1, X86InstInfo{"VINSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
@@ -452,33 +452,33 @@ std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps = []() consteval {
|
||||
{OPD(3, 0b01, 0x4B), 1, X86InstInfo{"VBLENDVPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x4C), 1, X86InstInfo{"VPBLENDVB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x5C), 1, X86InstInfo{"VFMADDSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x5D), 1, X86InstInfo{"VFMADDSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x5E), 1, X86InstInfo{"VFMSUBADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x5F), 1, X86InstInfo{"VFMSUBADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x5C), 1, X86InstInfo{"VFMADDSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x5D), 1, X86InstInfo{"VFMADDSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x5E), 1, X86InstInfo{"VFMSUBADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x5F), 1, X86InstInfo{"VFMSUBADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
|
||||
{OPD(3, 0b01, 0x60), 1, X86InstInfo{"VPCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x61), 1, X86InstInfo{"VPCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x62), 1, X86InstInfo{"VPCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x63), 1, X86InstInfo{"VPCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x68), 1, X86InstInfo{"VFMADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x69), 1, X86InstInfo{"VFMADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x6A), 1, X86InstInfo{"VFMADDSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x6B), 1, X86InstInfo{"VFMADDSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x6C), 1, X86InstInfo{"VFMSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x6D), 1, X86InstInfo{"VFMSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x6E), 1, X86InstInfo{"VFMSUBSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x6F), 1, X86InstInfo{"VFMSUBSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x68), 1, X86InstInfo{"VFMADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x69), 1, X86InstInfo{"VFMADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x6A), 1, X86InstInfo{"VFMADDSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x6B), 1, X86InstInfo{"VFMADDSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x6C), 1, X86InstInfo{"VFMSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x6D), 1, X86InstInfo{"VFMSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x6E), 1, X86InstInfo{"VFMSUBSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x6F), 1, X86InstInfo{"VFMSUBSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
|
||||
{OPD(3, 0b01, 0x78), 1, X86InstInfo{"VFNMADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x79), 1, X86InstInfo{"VFNMADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x7A), 1, X86InstInfo{"VFNMADDSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x7B), 1, X86InstInfo{"VFNMADDSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x7C), 1, X86InstInfo{"VFNMSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x7D), 1, X86InstInfo{"VFNMSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x7E), 1, X86InstInfo{"VFNMSUBSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x7F), 1, X86InstInfo{"VFNMSUBSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x78), 1, X86InstInfo{"VFNMADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x79), 1, X86InstInfo{"VFNMADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x7A), 1, X86InstInfo{"VFNMADDSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x7B), 1, X86InstInfo{"VFNMADDSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x7C), 1, X86InstInfo{"VFNMSUBPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x7D), 1, X86InstInfo{"VFNMSUBPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x7E), 1, X86InstInfo{"VFNMSUBSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
{OPD(3, 0b01, 0x7F), 1, X86InstInfo{"VFNMSUBSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, ///< FMA4
|
||||
|
||||
{OPD(3, 0b01, 0xDF), 1, X86InstInfo{"VAESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ constexpr uint32_t FLAG_LOCK = (1 << 2);
|
||||
constexpr uint32_t FLAG_LEGACY_PREFIX = (1 << 3);
|
||||
constexpr uint32_t FLAG_REX_PREFIX = (1 << 4);
|
||||
constexpr uint32_t FLAG_VSIB_BYTE = (1 << 5);
|
||||
// Hole where 1 << 6 is
|
||||
constexpr uint32_t FLAG_OPTION_AVX_W = (1 << 6);
|
||||
constexpr uint32_t FLAG_REX_WIDENING = (1 << 7);
|
||||
constexpr uint32_t FLAG_REX_XGPR_B = (1 << 8);
|
||||
constexpr uint32_t FLAG_REX_XGPR_X = (1 << 9);
|
||||
@@ -137,6 +137,10 @@ struct DecodedOperand {
|
||||
bool IsSIB() const {
|
||||
return Type == OpType::SIB;
|
||||
}
|
||||
uint64_t Literal() const {
|
||||
LOGMAN_THROW_A_FMT(IsLiteral(), "Precondition: must be a literal");
|
||||
return Data.Literal.Value;
|
||||
}
|
||||
|
||||
union TypeUnion {
|
||||
struct GPRType {
|
||||
|
||||
@@ -231,6 +231,17 @@
|
||||
],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = PushRoundingMode u8:$RoundMode": {
|
||||
"Desc": ["Override the current rounding mode options for the thread, returning old FPCR"
|
||||
],
|
||||
"DestSize": "8",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"PopRoundingMode GPR:$FPCR": {
|
||||
"Desc": ["Resets rounding mode after PushRoundingMode operation"
|
||||
],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"Print SSA:$Value": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Debug operation that prints an SSA value to the console",
|
||||
@@ -545,6 +556,20 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VLoadVectorGatherMasked u8:#RegisterSize, u8:#ElementSize, FPR:$Incoming, FPR:$Mask, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, u8:$VectorIndexElementSize, u8:$OffsetScale, u8:$DataElementOffsetStart, u8:$IndexElementOffsetStart": {
|
||||
"Desc": [
|
||||
"Does a masked load similar to VPGATHERD* where the upper bit of each element",
|
||||
"determines whether or not that element will be loaded from memory.",
|
||||
"Most of VSIB encoding is passed directly through to the IR operation."
|
||||
],
|
||||
"TiedSource": 0,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"EmitValidation": [
|
||||
"$VectorIndexElementSize == OpSize::i32Bit || $VectorIndexElementSize == OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"FPR = VLoadVectorElement u8:#RegisterSize, u8:#ElementSize, FPR:$DstSrc, u8:$Index, GPR:$Addr": {
|
||||
"Desc": ["Does a memory load to a single element of a vector.",
|
||||
"Leaves the rest of the vector's data intact.",
|
||||
@@ -629,6 +654,30 @@
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
"VStoreNonTemporal u8:#RegisterSize, FPR:$Value, GPR:$Addr, i8:$Offset": {
|
||||
"Desc": ["Does a non-temporal memory store of a vector.",
|
||||
"Matches arm64 SVE stnt1b semantics.",
|
||||
"Specifically weak-memory model ordered to match x86 non-temporal stores."
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"EmitValidation": [
|
||||
"_Offset % RegisterSize == 0",
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
]
|
||||
},
|
||||
"VStoreNonTemporalPair u8:#RegisterSize, FPR:$ValueLow, FPR:$ValueHigh, GPR:$Addr, i8:$Offset": {
|
||||
"Desc": ["Does a non-temporal memory store of two vector registers.",
|
||||
"Matches arm64 stnp semantics.",
|
||||
"Specifically weak-memory model ordered to match x86 non-temporal stores."
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"EmitValidation": [
|
||||
"_Offset % RegisterSize == 0",
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Atomic": {
|
||||
@@ -1917,7 +1966,7 @@
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VBic u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"FPR = VAndn u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -2262,6 +2311,42 @@
|
||||
"FPR = VFCADD u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, u16:$Rotate": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMLA u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (Vector1 * Vector2) + Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
},
|
||||
"FPR = VFMLS u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (Vector1 * Vector2) - Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
},
|
||||
"FPR = VFNMLA u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (-Vector1 * Vector2) + Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
},
|
||||
"FPR = VFNMLS u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FPR:$Addend": {
|
||||
"Desc": [
|
||||
"Dest = (-Vector1 * Vector2) - Addend",
|
||||
"This explicitly matches x86 FMA semantics because ARM semantics are mind-bending."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"TiedSource": 2
|
||||
}
|
||||
},
|
||||
"Conv": {
|
||||
@@ -2314,6 +2399,32 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DestElementSize"
|
||||
},
|
||||
|
||||
"FPR = VFCVTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": [
|
||||
"Vector op: Converts float from source element size to destination size (fp32->fp64)",
|
||||
"Selecting from the high half of the register."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)",
|
||||
"EmitValidation": [
|
||||
"RegisterSize != FEXCore::IR::OpSize::i256Bit && \"What does 256-bit mean in this context?\""
|
||||
]
|
||||
},
|
||||
"FPR = VFCVTN2 u8:#RegisterSize, u8:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"TiedSource": 0,
|
||||
"Desc": [
|
||||
"Vector op: Converts float from source element size and inserting in to the high bits.",
|
||||
"Bottom half is untouched",
|
||||
"Narrowing to the element size below what is passed in.",
|
||||
"F64->F32, F32->F16"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)",
|
||||
"EmitValidation": [
|
||||
"RegisterSize != FEXCore::IR::OpSize::i256Bit && \"What does 256-bit mean in this context?\""
|
||||
]
|
||||
},
|
||||
"FPR = Vector_FToI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, RoundType:$Round": {
|
||||
"Desc": ["Vector op: Rounds float to integral",
|
||||
"Rounding mode determined by argument"
|
||||
|
||||
@@ -183,6 +183,14 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
return "addsubpd_invert";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPD_INVERT_UPPER:
|
||||
return "addsubpd_invert_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PSUBADDPS_INVERT:
|
||||
return "subaddps_invert";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PSUBADDPS_INVERT_UPPER:
|
||||
return "subaddps_invert_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PSUBADDPD_INVERT:
|
||||
return "subaddpd_invert";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PSUBADDPD_INVERT_UPPER:
|
||||
return "subaddpd_invert_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_MOVMSKPS_SHIFT:
|
||||
return "movmskps_shift";
|
||||
case NamedVectorConstant::NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE:
|
||||
|
||||
@@ -70,7 +70,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateContextLoadStoreElimination(ctx->HostFeatures.SupportsAVX));
|
||||
InsertPass(CreateContextLoadStoreElimination(ctx->HostFeatures.SupportsAVX && ctx->HostFeatures.SupportsSVE256));
|
||||
InsertPass(CreateDeadStoreElimination());
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
|
||||
@@ -17,7 +17,7 @@ class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsSVE256);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
|
||||
@@ -298,8 +298,12 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 >> (Constant2 & ShiftMask)) & getMask(IROp);
|
||||
// The source is masked, which will produce a correctly masked
|
||||
// destination. Masking the destination without the source instead will
|
||||
// right-shift garbage into the upper bits instead of zeroes.
|
||||
Constant1 &= getMask(IROp);
|
||||
Constant2 &= (IROp->Size == 8 ? 63 : 31);
|
||||
uint64_t NewConstant = (Constant1 >> Constant2);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
@@ -85,9 +85,39 @@ struct ContextInfo {
|
||||
fextl::vector<ContextMemberInfo> ClassificationInfo;
|
||||
};
|
||||
|
||||
static void ClassifyContextStruct(ContextInfo* ContextClassificationInfo, bool SupportsAVX) {
|
||||
static void ClassifyContextStruct(ContextInfo* ContextClassificationInfo, bool SupportsAVX256) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader),
|
||||
sizeof(FEXCore::Core::CPUState::InlineJITBlockHeader),
|
||||
},
|
||||
LastAccessType::INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// DeferredSignalRefCount
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount),
|
||||
sizeof(FEXCore::Core::CPUState::DeferredSignalRefCount),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, avx_high[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
@@ -108,6 +138,50 @@ static void ClassifyContextStruct(ContextInfo* ContextClassificationInfo, bool S
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad),
|
||||
sizeof(FEXCore::Core::CPUState::_pad),
|
||||
},
|
||||
LastAccessType::INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
static_assert(offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) == 416, "What");
|
||||
static_assert(FEXCore::Core::CPUState::XMM_AVX_REG_SIZE == 32, "What");
|
||||
if (SupportsAVX256) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) + FEXCore::Core::CPUState::XMM_AVX_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_AVX_REG_SIZE,
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.pad[0][0]),
|
||||
static_cast<uint16_t>(FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * FEXCore::Core::CPUState::NUM_XMMS),
|
||||
},
|
||||
LastAccessType::INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es_idx),
|
||||
@@ -164,8 +238,8 @@ static void ClassifyContextStruct(ContextInfo* ContextClassificationInfo, bool S
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad),
|
||||
sizeof(FEXCore::Core::CPUState::_pad),
|
||||
offsetof(FEXCore::Core::CPUState, _pad2),
|
||||
sizeof(FEXCore::Core::CPUState::_pad2),
|
||||
},
|
||||
LastAccessType::INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
@@ -225,48 +299,6 @@ static void ClassifyContextStruct(ContextInfo* ContextClassificationInfo, bool S
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader),
|
||||
sizeof(FEXCore::Core::CPUState::InlineJITBlockHeader),
|
||||
},
|
||||
LastAccessType::INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
if (SupportsAVX) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) + FEXCore::Core::CPUState::XMM_AVX_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_AVX_REG_SIZE,
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.pad[0][0]),
|
||||
static_cast<uint16_t>(FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * FEXCore::Core::CPUState::NUM_XMMS),
|
||||
},
|
||||
LastAccessType::INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
@@ -337,21 +369,11 @@ static void ClassifyContextStruct(ContextInfo* ContextClassificationInfo, bool S
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// _pad2
|
||||
// _pad3
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad2),
|
||||
sizeof(FEXCore::Core::CPUState::_pad2),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// DeferredSignalRefCount
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount),
|
||||
sizeof(FEXCore::Core::CPUState::DeferredSignalRefCount),
|
||||
offsetof(FEXCore::Core::CPUState, _pad3),
|
||||
sizeof(FEXCore::Core::CPUState::_pad3),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
@@ -376,7 +398,7 @@ static void ClassifyContextStruct(ContextInfo* ContextClassificationInfo, bool S
|
||||
ContextClassificationInfo->Lookup.size(), sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
static void ResetClassificationAccesses(ContextInfo* ContextClassificationInfo, bool SupportsAVX) {
|
||||
static void ResetClassificationAccesses(ContextInfo* ContextClassificationInfo, bool SupportsAVX256) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
auto SetAccess = [&](size_t Offset, LastAccessType Access) {
|
||||
@@ -385,12 +407,41 @@ static void ResetClassificationAccesses(ContextInfo* ContextClassificationInfo,
|
||||
ContextClassification->at(Offset).AccessOffset = 0;
|
||||
ContextClassification->at(Offset).StoreNode = nullptr;
|
||||
};
|
||||
|
||||
size_t Offset = 0;
|
||||
|
||||
///< InlineJITBlockHeader
|
||||
SetAccess(Offset++, LastAccessType::INVALID);
|
||||
|
||||
// DeferredSignalRefCount
|
||||
SetAccess(Offset++, LastAccessType::INVALID);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
///< avx_high
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
// rip
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
///< gregs
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
// pad
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
// xmm
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
// xmm_pad
|
||||
if (!SupportsAVX256) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
// Segment indexes
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
@@ -399,7 +450,7 @@ static void ResetClassificationAccesses(ContextInfo* ContextClassificationInfo,
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
// Pad
|
||||
// Pad2
|
||||
SetAccess(Offset++, LastAccessType::INVALID);
|
||||
|
||||
// Segments
|
||||
@@ -410,37 +461,34 @@ static void ResetClassificationAccesses(ContextInfo* ContextClassificationInfo,
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
// Pad2
|
||||
SetAccess(Offset++, LastAccessType::INVALID);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
if (!SupportsAVX) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
///< flags
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
// PF/AF
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
///< pf_raw
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
///< af_raw
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
///< mm
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
///< gdt
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
///< FCW
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
SetAccess(Offset++, LastAccessType::INVALID);
|
||||
///< AbridgedFTW
|
||||
SetAccess(Offset++, LastAccessType::NONE);
|
||||
|
||||
// pad3
|
||||
SetAccess(Offset++, LastAccessType::INVALID);
|
||||
}
|
||||
|
||||
@@ -453,16 +501,16 @@ struct BlockInfo {
|
||||
|
||||
class RCLSE final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit RCLSE(bool SupportsAVX_)
|
||||
: SupportsAVX {SupportsAVX_} {
|
||||
ClassifyContextStruct(&ClassifiedStruct, SupportsAVX);
|
||||
explicit RCLSE(bool SupportsAVX256)
|
||||
: SupportsAVX256 {SupportsAVX256} {
|
||||
ClassifyContextStruct(&ClassifiedStruct, SupportsAVX256);
|
||||
}
|
||||
void Run(FEXCore::IR::IREmitter* IREmit) override;
|
||||
private:
|
||||
ContextInfo ClassifiedStruct;
|
||||
fextl::unordered_map<FEXCore::IR::NodeID, BlockInfo> OffsetToBlockMap;
|
||||
|
||||
bool SupportsAVX;
|
||||
bool SupportsAVX256;
|
||||
|
||||
ContextMemberInfo* FindMemberInfo(ContextInfo* ClassifiedInfo, uint32_t Offset, uint8_t Size);
|
||||
ContextMemberInfo* RecordAccess(ContextMemberInfo* Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size,
|
||||
@@ -628,7 +676,7 @@ void RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit) {
|
||||
auto BlockOp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
auto BlockEnd = IREmit->GetIterator(BlockOp->Last);
|
||||
|
||||
ResetClassificationAccesses(&LocalInfo, SupportsAVX);
|
||||
ResetClassificationAccesses(&LocalInfo, SupportsAVX256);
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_STORECONTEXT) {
|
||||
@@ -696,11 +744,11 @@ void RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit) {
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH) != FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH) {
|
||||
// We can't track through these
|
||||
ResetClassificationAccesses(&LocalInfo, SupportsAVX);
|
||||
ResetClassificationAccesses(&LocalInfo, SupportsAVX256);
|
||||
}
|
||||
} else if (IROp->Op == OP_STORECONTEXTINDEXED || IROp->Op == OP_LOADCONTEXTINDEXED || IROp->Op == OP_BREAK) {
|
||||
// We can't track through these
|
||||
ResetClassificationAccesses(&LocalInfo, SupportsAVX);
|
||||
ResetClassificationAccesses(&LocalInfo, SupportsAVX256);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -717,8 +765,8 @@ void RCLSE::Run(FEXCore::IR::IREmitter* IREmit) {
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX) {
|
||||
return fextl::make_unique<RCLSE>(SupportsAVX);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX256) {
|
||||
return fextl::make_unique<RCLSE>(SupportsAVX256);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -135,12 +135,16 @@ private:
|
||||
// Maps Old defs to their assigned spill slot + 1, or 0 if not spilled.
|
||||
fextl::vector<unsigned> SpillSlots;
|
||||
|
||||
bool Rematerializable(IROp_Header* IROp) {
|
||||
return IROp->Op == OP_CONSTANT;
|
||||
}
|
||||
|
||||
Ref InsertFill(Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Precondition");
|
||||
IROp_Header* IROp = IR->GetOp<IROp_Header>(Old);
|
||||
|
||||
// Remat if we can
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
if (Rematerializable(IROp)) {
|
||||
uint64_t Const = IROp->C<IR::IROp_Constant>()->Constant;
|
||||
return IREmit->_Constant(Const);
|
||||
}
|
||||
@@ -265,37 +269,44 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
// Helper macro to walk the set bits b in a 32-bit word x, using ffs to get
|
||||
// the next set bit and then clearing on each iteration.
|
||||
#define foreach_bit(b, x) for (uint32_t __x = (x), b; ((b) = __builtin_ffs(__x) - 1, __x); __x &= ~(1 << (b)))
|
||||
|
||||
void SpillReg(RegisterClass* Class, IROp_Header* Exclude, bool Pair) {
|
||||
// Find the best node to spill according to the "furthest-first" heuristic.
|
||||
// Since we defined IPs relative to the end of the block, the furthest
|
||||
// next-use has the /smallest/ unsigned IP.
|
||||
//
|
||||
// TODO: Prioritize constants, as they are cheaper to rematerialize.
|
||||
Ref Candidate = nullptr;
|
||||
uint32_t BestDistance = UINT32_MAX;
|
||||
uint8_t BestReg = ~0;
|
||||
|
||||
for (int i = 0; i < Class->Count; ++i) {
|
||||
foreach_bit(i, Class->Allocated) {
|
||||
// We have to prioritize the pair region if we're allocating for a Pair.
|
||||
// See the comment at the call site in AssignReg.
|
||||
if (Pair && Candidate != nullptr && i >= PairRegs) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (Class->Allocated & (1u << i)) {
|
||||
Ref Old = Class->RegToSSA[i];
|
||||
Ref Old = Class->RegToSSA[i];
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Old != nullptr, "Invariant3");
|
||||
LOGMAN_THROW_A_FMT(SSAToReg[IR->GetID(Map(Old)).Value].Reg == i, "Invariant4");
|
||||
LOGMAN_THROW_AA_FMT(Old != nullptr, "Invariant3");
|
||||
LOGMAN_THROW_A_FMT(SSAToReg[IR->GetID(Map(Old)).Value].Reg == i, "Invariant4");
|
||||
|
||||
// Skip any source used by the current instruction, it is unspillable.
|
||||
if (!HasSource(Exclude, Old)) {
|
||||
uint32_t NextUse = NextUses[IR->GetID(Old).Value];
|
||||
if (NextUse < BestDistance) {
|
||||
BestDistance = NextUse;
|
||||
BestReg = i;
|
||||
Candidate = Old;
|
||||
}
|
||||
// Skip any source used by the current instruction, it is unspillable.
|
||||
if (!HasSource(Exclude, Old)) {
|
||||
uint32_t NextUse = NextUses[IR->GetID(Old).Value];
|
||||
|
||||
// Prioritize remat over spilling. It is typically cheaper to remat a
|
||||
// constant multiple times than to spill a single value.
|
||||
if (!Rematerializable(IR->GetOp<IROp_Header>(Old))) {
|
||||
NextUse += 100000;
|
||||
}
|
||||
|
||||
if (NextUse < BestDistance) {
|
||||
BestDistance = NextUse;
|
||||
BestReg = i;
|
||||
Candidate = Old;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -81,7 +81,7 @@ ssize_t LoadFileToBuffer(const fextl::string& Filepath, std::span<char> Buffer)
|
||||
#else
|
||||
template<typename T>
|
||||
static bool LoadFileImpl(T& Data, const fextl::string& Filepath, size_t FixedSize) {
|
||||
std::ifstream f(Filepath, std::ios::binary | std::ios::ate);
|
||||
std::ifstream f(Filepath.c_str(), std::ios::binary | std::ios::ate);
|
||||
if (f.fail()) {
|
||||
return false;
|
||||
}
|
||||
@@ -93,7 +93,7 @@ static bool LoadFileImpl(T& Data, const fextl::string& Filepath, size_t FixedSiz
|
||||
}
|
||||
|
||||
ssize_t LoadFileToBuffer(const fextl::string& Filepath, std::span<char> Buffer) {
|
||||
std::ifstream f(Filepath, std::ios::binary | std::ios::ate);
|
||||
std::ifstream f(Filepath.c_str(), std::ios::binary | std::ios::ate);
|
||||
return f.readsome(Buffer.data(), Buffer.size());
|
||||
}
|
||||
|
||||
|
||||
@@ -47,6 +47,8 @@ uint64_t SetSignalMask(uint64_t Mask) {
|
||||
#ifndef _WIN32
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &Mask, 8);
|
||||
return Mask;
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -1,36 +0,0 @@
|
||||
# FEXCore IR Optimization passes
|
||||
---
|
||||
**This is very much a WIP since these optimization passes aren't in code yet**
|
||||
## Pass Managers
|
||||
* Need Function level optimization pass manager
|
||||
* Need block level optimization pass manager
|
||||
### Dead Store Elimination
|
||||
We need to do dead store elimination because LLVM can't always handle elimination of our loadstores
|
||||
This is very apparent when we are doing flag calculations and LLVM isn't able to remove them
|
||||
This is mainly just an issue around the context loadstores.
|
||||
We will want this more when the IRJIT comes online.
|
||||
### Dead flag elimination
|
||||
X86-64 is a fairly disguting ISA in that it calculates a bunch of flags on almost all instructions.
|
||||
We need eliminate redundant flag calculations that end up being being overwritten without being used.
|
||||
This happens *constantly* and in most cases the flag calculation takes significantly more work than the basic op by itself
|
||||
Good chance that breaking out the flags to independent memory locations will make this easier. Or just adding ops for flag handling.
|
||||
### Dead Code Elimination
|
||||
There are a lot of cases that code will be generated that is immediately dead afterwards.
|
||||
Flag calculation elimination will produce a lot of dead code that needs to get removed.
|
||||
Additionally there are a decent amount of x86-64 instructions that store their results in to multiple registers and then the next instruction overwrites one of those instructions.
|
||||
Multiply and Divide being a big one, since x86 calculates these at higher precision.
|
||||
These can rely significantly tracking liveness between LoadContext and StoreContext ops
|
||||
### ABI register elimination pass
|
||||
This one is very fun and will reduce a decent amount of work that the JIT needs to do.
|
||||
When we are targeting a specific x86-64 ABI and we know that we have translated a block of code that is the entire function.
|
||||
We can eliminate stores to the context that by ABI standards is a temporary register.
|
||||
We will be able to know exactly that these are dead and just remove the store (and run all the passes that optimize the rest away afterwards).
|
||||
### Loadstore coalescing pass
|
||||
Large amount of x86-64 instructions load or store registers in order from the context.
|
||||
We can merge these in to loadstore pair ops to improve perf
|
||||
### Function level heuristic pass
|
||||
Once we know that a function is a true full recompile we can do some additional optimizations.
|
||||
Remove any final flag stores. We know that a compiler won't pass flags past a function call boundary(It doesn't exist in the ABI)
|
||||
Remove any loadstores to the context mid function, only do a final store at the end of the function and do loads at the start. Which means ops just map registers directly throughout the entire function.
|
||||
### SIMD coalescing pass?
|
||||
When operating on older MMX ops(64bit SIMD) and they may end up up generating some independent ops that can be coalesced in to a 128bit op
|
||||
@@ -194,6 +194,11 @@ public:
|
||||
///< Sets FEX's internal EFLAGS representation to the passed in compacted form.
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, uint32_t EFLAGS) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void
|
||||
ReconstructXMMRegisters(const FEXCore::Core::InternalThreadState* Thread, __uint128_t* XMM_Low, __uint128_t* YMM_High) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void
|
||||
SetXMMRegistersFromState(FEXCore::Core::InternalThreadState* Thread, const __uint128_t* XMM_Low, const __uint128_t* YMM_High) = 0;
|
||||
|
||||
/**
|
||||
* @brief Create a new thread object that doesn't inherit any state.
|
||||
* Used to create FEX thread objects in preparation for creating a true OS thread.
|
||||
|
||||
@@ -88,19 +88,28 @@ struct CPUState {
|
||||
SSE sse;
|
||||
};
|
||||
|
||||
uint64_t InlineJITBlockHeader {};
|
||||
// Reference counter for FEX's per-thread deferred signals.
|
||||
// Counts the nesting depth of program sections that cause signals to be deferred.
|
||||
NonAtomicRefCounter<uint64_t> DeferredSignalRefCount;
|
||||
|
||||
// The high 128-bits of AVX registers when not being emulated by SVE256.
|
||||
uint64_t avx_high[16][2];
|
||||
|
||||
uint64_t rip {}; ///< Current core's RIP. May not be entirely accurate while JIT is active
|
||||
uint64_t gregs[16] {};
|
||||
uint64_t _pad {};
|
||||
XMMRegs xmm {};
|
||||
|
||||
// Raw segment register indexes
|
||||
uint16_t es_idx {}, cs_idx {}, ss_idx {}, ds_idx {};
|
||||
uint16_t gs_idx {}, fs_idx {};
|
||||
uint16_t _pad[2];
|
||||
uint16_t _pad2[2];
|
||||
|
||||
// Segment registers holding base addresses
|
||||
uint32_t es_cached {}, cs_cached {}, ss_cached {}, ds_cached {};
|
||||
uint64_t gs_cached {};
|
||||
uint64_t fs_cached {};
|
||||
uint64_t InlineJITBlockHeader {};
|
||||
XMMRegs xmm {};
|
||||
uint8_t flags[48] {};
|
||||
uint64_t pf_raw {};
|
||||
uint64_t af_raw {};
|
||||
@@ -113,11 +122,7 @@ struct CPUState {
|
||||
uint16_t FCW {0x37F};
|
||||
uint8_t AbridgedFTW {};
|
||||
|
||||
uint8_t _pad2[5];
|
||||
// Reference counter for FEX's per-thread deferred signals.
|
||||
// Counts the nesting depth of program sections that cause signals to be deferred.
|
||||
NonAtomicRefCounter<uint64_t> DeferredSignalRefCount;
|
||||
|
||||
uint8_t _pad3[5];
|
||||
// PF/AF are statically mapped as-if they were r16/r17 (which do not exist in
|
||||
// x86 otherwise). This allows a straightforward mapping for SRA.
|
||||
static constexpr uint8_t PF_AS_GREG = 16;
|
||||
@@ -161,8 +166,10 @@ struct CPUState {
|
||||
};
|
||||
static_assert(std::is_trivially_copyable_v<CPUState>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<CPUState>, "This needs to be standard layout");
|
||||
static_assert(offsetof(CPUState, avx_high) % 16 == 0, "avx_high needs to be 128-bit aligned!");
|
||||
static_assert(offsetof(CPUState, xmm) % 32 == 0, "xmm needs to be 256-bit aligned!");
|
||||
static_assert(offsetof(CPUState, mm) % 16 == 0, "mm needs to be 128-bit aligned!");
|
||||
static_assert(offsetof(CPUState, gregs[15]) <= 504, "gregs maximum offset must be <= 504 for ldp/stp to work");
|
||||
static_assert(offsetof(CPUState, DeferredSignalRefCount) % 8 == 0, "Needs to be 8-byte aligned");
|
||||
|
||||
struct InternalThreadState;
|
||||
@@ -342,7 +349,6 @@ struct CpuStateFrame {
|
||||
JITPointers Pointers;
|
||||
};
|
||||
static_assert(offsetof(CpuStateFrame, State) == 0, "CPUState must be first member in CpuStateFrame");
|
||||
static_assert(offsetof(CpuStateFrame, State.rip) == 0, "rip must be zero offset in CpuStateFrame");
|
||||
static_assert(offsetof(CpuStateFrame, Pointers) % 8 == 0, "JITPointers need to be aligned to 8 bytes");
|
||||
static_assert(offsetof(CpuStateFrame, Pointers) + sizeof(CpuStateFrame::Pointers) <= 32760, "JITPointers maximum pointer needs to be less "
|
||||
"than architecture maximum 32768");
|
||||
|
||||
@@ -25,8 +25,8 @@ public:
|
||||
bool Supports3DNow {};
|
||||
bool SupportsSSE4A {};
|
||||
bool SupportsAVX {};
|
||||
bool SupportsAVX2 {};
|
||||
bool SupportsSVE {};
|
||||
bool SupportsSVE128 {};
|
||||
bool SupportsSVE256 {};
|
||||
bool SupportsSHA {};
|
||||
bool SupportsBMI1 {};
|
||||
bool SupportsBMI2 {};
|
||||
@@ -38,6 +38,7 @@ public:
|
||||
bool SupportsFlagM2 {};
|
||||
bool SupportsRPRES {};
|
||||
bool SupportsPreserveAllABI {};
|
||||
bool SupportsAES256 {};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
|
||||
@@ -49,6 +49,10 @@ enum NamedVectorConstant : uint8_t {
|
||||
NAMED_VECTOR_PADDSUBPS_INVERT_UPPER,
|
||||
NAMED_VECTOR_PADDSUBPD_INVERT,
|
||||
NAMED_VECTOR_PADDSUBPD_INVERT_UPPER,
|
||||
NAMED_VECTOR_PSUBADDPS_INVERT,
|
||||
NAMED_VECTOR_PSUBADDPS_INVERT_UPPER,
|
||||
NAMED_VECTOR_PSUBADDPD_INVERT,
|
||||
NAMED_VECTOR_PSUBADDPD_INVERT_UPPER,
|
||||
NAMED_VECTOR_MOVMSKPS_SHIFT,
|
||||
NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE,
|
||||
NAMED_VECTOR_BLENDPS_0110B,
|
||||
|
||||
@@ -10,12 +10,9 @@ foreach(TEST ${TESTS})
|
||||
catch_discover_tests(FEXCore_Tests_${TEST_NAME} TEST_SUFFIX ".${TEST_NAME}.FEXCore_Tests")
|
||||
endforeach()
|
||||
|
||||
execute_process(COMMAND "nproc" OUTPUT_VARIABLE CORES)
|
||||
string(STRIP ${CORES} CORES)
|
||||
|
||||
add_custom_target(
|
||||
fexcore_apitests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}/"
|
||||
USES_TERMINAL
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" "-j${CORES}" "-R" "\.*.FEXCore_Tests$$")
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" ${TEST_JOB_FLAG} "-R" "\.*.FEXCore_Tests$$")
|
||||
|
||||
@@ -327,42 +327,25 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Advanced SIMD two-register m
|
||||
TEST_SINGLE(fsqrt<SubRegSize::i16Bit>(QReg::q30, QReg::q29), "fsqrt v30.8h, v29.8h");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Advanced SIMD three-register extension") {
|
||||
TEST_SINGLE(sdot(SubRegSize::i8Bit, QReg::q30, QReg::q29, QReg::q28), "sdot v30.16b, v29.16b, v28.16b");
|
||||
TEST_SINGLE(sdot(SubRegSize::i16Bit, QReg::q30, QReg::q29, QReg::q28), "sdot v30.8h, v29.16b, v28.16b");
|
||||
TEST_SINGLE(sdot(SubRegSize::i32Bit, QReg::q30, QReg::q29, QReg::q28), "sdot v30.4s, v29.16b, v28.16b");
|
||||
TEST_SINGLE(sdot(SubRegSize::i64Bit, QReg::q30, QReg::q29, QReg::q28), "sdot v30.2d, v29.16b, v28.16b");
|
||||
TEST_SINGLE(sdot(SubRegSize::i8Bit, DReg::d30, DReg::d29, DReg::d28), "sdot v30.8b, v29.8b, v28.8b");
|
||||
TEST_SINGLE(sdot(SubRegSize::i16Bit, DReg::d30, DReg::d29, DReg::d28), "sdot v30.4h, v29.8b, v28.8b");
|
||||
TEST_SINGLE(sdot(SubRegSize::i32Bit, DReg::d30, DReg::d29, DReg::d28), "sdot v30.2s, v29.8b, v28.8b");
|
||||
// TEST_SINGLE(sdot(SubRegSize::i64Bit, DReg::d30, DReg::d29, DReg::d28), "sdot v30.1d, v29.8b, v28.8b");
|
||||
|
||||
TEST_SINGLE(usdot(QReg::q30, QReg::q29, QReg::q28), "usdot v30.4s, v29.16b, v28.16b");
|
||||
TEST_SINGLE(usdot(DReg::d30, DReg::d29, DReg::d28), "usdot v30.2s, v29.8b, v28.8b");
|
||||
|
||||
TEST_SINGLE(sqrdmlah(SubRegSize::i8Bit, QReg::q30, QReg::q29, QReg::q28), "sqrdmlah v30.16b, v29.16b, v28.16b");
|
||||
TEST_SINGLE(sqrdmlah(SubRegSize::i16Bit, QReg::q30, QReg::q29, QReg::q28), "sqrdmlah v30.8h, v29.8h, v28.8h");
|
||||
TEST_SINGLE(sqrdmlah(SubRegSize::i32Bit, QReg::q30, QReg::q29, QReg::q28), "sqrdmlah v30.4s, v29.4s, v28.4s");
|
||||
TEST_SINGLE(sqrdmlah(SubRegSize::i64Bit, QReg::q30, QReg::q29, QReg::q28), "sqrdmlah v30.2d, v29.2d, v28.2d");
|
||||
TEST_SINGLE(sqrdmlah(SubRegSize::i8Bit, DReg::d30, DReg::d29, DReg::d28), "sqrdmlah v30.8b, v29.8b, v28.8b");
|
||||
TEST_SINGLE(sqrdmlah(SubRegSize::i16Bit, DReg::d30, DReg::d29, DReg::d28), "sqrdmlah v30.4h, v29.4h, v28.4h");
|
||||
TEST_SINGLE(sqrdmlah(SubRegSize::i32Bit, DReg::d30, DReg::d29, DReg::d28), "sqrdmlah v30.2s, v29.2s, v28.2s");
|
||||
// TEST_SINGLE(sqrdmlah(SubRegSize::i64Bit, DReg::d30, DReg::d29, DReg::d28), "sqrdmlah v30.1d, v29.1d, v28.1d");
|
||||
|
||||
TEST_SINGLE(sqrdmlsh(SubRegSize::i8Bit, QReg::q30, QReg::q29, QReg::q28), "sqrdmlsh v30.16b, v29.16b, v28.16b");
|
||||
TEST_SINGLE(sqrdmlsh(SubRegSize::i16Bit, QReg::q30, QReg::q29, QReg::q28), "sqrdmlsh v30.8h, v29.8h, v28.8h");
|
||||
TEST_SINGLE(sqrdmlsh(SubRegSize::i32Bit, QReg::q30, QReg::q29, QReg::q28), "sqrdmlsh v30.4s, v29.4s, v28.4s");
|
||||
TEST_SINGLE(sqrdmlsh(SubRegSize::i64Bit, QReg::q30, QReg::q29, QReg::q28), "sqrdmlsh v30.2d, v29.2d, v28.2d");
|
||||
TEST_SINGLE(sqrdmlsh(SubRegSize::i8Bit, DReg::d30, DReg::d29, DReg::d28), "sqrdmlsh v30.8b, v29.8b, v28.8b");
|
||||
TEST_SINGLE(sqrdmlsh(SubRegSize::i16Bit, DReg::d30, DReg::d29, DReg::d28), "sqrdmlsh v30.4h, v29.4h, v28.4h");
|
||||
TEST_SINGLE(sqrdmlsh(SubRegSize::i32Bit, DReg::d30, DReg::d29, DReg::d28), "sqrdmlsh v30.2s, v29.2s, v28.2s");
|
||||
// TEST_SINGLE(sqrdmlsh(SubRegSize::i64Bit, DReg::d30, DReg::d29, DReg::d28), "sqrdmlsh v30.1d, v29.1d, v28.1d");
|
||||
|
||||
TEST_SINGLE(udot(SubRegSize::i8Bit, QReg::q30, QReg::q29, QReg::q28), "udot v30.16b, v29.16b, v28.16b");
|
||||
TEST_SINGLE(udot(SubRegSize::i16Bit, QReg::q30, QReg::q29, QReg::q28), "udot v30.8h, v29.16b, v28.16b");
|
||||
TEST_SINGLE(udot(SubRegSize::i32Bit, QReg::q30, QReg::q29, QReg::q28), "udot v30.4s, v29.16b, v28.16b");
|
||||
TEST_SINGLE(udot(SubRegSize::i64Bit, QReg::q30, QReg::q29, QReg::q28), "udot v30.2d, v29.16b, v28.16b");
|
||||
TEST_SINGLE(udot(SubRegSize::i8Bit, DReg::d30, DReg::d29, DReg::d28), "udot v30.8b, v29.8b, v28.8b");
|
||||
TEST_SINGLE(udot(SubRegSize::i16Bit, DReg::d30, DReg::d29, DReg::d28), "udot v30.4h, v29.8b, v28.8b");
|
||||
TEST_SINGLE(udot(SubRegSize::i32Bit, DReg::d30, DReg::d29, DReg::d28), "udot v30.2s, v29.8b, v28.8b");
|
||||
// TEST_SINGLE(udot(SubRegSize::i64Bit, DReg::d30, DReg::d29, DReg::d28), "udot v30.1d, v29.8b, v28.8b");
|
||||
|
||||
|
||||
@@ -11,14 +11,11 @@ if (COMPILE_VIXL_DISASSEMBLER)
|
||||
catch_discover_tests(Emitter_${TEST_NAME} TEST_SUFFIX ".${TEST_NAME}.Emitter")
|
||||
endforeach()
|
||||
|
||||
execute_process(COMMAND "nproc" OUTPUT_VARIABLE CORES)
|
||||
string(STRIP ${CORES} CORES)
|
||||
|
||||
add_custom_target(
|
||||
emitter_tests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}/"
|
||||
USES_TERMINAL
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" "-j${CORES}" "-R" "\.*.Emitter$$")
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" ${TEST_JOB_FLAG} "-R" "\.*.Emitter$$")
|
||||
else()
|
||||
message(AUTHOR_WARNING "Tests are enabled but vixl disassembler is not. Emitter tests won't be built.")
|
||||
endif()
|
||||
@@ -4539,6 +4539,24 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE store multiple structures
|
||||
"[x29, x30, lsl #3]");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE contiguous non-temporal store (scalar plus immediate)") {
|
||||
TEST_SINGLE(stnt1b(ZReg::z31, PReg::p6, Reg::r29, 0), "stnt1b {z31.b}, p6, [x29]");
|
||||
TEST_SINGLE(stnt1b(ZReg::z31, PReg::p6, Reg::r29, -8), "stnt1b {z31.b}, p6, [x29, #-8, mul vl]");
|
||||
TEST_SINGLE(stnt1b(ZReg::z31, PReg::p6, Reg::r29, 7), "stnt1b {z31.b}, p6, [x29, #7, mul vl]");
|
||||
|
||||
TEST_SINGLE(stnt1h(ZReg::z31, PReg::p6, Reg::r29, 0), "stnt1h {z31.h}, p6, [x29]");
|
||||
TEST_SINGLE(stnt1h(ZReg::z31, PReg::p6, Reg::r29, -8), "stnt1h {z31.h}, p6, [x29, #-8, mul vl]");
|
||||
TEST_SINGLE(stnt1h(ZReg::z31, PReg::p6, Reg::r29, 7), "stnt1h {z31.h}, p6, [x29, #7, mul vl]");
|
||||
|
||||
TEST_SINGLE(stnt1w(ZReg::z31, PReg::p6, Reg::r29, 0), "stnt1w {z31.s}, p6, [x29]");
|
||||
TEST_SINGLE(stnt1w(ZReg::z31, PReg::p6, Reg::r29, -8), "stnt1w {z31.s}, p6, [x29, #-8, mul vl]");
|
||||
TEST_SINGLE(stnt1w(ZReg::z31, PReg::p6, Reg::r29, 7), "stnt1w {z31.s}, p6, [x29, #7, mul vl]");
|
||||
|
||||
TEST_SINGLE(stnt1d(ZReg::z31, PReg::p6, Reg::r29, 0), "stnt1d {z31.d}, p6, [x29]");
|
||||
TEST_SINGLE(stnt1d(ZReg::z31, PReg::p6, Reg::r29, -8), "stnt1d {z31.d}, p6, [x29, #-8, mul vl]");
|
||||
TEST_SINGLE(stnt1d(ZReg::z31, PReg::p6, Reg::r29, 7), "stnt1d {z31.d}, p6, [x29, #7, mul vl]");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE store multiple structures (scalar plus immediate)") {
|
||||
TEST_SINGLE(st2b(ZReg::z31, ZReg::z0, PReg::p6, Reg::r29, 0), "st2b {z31.b, z0.b}, p6, [x29]");
|
||||
TEST_SINGLE(st2b(ZReg::z26, ZReg::z27, PReg::p6, Reg::r29, 0), "st2b {z26.b, z27.b}, p6, [x29]");
|
||||
|
||||
@@ -3,6 +3,7 @@ import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import re
|
||||
|
||||
_Arch = None
|
||||
def GetArch():
|
||||
@@ -131,7 +132,7 @@ def GetCPUFeaturesVersion():
|
||||
return _ArchVersion
|
||||
|
||||
_PPAInstalled = None
|
||||
FEXPPA = "http://ppa.launchpad.net/fex-emu/fex/ubuntu"
|
||||
FEXPPA_REGEX = r".*\/fex-emu\/fex\/ubuntu$"
|
||||
|
||||
def GetPPAStatus():
|
||||
global _PPAInstalled
|
||||
@@ -147,7 +148,7 @@ def GetPPAStatus():
|
||||
LineSplit = Line.split(" ")
|
||||
|
||||
# 'status' 'URL' 'series' 'arch' 'type'
|
||||
if LineSplit[1] == FEXPPA:
|
||||
if re.match(FEXPPA_REGEX, LineSplit[1]):
|
||||
_PPAInstalled = True
|
||||
break
|
||||
|
||||
|
||||
@@ -53,6 +53,7 @@ class HostFeatures(Flag) :
|
||||
FEATURE_FLAGM = (1 << 8)
|
||||
FEATURE_FLAGM2 = (1 << 9)
|
||||
FEATURE_CRYPTO = (1 << 10)
|
||||
FEATURE_AES256 = (1 << 11)
|
||||
|
||||
HostFeaturesLookup = {
|
||||
"SVE128" : HostFeatures.FEATURE_SVE128,
|
||||
@@ -66,6 +67,7 @@ HostFeaturesLookup = {
|
||||
"FLAGM" : HostFeatures.FEATURE_FLAGM,
|
||||
"FLAGM2" : HostFeatures.FEATURE_FLAGM2,
|
||||
"CRYPTO" : HostFeatures.FEATURE_CRYPTO,
|
||||
"AES256" : HostFeatures.FEATURE_AES256,
|
||||
}
|
||||
|
||||
def GetHostFeatures(data):
|
||||
|
||||
@@ -74,6 +74,7 @@ class HostFeatures(Flag) :
|
||||
FEATURE_CLWB = (1 << 8)
|
||||
FEATURE_LINUX = (1 << 9)
|
||||
FEATURE_AVX2 = (1 << 10)
|
||||
FEATURE_AES256 = (1 << 11)
|
||||
|
||||
RegStringLookup = {
|
||||
"NONE": Regs.REG_NONE,
|
||||
@@ -147,6 +148,7 @@ HostFeaturesLookup = {
|
||||
"CLWB" : HostFeatures.FEATURE_CLWB,
|
||||
"LINUX" : HostFeatures.FEATURE_LINUX,
|
||||
"AVX2" : HostFeatures.FEATURE_AVX2,
|
||||
"AES256" : HostFeatures.FEATURE_AES256,
|
||||
}
|
||||
|
||||
def parse_hexstring(s):
|
||||
|
||||
Executable
+14
@@ -0,0 +1,14 @@
|
||||
#! /bin/bash
|
||||
set -e
|
||||
|
||||
# Make sure we actually build
|
||||
ninja
|
||||
|
||||
# Run tests, ignoring the retval since there will be changes.
|
||||
ninja instcountci_tests || true
|
||||
|
||||
# Now we can update.
|
||||
ninja instcountci_update_tests
|
||||
|
||||
# Commit the result in bulk.
|
||||
git commit -sam "InstCountCI: Update"
|
||||
@@ -5,6 +5,7 @@ set(SRCS
|
||||
Config.cpp
|
||||
ArgumentLoader.cpp
|
||||
EnvironmentLoader.cpp
|
||||
JSONPool.cpp
|
||||
StringUtil.cpp)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/ArgumentLoader.h"
|
||||
#include "Common/Config.h"
|
||||
#include "Common/JSONPool.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
@@ -22,38 +23,15 @@
|
||||
|
||||
namespace FEX::Config {
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
fextl::unique_ptr<fextl::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = fextl::make_unique<fextl::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
static void LoadJSonConfig(const fextl::string& Config, std::function<void(const char* Name, const char* ConfigSring)> Func) {
|
||||
fextl::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject =
|
||||
{
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
FEX::JSON::JsonAllocator Pool {};
|
||||
const json_t* json = FEX::JSON::CreateJSON(Data, Pool);
|
||||
|
||||
const json_t* json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/JSONPool.h"
|
||||
|
||||
namespace FEX::JSON {
|
||||
json_t* PoolInit(jsonPool_t* Pool);
|
||||
json_t* PoolAlloc(jsonPool_t* Pool);
|
||||
|
||||
JsonAllocator::JsonAllocator()
|
||||
: PoolObject {
|
||||
.init = FEX::JSON::PoolInit,
|
||||
.alloc = FEX::JSON::PoolAlloc,
|
||||
} {}
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects.emplace(alloc->json_objects.end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects.emplace(alloc->json_objects.end());
|
||||
}
|
||||
} // namespace FEX::JSON
|
||||
@@ -0,0 +1,25 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEX::JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
fextl::list<json_t> json_objects;
|
||||
|
||||
JsonAllocator();
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
template<typename T>
|
||||
const json_t* CreateJSON(T& Container, JsonAllocator& Allocator) {
|
||||
if (Container.empty()) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
return json_createWithPool(&Container.at(0), &Allocator.PoolObject);
|
||||
}
|
||||
} // namespace FEX::JSON
|
||||
@@ -514,7 +514,6 @@ int main(int argc, char** argv, char** const envp) {
|
||||
SVEWidth = 128;
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_SVE256) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEAVX);
|
||||
SVEWidth = 256;
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_CLZERO) {
|
||||
@@ -551,9 +550,6 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_SVE128) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLESVE);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_SVE256) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEAVX);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_CLZERO) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLECLZERO);
|
||||
}
|
||||
|
||||
@@ -307,7 +307,7 @@ public:
|
||||
FEATURE_BMI2 = (1 << 7),
|
||||
FEATURE_CLWB = (1 << 8),
|
||||
FEATURE_LINUX = (1 << 9),
|
||||
FEATURE_AVX2 = (1 << 10),
|
||||
FEATURE_AES256 = (1 << 10),
|
||||
};
|
||||
|
||||
bool Requires3DNow() const {
|
||||
@@ -319,9 +319,6 @@ public:
|
||||
bool RequiresAVX() const {
|
||||
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_AVX;
|
||||
}
|
||||
bool RequiresAVX2() const {
|
||||
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_AVX2;
|
||||
}
|
||||
bool RequiresRAND() const {
|
||||
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_RAND;
|
||||
}
|
||||
@@ -343,6 +340,9 @@ public:
|
||||
bool RequiresLinux() const {
|
||||
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_LINUX;
|
||||
}
|
||||
bool RequiresAES256() const {
|
||||
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_AES256;
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ConfigDumpGPRs, DUMPGPRS);
|
||||
@@ -488,9 +488,6 @@ public:
|
||||
bool RequiresAVX() const {
|
||||
return Config.RequiresAVX();
|
||||
}
|
||||
bool RequiresAVX2() const {
|
||||
return Config.RequiresAVX2();
|
||||
}
|
||||
bool RequiresRAND() const {
|
||||
return Config.RequiresRAND();
|
||||
}
|
||||
@@ -512,6 +509,9 @@ public:
|
||||
bool RequiresLinux() const {
|
||||
return Config.RequiresLinux();
|
||||
}
|
||||
bool RequiresAES256() const {
|
||||
return Config.RequiresAES256();
|
||||
}
|
||||
|
||||
private:
|
||||
constexpr static uint64_t STACK_OFFSET = 0xc000'0000;
|
||||
|
||||
@@ -563,14 +563,14 @@ void FillHackConfig() {
|
||||
bool MemcpyTSOEnabled = MemcpyTSO.has_value() && **MemcpyTSO == "1";
|
||||
bool HalfBarrierTSOEnabled = HalfBarrierTSO.has_value() && **HalfBarrierTSO == "1";
|
||||
|
||||
if (ImGui::Checkbox("TSO Enabled", &TSOEnabled)) {
|
||||
if (ImGui::Checkbox("TSO Emulation Enabled", &TSOEnabled)) {
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, TSOEnabled ? "1" : "0");
|
||||
ConfigChanged = true;
|
||||
}
|
||||
|
||||
if (TSOEnabled) {
|
||||
if (ImGui::TreeNodeEx("TSO sub-options", ImGuiTreeNodeFlags_Leaf)) {
|
||||
if (ImGui::Checkbox("Vector TSO Enabled", &VectorTSOEnabled)) {
|
||||
if (ImGui::TreeNodeEx("TSO Emulation sub-options", ImGuiTreeNodeFlags_Leaf)) {
|
||||
if (ImGui::Checkbox("Vector TSO Emulation Enabled", &VectorTSOEnabled)) {
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, VectorTSOEnabled ? "1" : "0");
|
||||
ConfigChanged = true;
|
||||
}
|
||||
@@ -580,7 +580,7 @@ void FillHackConfig() {
|
||||
ImGui::EndTooltip();
|
||||
}
|
||||
|
||||
if (ImGui::Checkbox("Memcpy TSO Enabled", &MemcpyTSOEnabled)) {
|
||||
if (ImGui::Checkbox("Memcpy TSO Emulation Enabled", &MemcpyTSOEnabled)) {
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, MemcpyTSOEnabled ? "1" : "0");
|
||||
ConfigChanged = true;
|
||||
}
|
||||
@@ -590,7 +590,7 @@ void FillHackConfig() {
|
||||
ImGui::EndTooltip();
|
||||
}
|
||||
|
||||
if (ImGui::Checkbox("Unaligned Half-Barrier TSO Enabled", &HalfBarrierTSOEnabled)) {
|
||||
if (ImGui::Checkbox("Unaligned Half-Barrier TSO Emulation Enabled", &HalfBarrierTSOEnabled)) {
|
||||
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_HALFBARRIERTSOENABLED, HalfBarrierTSOEnabled ? "1" : "0");
|
||||
ConfigChanged = true;
|
||||
}
|
||||
|
||||
@@ -12,6 +12,99 @@
|
||||
|
||||
#include <stdio.h>
|
||||
#include <string>
|
||||
#include <sys/prctl.h>
|
||||
|
||||
namespace {
|
||||
struct TSOEmulationFacts {
|
||||
bool LSE {}, LSE2 {};
|
||||
bool HardwareTSO {};
|
||||
bool LRCPC1 {}, LRCPC2 {}, LRCPC3 {};
|
||||
};
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
bool CheckForHardwareTSO() {
|
||||
// We need to check if these are defined or not. This is a very fresh feature.
|
||||
#ifndef PR_GET_MEM_MODEL
|
||||
#define PR_GET_MEM_MODEL 0x6d4d444c
|
||||
#endif
|
||||
#ifndef PR_SET_MEM_MODEL
|
||||
#define PR_SET_MEM_MODEL 0x4d4d444c
|
||||
#endif
|
||||
#ifndef PR_SET_MEM_MODEL_DEFAULT
|
||||
#define PR_SET_MEM_MODEL_DEFAULT 0
|
||||
#endif
|
||||
#ifndef PR_SET_MEM_MODEL_TSO
|
||||
#define PR_SET_MEM_MODEL_TSO 1
|
||||
#endif
|
||||
// Check to see if this is supported.
|
||||
auto Result = prctl(PR_GET_MEM_MODEL, 0, 0, 0, 0);
|
||||
if (Result == -1) {
|
||||
// Unsupported, early exit.
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Result == PR_SET_MEM_MODEL_DEFAULT) {
|
||||
// Try to set the TSO mode if we are currently default.
|
||||
Result = prctl(PR_SET_MEM_MODEL, PR_SET_MEM_MODEL_TSO, 0, 0, 0);
|
||||
if (Result == 0) {
|
||||
Result = prctl(PR_SET_MEM_MODEL, PR_SET_MEM_MODEL_DEFAULT, 0, 0, 0);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
enum ISAR0_FIELDS {
|
||||
LSE = 20,
|
||||
};
|
||||
|
||||
enum ISAR1_FIELDS {
|
||||
LRCPC = 20,
|
||||
};
|
||||
|
||||
enum MMFR2_FIELDS {
|
||||
AT = 32,
|
||||
};
|
||||
|
||||
constexpr static uint32_t IDFIELDMASK = 0b1111;
|
||||
uint64_t GetISAR0() {
|
||||
uint64_t Result {};
|
||||
asm("mrs %0, ID_AA64ISAR0_EL1;" : "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint64_t GetISAR1() {
|
||||
uint64_t Result {};
|
||||
asm("mrs %0, ID_AA64ISAR1_EL1;" : "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint64_t GetMMFR2() {
|
||||
uint64_t Result {};
|
||||
asm("mrs %0, ID_AA64MMFR2_EL1;" : "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
TSOEmulationFacts GetTSOEmulationFacts() {
|
||||
const auto ISAR0 = GetISAR0();
|
||||
const auto ISAR1 = GetISAR1();
|
||||
const auto MMFR2 = GetMMFR2();
|
||||
|
||||
return {
|
||||
.LSE = ((ISAR0 >> ISAR0_FIELDS::LSE) & IDFIELDMASK) >= 0b0010,
|
||||
.LSE2 = ((MMFR2 >> MMFR2_FIELDS::AT) & IDFIELDMASK) >= 0b0001,
|
||||
.HardwareTSO = CheckForHardwareTSO(),
|
||||
.LRCPC1 = ((ISAR1 >> ISAR1_FIELDS::LRCPC) & IDFIELDMASK) >= 0b0001,
|
||||
.LRCPC2 = ((ISAR1 >> ISAR1_FIELDS::LRCPC) & IDFIELDMASK) >= 0b0010,
|
||||
.LRCPC3 = ((ISAR1 >> ISAR1_FIELDS::LRCPC) & IDFIELDMASK) >= 0b0011,
|
||||
};
|
||||
}
|
||||
#else
|
||||
TSOEmulationFacts GetTSOEmulationFacts() {
|
||||
return {};
|
||||
}
|
||||
#endif
|
||||
} // namespace
|
||||
|
||||
int main(int argc, char** argv, char** envp) {
|
||||
FEXCore::Config::Initialize();
|
||||
@@ -30,6 +123,8 @@ int main(int argc, char** argv, char** envp) {
|
||||
|
||||
Parser.add_option("--current-rootfs").action("store_true").help("Print the directory that contains the FEX rootfs. Mounted in the case of squashfs");
|
||||
|
||||
Parser.add_option("--tso-emulation-info").action("store_true").help("Print how FEX is emulating the x86-TSO memory model.");
|
||||
|
||||
Parser.add_option("--version").action("store_true").help("Print the installed FEX-Emu version");
|
||||
|
||||
optparse::Values Options = Parser.parse_args(argc, argv);
|
||||
@@ -71,5 +166,69 @@ int main(int argc, char** argv, char** envp) {
|
||||
}
|
||||
}
|
||||
|
||||
if (Options.is_set_by_user("tso_emulation_info")) {
|
||||
auto TSOFacts = GetTSOEmulationFacts();
|
||||
const char* GPRMemoryTSOEmulation {};
|
||||
const char* MemcpyMemoryTSOEmulation {};
|
||||
const char* VectorMemoryTSOEmulation {};
|
||||
const char* UnalignedMemoryLoadStoreTSOEmulation {};
|
||||
|
||||
if (TSOFacts.HardwareTSO) {
|
||||
GPRMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
|
||||
} else if (TSOFacts.LRCPC3) {
|
||||
GPRMemoryTSOEmulation = "\e[32mLRCPC3\e[0m";
|
||||
} else if (TSOFacts.LRCPC2) {
|
||||
GPRMemoryTSOEmulation = "\e[32mLRCPC2\e[0m";
|
||||
} else if (TSOFacts.LRCPC1) {
|
||||
GPRMemoryTSOEmulation = "\e[32mLRCPC\e[0m";
|
||||
} else {
|
||||
GPRMemoryTSOEmulation = "\e[31mAtomics\e[0m";
|
||||
}
|
||||
|
||||
// Memcpy only uses Hardware TSO, LRCPC, and Atomics.
|
||||
if (TSOFacts.HardwareTSO) {
|
||||
MemcpyMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
|
||||
} else if (TSOFacts.LRCPC1) {
|
||||
MemcpyMemoryTSOEmulation = "\e[32mLRCPC\e[0m";
|
||||
} else {
|
||||
MemcpyMemoryTSOEmulation = "\e[31mAtomics\e[0m";
|
||||
}
|
||||
|
||||
if (TSOFacts.HardwareTSO) {
|
||||
VectorMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
|
||||
} else if (TSOFacts.LRCPC3) {
|
||||
VectorMemoryTSOEmulation = "\e[32mLRCPC3\e[0m";
|
||||
} else {
|
||||
VectorMemoryTSOEmulation = "\e[31mHalf-Barriers\e[0m";
|
||||
}
|
||||
|
||||
if (TSOFacts.HardwareTSO) {
|
||||
UnalignedMemoryLoadStoreTSOEmulation = "\e[32mHardware TSO\e[0m";
|
||||
} else {
|
||||
UnalignedMemoryLoadStoreTSOEmulation = "\e[31mHalf-Barriers\e[0m";
|
||||
}
|
||||
|
||||
fprintf(stdout, "Hardware Features:\n");
|
||||
fprintf(stdout, "\tMemory atomics emulation method: %s\n", TSOFacts.LSE ? "\e[32mLSE\e[0m" : "\e[31mLL/SC\e[0m");
|
||||
fprintf(stdout, "\tUnaligned atomic memory granularity: %s\n", TSOFacts.LSE2 ? "\e[32m16-byte\e[0m" : "\e[31mNatural alignment\e[0m");
|
||||
///< TODO: Once TME is supported by hardware this can change.
|
||||
fprintf(stdout, "\tUnaligned memory atomic emulation: %s\n", TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m");
|
||||
fprintf(stdout, "\tUnaligned memory loadstore emulation: %s\n", UnalignedMemoryLoadStoreTSOEmulation);
|
||||
fprintf(stdout, "\tGPR memory model emulation: %s\n", GPRMemoryTSOEmulation);
|
||||
fprintf(stdout, "\tMemcpy memory model emulation: %s\n", MemcpyMemoryTSOEmulation);
|
||||
fprintf(stdout, "\tVector memory model emulation: %s\n", VectorMemoryTSOEmulation);
|
||||
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
|
||||
|
||||
fprintf(stdout, "\nConfiguration:\n");
|
||||
fprintf(stdout, "\tTSO Emulation: %s\n", TSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\tMemcpy TSO Emulation: %s\n", TSOEnabled() && MemcpySetTSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\tVector TSO Emulation: %s\n", TSOEnabled() && VectorTSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\tHalf-barrier unaligned TSO emulation: %s\n", TSOEnabled() && HalfBarrierTSOEnabled() ? "Enabled" : "Disabled");
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -3,6 +3,7 @@
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include "Common/cpp-optparse/OptionParser.h"
|
||||
#include "Common/JSONPool.h"
|
||||
#include "XXFileHash.h"
|
||||
|
||||
#include "Common/ArgumentLoader.h"
|
||||
@@ -510,23 +511,6 @@ bool DownloadToPathWithZenityProgress(const fextl::string& URL, const fextl::str
|
||||
return Exec::ExecAndWaitForResponse(ExecveArgs[0], const_cast<char* const*>(ExecveArgs.data())) == 0;
|
||||
}
|
||||
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
std::unique_ptr<std::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = std::make_unique<std::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
std::optional<std::vector<FileTargets>> GetRootFSLinks() {
|
||||
// Decode the filetargets
|
||||
std::string Data = DownloadToString(DownloadURL);
|
||||
@@ -535,15 +519,9 @@ std::optional<std::vector<FileTargets>> GetRootFSLinks() {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject =
|
||||
{
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
FEX::JSON::JsonAllocator Pool {};
|
||||
const json_t* json = FEX::JSON::CreateJSON(Data, Pool);
|
||||
|
||||
const json_t* json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
fprintf(stderr, "Couldn't create json");
|
||||
return {};
|
||||
|
||||
@@ -162,11 +162,8 @@ foreach(Index RANGE 0 ${ARG_COUNT} 2)
|
||||
set_property(TEST ${TEST_NAME_ARCH}_aarch64 APPEND PROPERTY DEPENDS "${HEADER}")
|
||||
endforeach()
|
||||
|
||||
execute_process(COMMAND "nproc" OUTPUT_VARIABLE CORES)
|
||||
string(STRIP ${CORES} CORES)
|
||||
|
||||
add_custom_target(
|
||||
struct_verifier
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}"
|
||||
USES_TERMINAL
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" "-j${CORES}" "-R" "Test_verify*")
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" ${TEST_JOB_FLAG} "-R" "Test_verify*")
|
||||
@@ -375,6 +375,16 @@ fextl::string GenerateCPUInfo(FEXCore::Context::Context* ctx, uint32_t CPUCores)
|
||||
FLAG(res_7.edx & (1 << 28), "flush_l1d") FLAG(res_7.edx & (1 << 29), "arch_capabilities")
|
||||
}
|
||||
|
||||
// Get the cycle counter frequency from CPUID function 15h.
|
||||
auto res_15 = ctx->RunCPUIDFunction(0x15, 0);
|
||||
// Frequency is calculated in Hz, we need to convert it to megahertz since FEX is guaranteed to return >= 1Ghz.
|
||||
constexpr double HzInMhz = 1000000.0;
|
||||
double Frequency = 1.0 / (static_cast<double>(res_15.eax) / (static_cast<double>(res_15.ebx) * static_cast<double>(res_15.ecx)));
|
||||
Frequency /= HzInMhz;
|
||||
// Generate the cycle counter frequency string in the format expected by cpuinfo.
|
||||
// ex: `4000.000`
|
||||
const auto FrequencyString = fextl::fmt::format("{:.3f}", Frequency);
|
||||
|
||||
for (int i = 0; i < CPUCores; ++i) {
|
||||
cpu_stream << "processor\t: " << i << std::endl; // Logical id
|
||||
cpu_stream << "vendor_id\t: " << vendorid.Str << std::endl;
|
||||
@@ -392,7 +402,7 @@ fextl::string GenerateCPUInfo(FEXCore::Context::Context* ctx, uint32_t CPUCores)
|
||||
cpu_stream << "model name\t: " << modelname.Str << std::endl;
|
||||
cpu_stream << "stepping\t: " << info.Stepping << std::endl;
|
||||
cpu_stream << "microcode\t: 0x0" << std::endl;
|
||||
cpu_stream << "cpu MHz\t\t: 3000" << std::endl;
|
||||
cpu_stream << "cpu MHz\t\t: " << FrequencyString << std::endl;
|
||||
cpu_stream << "cache size\t: 512 KB" << std::endl;
|
||||
cpu_stream << "physical id\t: 0" << std::endl; // Socket id (always 0 for a single socket system)
|
||||
cpu_stream << "siblings\t: " << CPUCores << std::endl; // Number of logical cores
|
||||
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include "Common/Config.h"
|
||||
#include "Common/FDUtils.h"
|
||||
#include "Common/JSONPool.h"
|
||||
|
||||
#include "FEXCore/Config/Config.h"
|
||||
#include "LinuxSyscalls/FileManagement.h"
|
||||
@@ -42,25 +43,6 @@ $end_info$
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
fextl::unique_ptr<fextl::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = fextl::make_unique<fextl::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
} // namespace JSON
|
||||
|
||||
namespace FEX::HLE {
|
||||
bool FileManager::RootFSPathExists(const char* Filepath) {
|
||||
LOGMAN_THROW_A_FMT(Filepath && Filepath[0] == '/', "Filepath needs to be absolute");
|
||||
@@ -71,7 +53,6 @@ void FileManager::LoadThunkDatabase(fextl::unordered_map<fextl::string, ThunkDBO
|
||||
auto ThunkDBPath = FEXCore::Config::GetConfigDirectory(Global) + "ThunksDB.json";
|
||||
fextl::vector<char> FileData;
|
||||
if (FEXCore::FileLoading::LoadFile(FileData, ThunkDBPath)) {
|
||||
FileData.push_back(0);
|
||||
|
||||
// If the thunksDB file exists then we need to check if the rootfs supports multi-arch or not.
|
||||
const bool RootFSIsMultiarch = RootFSPathExists("/usr/lib/x86_64-linux-gnu/") || RootFSPathExists("/usr/lib/i386-linux-gnu/");
|
||||
@@ -112,15 +93,12 @@ void FileManager::LoadThunkDatabase(fextl::unordered_map<fextl::string, ThunkDBO
|
||||
}
|
||||
}
|
||||
|
||||
JSON::JsonAllocator Pool {
|
||||
.PoolObject =
|
||||
{
|
||||
.init = JSON::PoolInit,
|
||||
.alloc = JSON::PoolAlloc,
|
||||
},
|
||||
};
|
||||
FEX::JSON::JsonAllocator Pool {};
|
||||
const json_t* json = FEX::JSON::CreateJSON(FileData, Pool);
|
||||
|
||||
const json_t* json = json_createWithPool(&FileData.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
return;
|
||||
}
|
||||
|
||||
const json_t* DB = json_getProperty(json, "DB");
|
||||
if (!DB || JSON_OBJ != json_getType(DB)) {
|
||||
@@ -258,16 +236,14 @@ FileManager::FileManager(FEXCore::Context::Context* ctx)
|
||||
for (const auto& Path : ConfigPaths) {
|
||||
fextl::vector<char> FileData;
|
||||
if (FEXCore::FileLoading::LoadFile(FileData, Path)) {
|
||||
JSON::JsonAllocator Pool {
|
||||
.PoolObject =
|
||||
{
|
||||
.init = JSON::PoolInit,
|
||||
.alloc = JSON::PoolAlloc,
|
||||
},
|
||||
};
|
||||
FEX::JSON::JsonAllocator Pool {};
|
||||
|
||||
// If a thunks DB property exists then we pull in data from the thunks database
|
||||
const json_t* json = json_createWithPool(&FileData.at(0), &Pool.PoolObject);
|
||||
const json_t* json = FEX::JSON::CreateJSON(FileData, Pool);
|
||||
if (!json) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const json_t* ThunksDB = json_getProperty(json, "ThunksDB");
|
||||
if (!ThunksDB) {
|
||||
continue;
|
||||
|
||||
@@ -439,14 +439,9 @@ void SignalDelegator::RestoreFrame_x64(FEXCore::Core::InternalThreadState* Threa
|
||||
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][0], &fpstate->_xmm[i], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][2], &xstate->ymmh.ymmh_space[i], sizeof(__uint128_t));
|
||||
}
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
@@ -517,14 +512,9 @@ void SignalDelegator::RestoreFrame_ia32(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
// Extended XMM state
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][0], &fpstate->_xmm[i], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][2], &xstate->ymmh.ymmh_space[i], sizeof(__uint128_t));
|
||||
}
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
@@ -596,14 +586,9 @@ void SignalDelegator::RestoreRTFrame_ia32(FEXCore::Core::InternalThreadState* Th
|
||||
|
||||
// Extended XMM state
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][0], &fpstate->_xmm[i], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][2], &xstate->ymmh.ymmh_space[i], sizeof(__uint128_t));
|
||||
}
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
@@ -726,14 +711,9 @@ uint64_t SignalDelegator::SetupFrame_x64(FEXCore::Core::InternalThreadState* Thr
|
||||
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
@@ -771,9 +751,9 @@ uint64_t SignalDelegator::SetupFrame_x64(FEXCore::Core::InternalThreadState* Thr
|
||||
return NewGuestSP;
|
||||
}
|
||||
|
||||
uint64_t SignalDelegator::SetupFrame_ia32(ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame* Frame,
|
||||
int Signal, siginfo_t* HostSigInfo, void* ucontext, GuestSigAction* GuestAction,
|
||||
stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags) {
|
||||
uint64_t SignalDelegator::SetupFrame_ia32(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* ContextBackup,
|
||||
FEXCore::Core::CpuStateFrame* Frame, int Signal, siginfo_t* HostSigInfo, void* ucontext,
|
||||
GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags) {
|
||||
|
||||
const bool IsAVXEnabled = Config.SupportsAVX;
|
||||
const uint64_t SignalReturn = reinterpret_cast<uint64_t>(VDSOPointers.VDSO_kernel_sigreturn);
|
||||
@@ -851,15 +831,11 @@ uint64_t SignalDelegator::SetupFrame_ia32(ArchHelpers::Context::ContextBackup* C
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
@@ -904,9 +880,9 @@ uint64_t SignalDelegator::SetupFrame_ia32(ArchHelpers::Context::ContextBackup* C
|
||||
return NewGuestSP;
|
||||
}
|
||||
|
||||
uint64_t SignalDelegator::SetupRTFrame_ia32(ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame* Frame,
|
||||
int Signal, siginfo_t* HostSigInfo, void* ucontext, GuestSigAction* GuestAction,
|
||||
stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags) {
|
||||
uint64_t SignalDelegator::SetupRTFrame_ia32(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* ContextBackup,
|
||||
FEXCore::Core::CpuStateFrame* Frame, int Signal, siginfo_t* HostSigInfo, void* ucontext,
|
||||
GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags) {
|
||||
|
||||
const bool IsAVXEnabled = Config.SupportsAVX;
|
||||
const uint64_t SignalReturn = reinterpret_cast<uint64_t>(VDSOPointers.VDSO_kernel_rt_sigreturn);
|
||||
@@ -990,15 +966,11 @@ uint64_t SignalDelegator::SetupRTFrame_ia32(ArchHelpers::Context::ContextBackup*
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, xstate->ymmh.ymmh_space);
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
@@ -1193,9 +1165,9 @@ bool SignalDelegator::HandleDispatcherGuestSignal(FEXCore::Core::InternalThreadS
|
||||
} else {
|
||||
const bool SigInfoFrame = (GuestAction->sa_flags & SA_SIGINFO) == SA_SIGINFO;
|
||||
if (SigInfoFrame) {
|
||||
NewGuestSP = SetupRTFrame_ia32(ContextBackup, Frame, Signal, HostSigInfo, ucontext, GuestAction, GuestStack, NewGuestSP, eflags);
|
||||
NewGuestSP = SetupRTFrame_ia32(Thread, ContextBackup, Frame, Signal, HostSigInfo, ucontext, GuestAction, GuestStack, NewGuestSP, eflags);
|
||||
} else {
|
||||
NewGuestSP = SetupFrame_ia32(ContextBackup, Frame, Signal, HostSigInfo, ucontext, GuestAction, GuestStack, NewGuestSP, eflags);
|
||||
NewGuestSP = SetupFrame_ia32(Thread, ContextBackup, Frame, Signal, HostSigInfo, ucontext, GuestAction, GuestStack, NewGuestSP, eflags);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -256,14 +256,14 @@ private:
|
||||
GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags);
|
||||
|
||||
///< Setup the signal frame for a 32-bit signal without SA_SIGINFO.
|
||||
uint64_t SetupFrame_ia32(ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame* Frame, int Signal,
|
||||
siginfo_t* HostSigInfo, void* ucontext, GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP,
|
||||
const uint32_t eflags);
|
||||
uint64_t SetupFrame_ia32(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* ContextBackup,
|
||||
FEXCore::Core::CpuStateFrame* Frame, int Signal, siginfo_t* HostSigInfo, void* ucontext,
|
||||
GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags);
|
||||
|
||||
///< Setup the signal frame for a 32-bit signal with SA_SIGINFO.
|
||||
uint64_t SetupRTFrame_ia32(ArchHelpers::Context::ContextBackup* ContextBackup, FEXCore::Core::CpuStateFrame* Frame, int Signal,
|
||||
siginfo_t* HostSigInfo, void* ucontext, GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP,
|
||||
const uint32_t eflags);
|
||||
uint64_t SetupRTFrame_ia32(FEXCore::Core::InternalThreadState* Thread, ArchHelpers::Context::ContextBackup* ContextBackup,
|
||||
FEXCore::Core::CpuStateFrame* Frame, int Signal, siginfo_t* HostSigInfo, void* ucontext,
|
||||
GuestSigAction* GuestAction, stack_t* GuestStack, uint64_t NewGuestSP, const uint32_t eflags);
|
||||
|
||||
enum class RestoreType {
|
||||
TYPE_REALTIME, ///< Signal restore type is from a `realtime` signal.
|
||||
|
||||
@@ -463,7 +463,7 @@ static void PrintFlags(uint64_t Flags) {
|
||||
|
||||
static uint64_t Clone2Handler(FEXCore::Core::CpuStateFrame* Frame, FEX::HLE::clone3_args* args) {
|
||||
StackFrameData* Data = (StackFrameData*)FEXCore::Allocator::malloc(sizeof(StackFrameData));
|
||||
Data->Thread = static_cast<FEX::HLE::ThreadStateObject*>(Frame->Thread->FrontendPtr);
|
||||
Data->Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
Data->CTX = Frame->Thread->CTX;
|
||||
Data->GuestArgs = *args;
|
||||
|
||||
@@ -489,7 +489,7 @@ static uint64_t Clone3Handler(FEXCore::Core::CpuStateFrame* Frame, FEX::HLE::clo
|
||||
constexpr size_t Offset = sizeof(StackFramePlusRet);
|
||||
StackFramePlusRet* Data = (StackFramePlusRet*)(reinterpret_cast<uint64_t>(args->NewStack) + args->StackSize - Offset);
|
||||
Data->Ret = (uint64_t)Clone3HandlerRet;
|
||||
Data->Data.Thread = static_cast<FEX::HLE::ThreadStateObject*>(Frame->Thread->FrontendPtr);
|
||||
Data->Data.Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
Data->Data.CTX = Frame->Thread->CTX;
|
||||
Data->Data.GuestArgs = *args;
|
||||
|
||||
|
||||
@@ -385,7 +385,7 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
// TLS/DTV teardown is something FEX can't control. Disable glibc checking when we leave a pthread.
|
||||
// Since this thread is hard stopping, we can't track the TLS/DTV teardown in FEX's thread handling.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator::HardDisable();
|
||||
auto ThreadObject = static_cast<FEX::HLE::ThreadStateObject*>(Thread->FrontendPtr);
|
||||
auto ThreadObject = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
if (Thread->ThreadManager.clear_child_tid) {
|
||||
std::atomic<uint32_t>* Addr = reinterpret_cast<std::atomic<uint32_t>*>(Thread->ThreadManager.clear_child_tid);
|
||||
|
||||
@@ -245,7 +245,14 @@ void SyscallHandler::VMATracking::ClearUnsafe(FEXCore::Context::Context* CTX, ui
|
||||
|
||||
// Change flags of mappings in a range and split the mappings if needed
|
||||
void SyscallHandler::VMATracking::ChangeUnsafe(uintptr_t Base, uintptr_t Length, VMAProt NewProt) {
|
||||
const auto Top = Base + Length;
|
||||
// This needs to handle multiple split-merge strategies:
|
||||
// 1) Exact overlap - No Split, no Merge. Only protection tracking changes.
|
||||
// 2) Exact base overlap - Single insert, can never fail.
|
||||
// 3) Insert in middle of VMA range. 1 or 2 inserts, can never fail.
|
||||
// 4) Partial overlapping merge. The most interesting strategy.
|
||||
// - More information below about this one.
|
||||
|
||||
auto Top = Base + Length;
|
||||
|
||||
// find the first Mapping at or after the Range ends, or ::end()
|
||||
// Top is the address after the end
|
||||
@@ -256,70 +263,213 @@ void SyscallHandler::VMATracking::ChangeUnsafe(uintptr_t Base, uintptr_t Length,
|
||||
MappingIter--;
|
||||
|
||||
auto Current = &MappingIter->second;
|
||||
const auto MapBase = Current->Base;
|
||||
const auto MapTop = MapBase + Current->Length;
|
||||
const auto MapFlags = Current->Flags;
|
||||
const auto MapProt = Current->Prot;
|
||||
|
||||
const auto OffsetDiff = Current->Offset - MapBase;
|
||||
|
||||
if (MapTop <= Base) {
|
||||
// Mapping ends before the Range start, exit
|
||||
if (Current->Base <= Base || Current->Base + Current->Length < Top) {
|
||||
break;
|
||||
} else if (MapProt.All == NewProt.All) {
|
||||
// Mapping already has the needed prots
|
||||
continue;
|
||||
} else {
|
||||
const bool HasFirstPart = MapBase < Base;
|
||||
const bool HasTrailingPart = MapTop > Top;
|
||||
}
|
||||
|
||||
if (HasFirstPart) {
|
||||
// Mapping starts before range, split first part
|
||||
const auto CurrentBase = Current->Base;
|
||||
const auto CurrentTop = CurrentBase + Current->Length;
|
||||
const auto CurrentFlags = Current->Flags;
|
||||
const auto CurrentProt = Current->Prot;
|
||||
|
||||
// Trim end of original mapping
|
||||
Current->Length = Base - MapBase;
|
||||
///< Resource mapping base.
|
||||
const auto OffsetDiff = Current->Offset - CurrentBase;
|
||||
|
||||
// Make new VMA with new flags, insert for length of range
|
||||
auto NewOffset = OffsetDiff + Base;
|
||||
auto NewLength = Top - Base;
|
||||
// Merge strategy 4)
|
||||
// CurrentBase range doesn't fully overlap the starting range but does overlap the tail.
|
||||
// This is the most confusing strategy as it requires splitting the protect range itself.
|
||||
//
|
||||
// if the VMA has tail data after the protection range we must first deal with that:
|
||||
// 1) Split the tail data in to new VMA range with original protections. Must not fail.
|
||||
// 2) Adjust the overlapping VMA protections to the new protections and the truncated length
|
||||
// 3) Truncate the mprotecting length and top to be that untouched range. Next loop will continue inserting.
|
||||
// [ Incoming Ranges ]
|
||||
// CurrentVMA: [CurrentBase ====== CurrentTop)
|
||||
// CurrentMProtectRange: [Base =============== Top)**********************
|
||||
// [ Modified Ranges ]
|
||||
// New Tail Range: [TailBase === Tail Top)
|
||||
// CurrentVMA Modified Range: [=======)
|
||||
// Remaining Tracking: [Base ==== NewTop)
|
||||
//
|
||||
// Next loop iterations will decompose the remaining mprotects in to more merge strategies.
|
||||
|
||||
auto [Iter, Inserted] =
|
||||
VMAs.emplace(Base, VMAEntry {Current->Resource, Current, Current->ResourceNextVMA, Base, NewOffset, NewLength, MapFlags, NewProt});
|
||||
LOGMAN_THROW_A_FMT(Inserted == true, "VMA tracking error");
|
||||
auto RestOfMapping = &Iter->second;
|
||||
// Steps:
|
||||
// 1) Split VMA if Top != CurrentTop
|
||||
// 2) Change [CurrentBase, Top) protections
|
||||
// 3) Change CurrentVMA length
|
||||
// 4) Adjust searching length for [Base, CurrentBase)
|
||||
const bool HasTailData = CurrentTop > Top;
|
||||
|
||||
if (Current->Resource) {
|
||||
ListInsertAfter(Current, RestOfMapping);
|
||||
}
|
||||
if (HasTailData) {
|
||||
// We now need to insert another VMA entry afterwards to ensure consistency.
|
||||
// This will have the original VMA's protection flags.
|
||||
|
||||
Current = RestOfMapping;
|
||||
} else {
|
||||
// Mapping starts in range, just change Prot
|
||||
Current->Prot = NewProt;
|
||||
// Make new VMA with new flags, insert for length of range
|
||||
auto NewOffset = OffsetDiff + CurrentBase;
|
||||
auto NewLength = CurrentTop - Top;
|
||||
|
||||
auto [Iter, Inserted] = VMAs.emplace(Top, VMAEntry {.Resource = Current->Resource,
|
||||
.ResourcePrevVMA = Current,
|
||||
.ResourceNextVMA = Current->ResourceNextVMA,
|
||||
.Base = Top,
|
||||
.Offset = NewOffset,
|
||||
.Length = NewLength,
|
||||
.Flags = CurrentFlags,
|
||||
.Prot = CurrentProt});
|
||||
|
||||
if (!Inserted) {
|
||||
// We can't recover from this.
|
||||
// Shouldn't ever happen.
|
||||
ERROR_AND_DIE_FMT("{}:{}: VMA tracking error", __func__, __LINE__);
|
||||
}
|
||||
|
||||
if (HasTrailingPart) {
|
||||
// ends after Range, split last part and insert with original flags
|
||||
|
||||
// Trim the mapping (possibly already trimmed)
|
||||
Current->Length = Top - Current->Base;
|
||||
|
||||
// prot has already been changed
|
||||
|
||||
// Make new VMA with original flags, insert for remaining length
|
||||
auto NewOffset = OffsetDiff + Top;
|
||||
auto NewLength = MapTop - Top;
|
||||
|
||||
auto [Iter, Inserted] =
|
||||
VMAs.emplace(Top, VMAEntry {Current->Resource, Current, Current->ResourceNextVMA, Top, NewOffset, NewLength, MapFlags, MapProt});
|
||||
LOGMAN_THROW_A_FMT(Inserted == true, "VMA tracking error");
|
||||
auto TrailingMapping = &Iter->second;
|
||||
|
||||
if (Current->Resource) {
|
||||
ListInsertAfter(Current, TrailingMapping);
|
||||
}
|
||||
if (Current->Resource) {
|
||||
ListInsertAfter(Current, &Iter->second);
|
||||
}
|
||||
}
|
||||
|
||||
// Change CurrentVMA's protections
|
||||
Current->Prot = NewProt;
|
||||
|
||||
// Change CurrentVMA's length
|
||||
Current->Length = Top - CurrentBase;
|
||||
|
||||
// Adjust the protection length we're searching for.
|
||||
// Next loop will pick up the next check.
|
||||
Length = CurrentBase - Base;
|
||||
Top = Base + Length;
|
||||
}
|
||||
|
||||
auto Current = &MappingIter->second;
|
||||
const auto CurrentBase = Current->Base;
|
||||
const auto CurrentTop = CurrentBase + Current->Length;
|
||||
const auto CurrentFlags = Current->Flags;
|
||||
const auto CurrentProt = Current->Prot;
|
||||
|
||||
///< Resource mapping base.
|
||||
const auto OffsetDiff = Current->Offset - CurrentBase;
|
||||
if (CurrentTop <= Base) {
|
||||
// Mapping is below what we care about
|
||||
// [CurrentBase === CurrentTop)
|
||||
// [Base === Top)
|
||||
} else if (CurrentBase == Base && CurrentTop == Top) {
|
||||
// Merge strategy 1)
|
||||
// Exact encompassing, quite common.
|
||||
// [CurrentBase ======================== CurrentTop)
|
||||
// [Base ====================================== Top)
|
||||
Current->Prot = NewProt;
|
||||
} else if (CurrentBase == Base && CurrentTop > Top) {
|
||||
// Merge strategy 2)
|
||||
// [CurrentBase ======================== CurrentTop)
|
||||
// [Base =============== Top)***********************
|
||||
// VMA fully encompasses with matching base.
|
||||
// VMA needs to split.
|
||||
|
||||
// Steps:
|
||||
// 1) Set new permissions for this VMA
|
||||
// 2) Trim VMA->Length to match [CurrentBase, CurrentBase+Length)
|
||||
// 2) Insert new node at [CurrentBase+Length, CurrentTop)
|
||||
|
||||
// 1) Set new permissions
|
||||
Current->Prot = NewProt;
|
||||
|
||||
// Trim end of original mapping
|
||||
// New length for Current VMA is Top - CurrentBase
|
||||
Current->Length = Top - CurrentBase;
|
||||
|
||||
// Make new VMA with original protections, insert for remaining length
|
||||
auto NewOffset = OffsetDiff + Top;
|
||||
auto NewLength = CurrentTop - Top;
|
||||
|
||||
auto [Iter, Inserted] = VMAs.emplace(Top, VMAEntry {.Resource = Current->Resource,
|
||||
.ResourcePrevVMA = Current,
|
||||
.ResourceNextVMA = Current->ResourceNextVMA,
|
||||
.Base = Top,
|
||||
.Offset = NewOffset,
|
||||
.Length = NewLength,
|
||||
.Flags = CurrentFlags,
|
||||
.Prot = CurrentProt});
|
||||
|
||||
if (!Inserted) [[unlikely]] {
|
||||
// We can't recover from this.
|
||||
// Shouldn't ever happen.
|
||||
ERROR_AND_DIE_FMT("{}:{}: VMA tracking error", __func__, __LINE__);
|
||||
}
|
||||
|
||||
if (Current->Resource) {
|
||||
ListInsertAfter(Current, &Iter->second);
|
||||
}
|
||||
} else if (CurrentBase < Base && CurrentTop >= Top) {
|
||||
// Merge strategy 3)
|
||||
// VMA fully encompasses, VMA needs to split.
|
||||
// Explicitly VMA base doesn't match current base.
|
||||
// [CurrentBase ======================== CurrentTop)
|
||||
// ***************[Base =============== Top)********
|
||||
|
||||
// Steps:
|
||||
// 1) Split the CurrentVMA
|
||||
// 2) Set new length of CurrentVMA
|
||||
// 3) If there is tail length still, Insert another new VMA with CurrentVMA data.
|
||||
|
||||
const bool HasTailData = CurrentTop > Top;
|
||||
|
||||
// Trim end of original mapping
|
||||
Current->Length = Base - CurrentBase;
|
||||
{
|
||||
// Make new VMA with new flags, insert for length of range
|
||||
auto NewOffset = OffsetDiff + Base;
|
||||
auto NewLength = Top - Base;
|
||||
|
||||
auto [Iter, Inserted] = VMAs.emplace(Base, VMAEntry {.Resource = Current->Resource,
|
||||
.ResourcePrevVMA = Current,
|
||||
.ResourceNextVMA = Current->ResourceNextVMA,
|
||||
.Base = Base,
|
||||
.Offset = NewOffset,
|
||||
.Length = NewLength,
|
||||
.Flags = CurrentFlags,
|
||||
.Prot = NewProt});
|
||||
|
||||
if (!Inserted) [[unlikely]] {
|
||||
// We can't recover from this.
|
||||
// Shouldn't ever happen.
|
||||
ERROR_AND_DIE_FMT("{}:{}: VMA tracking error", __func__, __LINE__);
|
||||
}
|
||||
|
||||
if (Current->Resource) {
|
||||
ListInsertAfter(Current, &Iter->second);
|
||||
}
|
||||
}
|
||||
|
||||
if (HasTailData) {
|
||||
// We now need to insert another VMA entry afterwards to ensure consistency.
|
||||
// This will have the original VMA's protection flags.
|
||||
|
||||
// Make new VMA with new flags, insert for length of range
|
||||
auto NewOffset = OffsetDiff + Top;
|
||||
auto NewLength = CurrentTop - Top;
|
||||
|
||||
auto [Iter, Inserted] = VMAs.emplace(Top, VMAEntry {.Resource = Current->Resource,
|
||||
.ResourcePrevVMA = Current,
|
||||
.ResourceNextVMA = Current->ResourceNextVMA,
|
||||
.Base = Top,
|
||||
.Offset = NewOffset,
|
||||
.Length = NewLength,
|
||||
.Flags = CurrentFlags,
|
||||
.Prot = CurrentProt});
|
||||
|
||||
if (!Inserted) {
|
||||
// We can't recover from this.
|
||||
// Shouldn't ever happen.
|
||||
ERROR_AND_DIE_FMT("{}:{}: VMA tracking error", __func__, __LINE__);
|
||||
}
|
||||
|
||||
if (Current->Resource) {
|
||||
ListInsertAfter(Current, &Iter->second);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
ERROR_AND_DIE_FMT("Unexpected {} Merge strategy! [0x{:x}, 0x{:x}) Versus [0x{:x}, 0x{:x})\n", __func__, CurrentBase, CurrentTop, Base, Top);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -217,8 +217,8 @@ void ThreadManager::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThre
|
||||
// Remove all threads but the live thread from Threads
|
||||
Threads.clear();
|
||||
|
||||
auto LiveThreadData = static_cast<FEX::HLE::ThreadStateObject*>(LiveThread->FrontendPtr);
|
||||
Threads.push_back(LiveThreadData);
|
||||
auto ThreadObject = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(LiveThread->CurrentFrame);
|
||||
Threads.push_back(ThreadObject);
|
||||
|
||||
// Clean up dead stacks
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
|
||||
@@ -30,6 +30,11 @@ public:
|
||||
|
||||
~ThreadManager();
|
||||
|
||||
///< Returns the ThreadStateObject from a CpuStateFrame object.
|
||||
static inline FEX::HLE::ThreadStateObject* GetStateObjectFromCPUState(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
return static_cast<FEX::HLE::ThreadStateObject*>(Frame->Thread->FrontendPtr);
|
||||
}
|
||||
|
||||
FEX::HLE::ThreadStateObject*
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::Core::CPUState* NewThreadState = nullptr, uint64_t ParentTID = 0);
|
||||
void TrackThread(FEX::HLE::ThreadStateObject* Thread) {
|
||||
|
||||
@@ -242,7 +242,6 @@ int main(int argc, char** argv, char** const envp) {
|
||||
#endif
|
||||
|
||||
bool SupportsAVX = false;
|
||||
bool SupportsAVX2 = false;
|
||||
FEXCore::Core::CPUState State;
|
||||
|
||||
FEXCore::Context::InitializeStaticTables(Loader.Is64BitMode() ? FEXCore::Context::MODE_64BIT : FEXCore::Context::MODE_32BIT);
|
||||
@@ -262,13 +261,12 @@ int main(int argc, char** argv, char** const envp) {
|
||||
// Skip any tests that the host doesn't support features for
|
||||
auto HostFeatures = CTX->GetHostFeatures();
|
||||
SupportsAVX = HostFeatures.SupportsAVX;
|
||||
SupportsAVX2 = HostFeatures.SupportsAVX2;
|
||||
|
||||
bool TestUnsupported = (!HostFeatures.Supports3DNow && Loader.Requires3DNow()) || (!HostFeatures.SupportsSSE4A && Loader.RequiresSSE4A()) ||
|
||||
(!SupportsAVX && Loader.RequiresAVX()) || (!SupportsAVX2 && Loader.RequiresAVX2()) ||
|
||||
(!HostFeatures.SupportsRAND && Loader.RequiresRAND()) || (!HostFeatures.SupportsSHA && Loader.RequiresSHA()) ||
|
||||
(!HostFeatures.SupportsCLZERO && Loader.RequiresCLZERO()) || (!HostFeatures.SupportsBMI1 && Loader.RequiresBMI1()) ||
|
||||
(!HostFeatures.SupportsBMI2 && Loader.RequiresBMI2()) || (!HostFeatures.SupportsCLWB && Loader.RequiresCLWB());
|
||||
(!SupportsAVX && Loader.RequiresAVX()) || (!HostFeatures.SupportsRAND && Loader.RequiresRAND()) ||
|
||||
(!HostFeatures.SupportsSHA && Loader.RequiresSHA()) || (!HostFeatures.SupportsCLZERO && Loader.RequiresCLZERO()) ||
|
||||
(!HostFeatures.SupportsBMI1 && Loader.RequiresBMI1()) || (!HostFeatures.SupportsBMI2 && Loader.RequiresBMI2()) ||
|
||||
(!HostFeatures.SupportsCLWB && Loader.RequiresCLWB()) || (!HostFeatures.SupportsAES256 && Loader.RequiresAES256());
|
||||
|
||||
#ifdef _WIN32
|
||||
TestUnsupported |= Loader.RequiresLinux();
|
||||
@@ -318,6 +316,19 @@ int main(int argc, char** argv, char** const envp) {
|
||||
// Just re-use compare state. It also checks against the expected values in config.
|
||||
memcpy(&State, &ParentThread->Thread->CurrentFrame->State, sizeof(State));
|
||||
|
||||
__uint128_t XMM_Low[FEXCore::Core::CPUState::NUM_XMMS];
|
||||
if (SupportsAVX) {
|
||||
///< Reconstruct the XMM registers even if they are in split view, then remerge them.
|
||||
__uint128_t YMM_High[FEXCore::Core::CPUState::NUM_XMMS];
|
||||
CTX->ReconstructXMMRegisters(ParentThread->Thread, XMM_Low, YMM_High);
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
memcpy(&State.xmm.avx.data[i][0], &XMM_Low[i], sizeof(__uint128_t));
|
||||
memcpy(&State.xmm.avx.data[i][2], &YMM_High[i], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
CTX->ReconstructXMMRegisters(ParentThread->Thread, reinterpret_cast<__uint128_t*>(State.xmm.sse.data), nullptr);
|
||||
}
|
||||
|
||||
SignalDelegation->UninstallTLSState(ParentThread);
|
||||
FEX::HLE::_SyscallHandler->TM.DestroyThread(ParentThread, true);
|
||||
|
||||
|
||||
@@ -9,4 +9,4 @@ EXPORTS
|
||||
NtGetContextThread
|
||||
NtContinue
|
||||
__wine_dbg_output
|
||||
__wine_unix_call
|
||||
__wine_unix_call_dispatcher DATA
|
||||
@@ -159,7 +159,7 @@ void LoadStateFromWowContext(FEXCore::Core::InternalThreadState* Thread, uint64_
|
||||
// Floating-point register state
|
||||
const auto* XSave = reinterpret_cast<XSAVE_FORMAT*>(Context->ExtendedRegisters);
|
||||
|
||||
memcpy(State.xmm.sse.data, XSave->XmmRegisters, sizeof(State.xmm.sse.data));
|
||||
CTX->SetXMMRegistersFromState(Thread, reinterpret_cast<const __uint128_t*>(XSave->XmmRegisters), nullptr);
|
||||
memcpy(State.mm, XSave->FloatRegisters, sizeof(State.mm));
|
||||
|
||||
State.FCW = XSave->ControlWord;
|
||||
@@ -199,7 +199,7 @@ void StoreWowContextFromState(FEXCore::Core::InternalThreadState* Thread, WOW64_
|
||||
|
||||
auto* XSave = reinterpret_cast<XSAVE_FORMAT*>(Context->ExtendedRegisters);
|
||||
|
||||
memcpy(XSave->XmmRegisters, State.xmm.sse.data, sizeof(State.xmm.sse.data));
|
||||
CTX->ReconstructXMMRegisters(Thread, reinterpret_cast<__uint128_t*>(XSave->XmmRegisters), nullptr);
|
||||
memcpy(XSave->FloatRegisters, State.mm, sizeof(State.mm));
|
||||
|
||||
XSave->ControlWord = State.FCW;
|
||||
|
||||
@@ -11,7 +11,11 @@ extern "C" {
|
||||
|
||||
typedef UINT64 unixlib_handle_t;
|
||||
|
||||
NTSTATUS WINAPI __wine_unix_call(unixlib_handle_t handle, unsigned int code, void* args);
|
||||
extern NTSTATUS(WINAPI* __wine_unix_call_dispatcher)(unixlib_handle_t, unsigned int, void*);
|
||||
|
||||
static inline NTSTATUS __wine_unix_call(unixlib_handle_t handle, unsigned int code, void* args) {
|
||||
return __wine_unix_call_dispatcher(handle, code, args);
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# FEX-2406
|
||||
# FEX-2407
|
||||
|
||||
## FEXCore
|
||||
See [FEXCore/Readme.md](../FEXCore/Readme.md) for more details
|
||||
@@ -58,6 +58,7 @@ Metadata that drives the frontend x86/64 decoding
|
||||
- [X86Tables.cpp](../FEXCore/Source/Interface/Core/X86Tables.cpp)
|
||||
|
||||
#### x86-to-ir
|
||||
- [AVX_128.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp): Handles x86/64 AVX instructions to 128-bit IR
|
||||
- [Crypto.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp): Handles x86/64 Crypto instructions to IR
|
||||
- [Flags.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp): Handles x86/64 flag generation
|
||||
- [Vector.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp): Handles x86/64 Vector instructions to IR
|
||||
@@ -124,6 +125,7 @@ IR to IR Optimization
|
||||
- [CPUID.cpp](../FEXCore/Source/Interface/Core/CPUID.cpp): Handles presented capability bits for guest cpu
|
||||
|
||||
#### dispatcher-implementations
|
||||
- [AVX_128.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp): Handles x86/64 AVX instructions to 128-bit IR
|
||||
- [Crypto.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp): Handles x86/64 Crypto instructions to IR
|
||||
- [Flags.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp): Handles x86/64 flag generation
|
||||
- [Vector.cpp](../FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp): Handles x86/64 Vector instructions to IR
|
||||
|
||||
@@ -110,13 +110,10 @@ endforeach()
|
||||
add_custom_target(32bit_asm_files ALL
|
||||
DEPENDS "${ASM_DEPENDS}")
|
||||
|
||||
execute_process(COMMAND "nproc" OUTPUT_VARIABLE CORES)
|
||||
string(STRIP ${CORES} CORES)
|
||||
|
||||
add_custom_target(
|
||||
32bit_asm_tests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}"
|
||||
USES_TERMINAL
|
||||
DEPENDS 32bit_asm_files
|
||||
DEPENDS "${CMAKE_BINARY_DIR}/Bin/TestHarnessRunner"
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" "-j${CORES}" "-R" "\.*32Bit\.*.asm$$")
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" ${TEST_JOB_FLAG} "-R" "\.*32Bit\.*.asm$$")
|
||||
@@ -14,14 +14,11 @@ foreach(API_TEST ${TESTS})
|
||||
TEST_SUFFIX ".${API_TEST}.APITest")
|
||||
endforeach()
|
||||
|
||||
execute_process(COMMAND "nproc" OUTPUT_VARIABLE CORES)
|
||||
string(STRIP ${CORES} CORES)
|
||||
|
||||
add_custom_target(
|
||||
api_tests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}"
|
||||
USES_TERMINAL
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" "-j${CORES}" "-R" "\.*.APITest")
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" ${TEST_JOB_FLAG} "-R" "\.*.APITest")
|
||||
|
||||
foreach(API_TEST ${TESTS})
|
||||
add_dependencies(api_tests ${API_TEST})
|
||||
|
||||
@@ -117,16 +117,13 @@ endforeach()
|
||||
add_custom_target(asm_files ALL
|
||||
DEPENDS "${ASM_DEPENDS}")
|
||||
|
||||
execute_process(COMMAND "nproc" OUTPUT_VARIABLE CORES)
|
||||
string(STRIP ${CORES} CORES)
|
||||
|
||||
add_custom_target(
|
||||
64bit_asm_tests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}"
|
||||
USES_TERMINAL
|
||||
DEPENDS asm_files
|
||||
DEPENDS "${CMAKE_BINARY_DIR}/Bin/TestHarnessRunner"
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" "-j${CORES}" "-R" "\.*64Bit\.*.asm$$")
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" ${TEST_JOB_FLAG} "-R" "\.*64Bit\.*.asm$$")
|
||||
|
||||
add_custom_target(
|
||||
asm_tests
|
||||
@@ -135,4 +132,4 @@ add_custom_target(
|
||||
DEPENDS asm_files
|
||||
DEPENDS 32bit_asm_files
|
||||
DEPENDS "${CMAKE_BINARY_DIR}/Bin/TestHarnessRunner"
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" "-j${CORES}" "-R" "\.*.asm$$")
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" ${TEST_JOB_FLAG} "-R" "\.*.asm$$")
|
||||
@@ -10,11 +10,16 @@ Test_VEX/vaesdeclast.asm
|
||||
Test_VEX/vaesenc.asm
|
||||
Test_VEX/vaesenclast.asm
|
||||
Test_VEX/vaesimc.asm
|
||||
Test_VEX/vaesdec256.asm
|
||||
Test_VEX/vaesdeclast256.asm
|
||||
Test_VEX/vaesenc256.asm
|
||||
Test_VEX/vaesenclast256.asm
|
||||
Test_VEX/vaeskeygenassist.asm
|
||||
|
||||
# PCMUL considered to be part of crypto operations. Simulator doesn't support this.
|
||||
Test_H0F3A/pclmulqdq.asm
|
||||
Test_VEX/vpclmulqdq.asm
|
||||
Test_VEX/vpclmulqdq_256.asm
|
||||
|
||||
# Simulator can't handle self-modifying code
|
||||
Test_SelfModifyingCode/Delinking.asm
|
||||
@@ -83,6 +88,14 @@ Test_VEX/vroundpd.asm
|
||||
Test_VEX/vroundps.asm
|
||||
Test_VEX/vroundsd.asm
|
||||
Test_VEX/vroundss.asm
|
||||
Test_VEX/vcvtps2ph_rtne.asm
|
||||
Test_VEX/vcvtps2ph_rd.asm
|
||||
Test_VEX/vcvtps2ph_ru.asm
|
||||
Test_VEX/vcvtps2ph_trunc.asm
|
||||
Test_VEX/vcvtps2ph_rtne_mxcsr.asm
|
||||
Test_VEX/vcvtps2ph_rd_mxcsr.asm
|
||||
Test_VEX/vcvtps2ph_ru_mxcsr.asm
|
||||
Test_VEX/vcvtps2ph_trunc_mxcsr.asm
|
||||
|
||||
# Simulator doesn't support cycle counter reading
|
||||
Test_TwoByte/0F_31.asm
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0",
|
||||
"RCX": "0x5a"
|
||||
},
|
||||
"HostFeatures": ["BMI1"]
|
||||
}
|
||||
%endif
|
||||
|
||||
mov rcx, 0x8f635a775ad3b9b4
|
||||
mov esi, 0x3018
|
||||
bextr ecx, ecx, esi
|
||||
cmp rcx, 0x5a
|
||||
jne .bad
|
||||
|
||||
.good:
|
||||
mov rax, 0
|
||||
hlt
|
||||
|
||||
.bad:
|
||||
mov rax, 1
|
||||
hlt
|
||||
@@ -0,0 +1,24 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "1"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX had a bug with relative call instructions.
|
||||
; It was incorrectly truncating the immediate displacement based on address size override AND operand size override.
|
||||
; Address size override doesn't actually change immediate representation on the call instruction.
|
||||
|
||||
mov rsp, 0xe000_1000
|
||||
mov rax, 0
|
||||
|
||||
jmp .after
|
||||
.test:
|
||||
mov rax, 1
|
||||
hlt
|
||||
|
||||
.after:
|
||||
a32 call .test
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,381 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0x4b497b9e152430ec", "0x019f45087baf8cb8"],
|
||||
"XMM1": ["0x089f12645cb5e036", "0x5a6af6f5102c523c"],
|
||||
"XMM2": ["0x4c619a6f28bed383", "0x6892c52557512e58"],
|
||||
"XMM3": ["0x8ee99e09628ebdc3", "0xa7688af8254ea454"],
|
||||
"XMM4": ["0x805080d92966f25a", "0x31f967965d3a07cb"],
|
||||
"XMM5": ["0x2828cb0ce87848be", "0xc3291045169390b4"],
|
||||
"XMM6": ["0x755ae99230a898c3", "0x3d209d2dd4bad59f"],
|
||||
"XMM7": ["0x3a670269bb42b2f8", "0x05173dbeda9e86ab"],
|
||||
"XMM8": ["0x275bf419e2f3b099", "0x276d21a284ab2912"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX-Emu had a bug where we were conflating x87 registers as mmx registers and vice-versa depending on caching behaviour.
|
||||
; This unittest semi-aggressively mixes x87 and mmx with xsave/xrstor that would have failed with FEX's caching.
|
||||
|
||||
fninit ; Initialize x87
|
||||
|
||||
; Load all x87 registers
|
||||
fld tword [rel .random_data + (0 * 10)]
|
||||
fld tword [rel .random_data + (1 * 10)]
|
||||
fld tword [rel .random_data + (2 * 10)]
|
||||
fld tword [rel .random_data + (3 * 10)]
|
||||
fld tword [rel .random_data + (4 * 10)]
|
||||
fld tword [rel .random_data + (5 * 10)]
|
||||
fld tword [rel .random_data + (6 * 10)]
|
||||
fld tword [rel .random_data + (7 * 10)]
|
||||
|
||||
; Save the data based on bits in EDX:EAX
|
||||
; Just save everything
|
||||
mov edx, -1
|
||||
mov eax, -1
|
||||
xsave64 [rel .xsave_data]
|
||||
|
||||
; Load all MMX registers (Data just past what x87 loaded.
|
||||
movq mm0, [rel .random_data + (8 * 10) + (0 * 8)]
|
||||
movq mm1, [rel .random_data + (8 * 10) + (1 * 8)]
|
||||
movq mm2, [rel .random_data + (8 * 10) + (2 * 8)]
|
||||
movq mm3, [rel .random_data + (8 * 10) + (3 * 8)]
|
||||
movq mm4, [rel .random_data + (8 * 10) + (4 * 8)]
|
||||
movq mm5, [rel .random_data + (8 * 10) + (5 * 8)]
|
||||
movq mm6, [rel .random_data + (8 * 10) + (6 * 8)]
|
||||
movq mm7, [rel .random_data + (8 * 10) + (7 * 8)]
|
||||
|
||||
; Do some operation on the MMX registers
|
||||
pxor mm0, mm1
|
||||
pxor mm1, mm2
|
||||
pxor mm2, mm3
|
||||
pxor mm3, mm4
|
||||
pxor mm4, mm5
|
||||
pxor mm5, mm6
|
||||
pxor mm6, mm7
|
||||
pxor mm7, mm0
|
||||
|
||||
; Store MMX registers
|
||||
movq [rel .temp_result + (0 * 8)], mm0
|
||||
movq [rel .temp_result + (1 * 8)], mm1
|
||||
movq [rel .temp_result + (2 * 8)], mm2
|
||||
movq [rel .temp_result + (3 * 8)], mm3
|
||||
movq [rel .temp_result + (4 * 8)], mm4
|
||||
movq [rel .temp_result + (5 * 8)], mm5
|
||||
movq [rel .temp_result + (6 * 8)], mm6
|
||||
movq [rel .temp_result + (7 * 8)], mm7
|
||||
|
||||
; Clear MMX state
|
||||
emms
|
||||
|
||||
; Load all x87 registers with new data
|
||||
; This ensures the top 16-bits of every x87 word is different.
|
||||
fld tword [rel .random_data + (8 * 10)]
|
||||
fld tword [rel .random_data + (9 * 10)]
|
||||
fld tword [rel .random_data + (10 * 10)]
|
||||
fld tword [rel .random_data + (11 * 10)]
|
||||
fld tword [rel .random_data + (12 * 10)]
|
||||
fld tword [rel .random_data + (13 * 10)]
|
||||
fld tword [rel .random_data + (14 * 10)]
|
||||
fld tword [rel .random_data + (15 * 10)]
|
||||
|
||||
; Reload context, including original x87 state
|
||||
mov edx, -1
|
||||
mov eax, -1
|
||||
xrstor64 [rel .xsave_data]
|
||||
|
||||
; Save the x87 registers.
|
||||
fstp tword [rel .temp_x87_result + (0 * 10)]
|
||||
fstp tword [rel .temp_x87_result + (1 * 10)]
|
||||
fstp tword [rel .temp_x87_result + (2 * 10)]
|
||||
fstp tword [rel .temp_x87_result + (3 * 10)]
|
||||
fstp tword [rel .temp_x87_result + (4 * 10)]
|
||||
fstp tword [rel .temp_x87_result + (5 * 10)]
|
||||
fstp tword [rel .temp_x87_result + (6 * 10)]
|
||||
fstp tword [rel .temp_x87_result + (7 * 10)]
|
||||
|
||||
; Load the results in to XMM registers
|
||||
; First load the MMX registers
|
||||
movups xmm0, [rel .temp_result + (0 * 16)]
|
||||
movups xmm1, [rel .temp_result + (1 * 16)]
|
||||
movups xmm2, [rel .temp_result + (2 * 16)]
|
||||
movups xmm3, [rel .temp_result + (3 * 16)]
|
||||
|
||||
; Now load the 80 bytes of x87 registers
|
||||
movups xmm4, [rel .temp_x87_result + (0 * 16)]
|
||||
movups xmm5, [rel .temp_x87_result + (1 * 16)]
|
||||
movups xmm6, [rel .temp_x87_result + (2 * 16)]
|
||||
movups xmm7, [rel .temp_x87_result + (3 * 16)]
|
||||
movups xmm8, [rel .temp_x87_result + (4 * 16)]
|
||||
|
||||
hlt
|
||||
align 32
|
||||
|
||||
.temp_x87_result:
|
||||
times (16 * 8) db 0
|
||||
|
||||
.temp_result:
|
||||
times (8 * 8) db 0
|
||||
|
||||
; 4096 bytes of random data.
|
||||
.random_data:
|
||||
db 0x5b, 0x27, 0x12, 0x29, 0xab, 0x84, 0xa2, 0x21, 0x6d, 0x27, 0xbe, 0x3d, 0x17, 0x05, 0x99, 0xb0
|
||||
db 0xf3, 0xe2, 0x19, 0xf4, 0x42, 0xbb, 0x69, 0x02, 0x67, 0x3a, 0xab, 0x86, 0x9e, 0xda, 0x9f, 0xd5
|
||||
db 0xba, 0xd4, 0x2d, 0x9d, 0x20, 0x3d, 0xf8, 0xb2, 0x29, 0xc3, 0xc3, 0x98, 0xa8, 0x30, 0x92, 0xe9
|
||||
db 0x5a, 0x75, 0x0c, 0xcb, 0x28, 0x28, 0xb4, 0x90, 0x93, 0x16, 0x45, 0x10, 0x3a, 0x5d, 0x96, 0x67
|
||||
db 0xf9, 0x31, 0xbe, 0x48, 0x78, 0xe8, 0x5a, 0xf2, 0x66, 0x29, 0xd9, 0x80, 0x50, 0x80, 0xcb, 0x07
|
||||
db 0xfe, 0xda, 0x19, 0x0f, 0x22, 0xea, 0x18, 0x5e, 0x12, 0xea, 0x3d, 0x1a, 0xbc, 0x91, 0x51, 0x15
|
||||
db 0xaa, 0x66, 0x92, 0x61, 0xb4, 0xd4, 0xce, 0x14, 0x9c, 0x86, 0x27, 0x3d, 0xd0, 0xc6, 0x51, 0x1c
|
||||
db 0xa0, 0xd4, 0x0b, 0x2d, 0x25, 0x30, 0x3b, 0x46, 0x23, 0x07, 0xb5, 0x05, 0x4a, 0xaa, 0x5a, 0x0a
|
||||
db 0x7b, 0x29, 0xe4, 0x52, 0x6f, 0x6f, 0xc8, 0x62, 0xb8, 0x94, 0x6a, 0x30, 0x66, 0xf1, 0x21, 0xec
|
||||
db 0xd1, 0xf2, 0x68, 0xda, 0xb7, 0x7f, 0x5a, 0x26, 0x38, 0x46, 0x48, 0xda, 0x5d, 0x64, 0x8d, 0x3d
|
||||
db 0x2f, 0xf6, 0xc3, 0x63, 0xb8, 0x09, 0x3a, 0xd0, 0x5b, 0xeb, 0x67, 0xd0, 0xaa, 0x63, 0x71, 0x19
|
||||
db 0x7e, 0x4e, 0x33, 0xe2, 0x15, 0xba, 0x87, 0xa7, 0x7b, 0x25, 0xe4, 0xbb, 0xb5, 0x26, 0x9a, 0xf1
|
||||
db 0xdd, 0x5a, 0x63, 0xd7, 0x16, 0xc0, 0xc3, 0xc8, 0x1b, 0xad, 0x00, 0x52, 0x63, 0x55, 0xc7, 0xe0
|
||||
db 0xd9, 0xe9, 0xf4, 0x4c, 0x53, 0xfb, 0x73, 0x57, 0xdc, 0xad, 0x0c, 0xca, 0x73, 0x44, 0x6b, 0xf3
|
||||
db 0xb7, 0x83, 0x3b, 0xfe, 0xf0, 0x15, 0xbf, 0xe5, 0x15, 0xca, 0xdf, 0x35, 0xeb, 0xe7, 0xe3, 0xa2
|
||||
db 0xbd, 0x20, 0xad, 0xff, 0x1b, 0x67, 0x0a, 0x9f, 0x60, 0x60, 0xff, 0xa7, 0xc9, 0x19, 0xde, 0xb3
|
||||
db 0x67, 0xf1, 0x4b, 0x77, 0x7f, 0x0b, 0xb1, 0x29, 0xee, 0xcb, 0xd6, 0x5d, 0x0d, 0xb9, 0x54, 0x49
|
||||
db 0x10, 0xe3, 0xbd, 0x8a, 0xa0, 0x69, 0xa3, 0x07, 0xbe, 0x8e, 0xea, 0xc6, 0x75, 0x27, 0x66, 0xae
|
||||
db 0x3c, 0xde, 0xc6, 0x13, 0x1b, 0x50, 0x37, 0x56, 0x7c, 0x01, 0xab, 0x8b, 0x46, 0xdc, 0x80, 0xed
|
||||
db 0xdf, 0x12, 0x6f, 0x64, 0xdf, 0xe6, 0xf9, 0xbf, 0x15, 0x95, 0xd9, 0x80, 0x19, 0x8c, 0x96, 0x33
|
||||
db 0x89, 0xbe, 0x25, 0x33, 0x34, 0x82, 0x92, 0x96, 0x05, 0x52, 0xa2, 0xcf, 0x5b, 0x3d, 0xfc, 0xd8
|
||||
db 0x43, 0x89, 0x2e, 0x16, 0x6d, 0xbd, 0x84, 0x97, 0x77, 0xb5, 0xd6, 0x2b, 0x6b, 0xb1, 0xc6, 0x38
|
||||
db 0x0a, 0xfe, 0xe1, 0xc9, 0x31, 0x32, 0x7f, 0xd5, 0xc1, 0x03, 0x4a, 0xb2, 0x86, 0x4d, 0x8d, 0x77
|
||||
db 0xd6, 0x62, 0x52, 0x75, 0xed, 0x27, 0x21, 0xe8, 0x69, 0x6f, 0x6a, 0x5b, 0x59, 0x4d, 0xd2, 0x6c
|
||||
db 0x2a, 0x97, 0x09, 0x03, 0xc5, 0x29, 0x0d, 0xe1, 0x31, 0x2e, 0x62, 0x21, 0x0e, 0xc2, 0x00, 0x7c
|
||||
db 0xa2, 0x4c, 0x19, 0x63, 0x24, 0xfc, 0x9b, 0x38, 0x11, 0xbf, 0x20, 0x53, 0x53, 0xac, 0x3f, 0xdb
|
||||
db 0xfd, 0x2b, 0x39, 0x3c, 0x39, 0x6b, 0xb4, 0x52, 0x1f, 0xf8, 0x8f, 0x3b, 0x47, 0x2b, 0x86, 0xcf
|
||||
db 0xd2, 0x38, 0xe9, 0x08, 0x73, 0x09, 0x32, 0x5f, 0x6c, 0x3a, 0xdb, 0xfc, 0x1d, 0x91, 0xa4, 0x26
|
||||
db 0xa3, 0x0c, 0xbc, 0x94, 0xf5, 0xbd, 0x29, 0xcf, 0x72, 0x3d, 0xee, 0x48, 0x06, 0x77, 0x63, 0x70
|
||||
db 0x47, 0xc9, 0x87, 0x21, 0xb1, 0x9a, 0xdd, 0x5f, 0x71, 0x08, 0xe3, 0x3b, 0xf6, 0x07, 0x9f, 0x2f
|
||||
db 0x20, 0xa3, 0x02, 0xc8, 0x4d, 0xc8, 0x18, 0xfa, 0x69, 0x32, 0x60, 0x97, 0x2d, 0x2f, 0x26, 0x84
|
||||
db 0x3d, 0x7a, 0xf6, 0x2f, 0xb1, 0xc9, 0xd2, 0xcd, 0x6e, 0x24, 0x18, 0xa8, 0x0d, 0xb0, 0xe2, 0x41
|
||||
db 0x1e, 0xdf, 0xc7, 0xee, 0xcd, 0x21, 0x5b, 0xc3, 0x26, 0x26, 0xb3, 0xb4, 0x33, 0x58, 0x79, 0xb5
|
||||
db 0xc3, 0x24, 0x7c, 0xe3, 0xd7, 0x78, 0x33, 0x22, 0xd5, 0x20, 0x21, 0x86, 0xcf, 0xca, 0x44, 0xba
|
||||
db 0xd8, 0x05, 0x84, 0x37, 0x69, 0x48, 0xb0, 0xe0, 0x7a, 0xe6, 0x74, 0x53, 0x1e, 0xd0, 0x0c, 0x3c
|
||||
db 0x33, 0x83, 0x15, 0x43, 0x16, 0x0e, 0x93, 0x39, 0x55, 0x2e, 0x55, 0x1c, 0x09, 0xbd, 0x7a, 0xc3
|
||||
db 0x80, 0x77, 0x4e, 0xd9, 0xf3, 0xa5, 0xee, 0x94, 0xbf, 0x8e, 0xd0, 0xec, 0x39, 0x33, 0x31, 0x8d
|
||||
db 0x74, 0x94, 0xd2, 0x24, 0x22, 0x4a, 0xde, 0x51, 0x99, 0xc5, 0x68, 0xf2, 0x2e, 0xd3, 0x8d, 0xc5
|
||||
db 0x32, 0x31, 0x26, 0xe7, 0x87, 0x47, 0x5f, 0xbc, 0x32, 0x80, 0x43, 0x83, 0x34, 0x36, 0xa1, 0x72
|
||||
db 0x6b, 0x38, 0x10, 0x93, 0xa7, 0xa3, 0x92, 0xb7, 0x3c, 0x61, 0x1c, 0x4e, 0x0b, 0x86, 0x43, 0xa9
|
||||
db 0x64, 0xf1, 0xf8, 0xd7, 0xd3, 0xf4, 0xd0, 0xe2, 0x17, 0xd4, 0xbb, 0xe9, 0x2c, 0xc8, 0x76, 0xc5
|
||||
db 0x87, 0x7f, 0x81, 0x55, 0xbe, 0x87, 0x0e, 0x6b, 0xf6, 0x4f, 0x44, 0x37, 0x92, 0x32, 0x7f, 0x30
|
||||
db 0xa6, 0x66, 0x09, 0x01, 0x7a, 0x6e, 0xb3, 0x3b, 0x7d, 0x8f, 0x32, 0x0e, 0x3c, 0xdc, 0xba, 0x2e
|
||||
db 0xf8, 0xec, 0xde, 0xd9, 0xb1, 0xf0, 0x3e, 0xbd, 0x20, 0x4d, 0x01, 0x5a, 0xf4, 0xda, 0x99, 0x23
|
||||
db 0x81, 0x01, 0x5f, 0x50, 0xce, 0xa8, 0xb9, 0xb1, 0x59, 0xe5, 0xde, 0x47, 0x5b, 0xba, 0x94, 0xd3
|
||||
db 0x21, 0x7c, 0x49, 0xeb, 0xb5, 0x14, 0xe5, 0x56, 0x93, 0x06, 0x3b, 0xd2, 0x3a, 0x11, 0xca, 0x7a
|
||||
db 0x14, 0x48, 0x54, 0xc7, 0x9f, 0x03, 0x40, 0x2c, 0x0b, 0x42, 0x8e, 0xac, 0xac, 0x08, 0x04, 0x8e
|
||||
db 0xb3, 0x15, 0xe5, 0x06, 0xa6, 0x5b, 0xf0, 0x57, 0x08, 0xfa, 0x0f, 0x00, 0x7e, 0x4a, 0x16, 0xa8
|
||||
db 0xb0, 0x4d, 0x07, 0x1b, 0xbc, 0x3d, 0xd0, 0x86, 0x15, 0xcd, 0x7c, 0xb2, 0xcc, 0x37, 0x6d, 0x15
|
||||
db 0x8b, 0xd1, 0xe6, 0x3e, 0xfb, 0x6e, 0xe4, 0xea, 0xd9, 0x1f, 0x69, 0x2a, 0xbc, 0xda, 0xd9, 0x78
|
||||
db 0xee, 0xcb, 0xb6, 0xff, 0x53, 0xfd, 0xd2, 0xb9, 0x18, 0x1f, 0xdf, 0x0e, 0x69, 0xfe, 0x36, 0xb0
|
||||
db 0x77, 0x28, 0x66, 0xe2, 0xf0, 0x80, 0x4c, 0x11, 0x11, 0xba, 0xb7, 0xfd, 0x67, 0x4f, 0x05, 0xed
|
||||
db 0x0c, 0xcc, 0x3e, 0x4d, 0xd9, 0xbc, 0x52, 0xe3, 0xec, 0xd9, 0x74, 0x29, 0x30, 0xf2, 0x66, 0xd6
|
||||
db 0xfb, 0xc3, 0x5c, 0xc1, 0xd8, 0xef, 0x86, 0x08, 0x22, 0xb1, 0x6d, 0xfd, 0xee, 0xc7, 0x12, 0x25
|
||||
db 0xda, 0xee, 0xd6, 0x28, 0x3b, 0x1d, 0xa7, 0x29, 0xdf, 0x45, 0x3a, 0xa4, 0x36, 0xe0, 0xa4, 0xda
|
||||
db 0xb1, 0x2c, 0x8a, 0xa5, 0x5c, 0x8c, 0x70, 0xd8, 0xcd, 0x0f, 0xb5, 0x63, 0xd3, 0xaf, 0x59, 0x2b
|
||||
db 0x7d, 0x86, 0x4a, 0xc4, 0xcc, 0x72, 0x9e, 0x89, 0xf4, 0x38, 0x89, 0x81, 0x64, 0x6f, 0xa5, 0xac
|
||||
db 0x13, 0x59, 0xc4, 0x0f, 0xfb, 0xcc, 0x4c, 0x1d, 0x67, 0x5a, 0xbf, 0x19, 0xfc, 0x06, 0x71, 0xbd
|
||||
db 0x7f, 0xb6, 0xb1, 0x95, 0xd3, 0x7b, 0x4c, 0x40, 0x91, 0xa9, 0x26, 0xdd, 0x28, 0x69, 0x90, 0xf6
|
||||
db 0x5d, 0x16, 0x9f, 0xa9, 0x75, 0x5e, 0xad, 0x8f, 0xc8, 0x0b, 0x57, 0x48, 0xf2, 0x74, 0x77, 0x22
|
||||
db 0x5d, 0xed, 0xc2, 0x79, 0x27, 0x46, 0x0c, 0x9e, 0x6f, 0x9a, 0x9a, 0xdc, 0xe0, 0x3d, 0x24, 0xc9
|
||||
db 0xce, 0xf3, 0x34, 0x66, 0x45, 0x07, 0x0b, 0x83, 0x8c, 0xb7, 0xd9, 0x1e, 0xac, 0xc6, 0xf7, 0xef
|
||||
db 0xe7, 0xd1, 0xbc, 0xa3, 0x21, 0x85, 0x3d, 0x25, 0x90, 0x24, 0x48, 0xb1, 0x00, 0xb0, 0xd2, 0xa6
|
||||
db 0xd8, 0x4e, 0x46, 0x7c, 0xc4, 0x79, 0x40, 0x95, 0x81, 0xb4, 0xb9, 0xa8, 0x70, 0xf0, 0x12, 0xd6
|
||||
db 0xdc, 0xb2, 0x7c, 0x0f, 0x47, 0xad, 0x7d, 0x46, 0x78, 0x18, 0x6e, 0xdd, 0x5f, 0xe5, 0xd7, 0x63
|
||||
db 0x11, 0xf0, 0x5b, 0xa0, 0x48, 0x15, 0xe2, 0x55, 0xc6, 0x7f, 0xf4, 0x2e, 0x0e, 0x49, 0x39, 0x65
|
||||
db 0x3e, 0x69, 0xc1, 0x27, 0x39, 0xb3, 0x10, 0x1b, 0xf2, 0x35, 0x88, 0x0c, 0x1b, 0xac, 0x4a, 0x15
|
||||
db 0x31, 0x81, 0x63, 0xe5, 0x3d, 0x56, 0x6f, 0x34, 0x06, 0x5b, 0x1d, 0xa0, 0xea, 0x0c, 0x92, 0x6a
|
||||
db 0x22, 0x2b, 0x2d, 0xbb, 0xaf, 0xc5, 0x6d, 0x44, 0x1b, 0xb0, 0x69, 0x06, 0x27, 0x54, 0xa5, 0x7f
|
||||
db 0x07, 0xd4, 0xdc, 0xe5, 0x5c, 0x78, 0x9e, 0xf7, 0x4a, 0x47, 0x9b, 0x21, 0xf6, 0x87, 0x89, 0xad
|
||||
db 0xec, 0xe4, 0xd6, 0x83, 0xd3, 0x7b, 0x34, 0x00, 0x0b, 0x75, 0xba, 0x4c, 0x0f, 0x46, 0xd2, 0x0c
|
||||
db 0x58, 0x1b, 0x0f, 0x19, 0xb5, 0xf5, 0xba, 0x8f, 0xbd, 0x17, 0x51, 0xaf, 0xa6, 0x1a, 0x97, 0x8c
|
||||
db 0x44, 0x30, 0x7c, 0x73, 0x50, 0xca, 0x05, 0xe8, 0x3e, 0x19, 0x4a, 0x5a, 0x6b, 0x4d, 0x01, 0x05
|
||||
db 0xea, 0x1b, 0x70, 0xb6, 0xe6, 0x39, 0x5d, 0x99, 0x3b, 0xae, 0xed, 0x7c, 0xa6, 0xc7, 0x29, 0x6f
|
||||
db 0xeb, 0x0a, 0xba, 0x03, 0xd3, 0xba, 0x62, 0x21, 0xa0, 0xb7, 0xb5, 0xbf, 0x40, 0xb8, 0x4e, 0xc3
|
||||
db 0x89, 0xa0, 0xa9, 0xe8, 0xc8, 0x2b, 0xfd, 0x23, 0x32, 0x53, 0xe5, 0x35, 0xc1, 0x23, 0x97, 0xc1
|
||||
db 0x87, 0x10, 0x41, 0x21, 0xb3, 0xf6, 0x53, 0xcf, 0x28, 0x47, 0x9c, 0x69, 0x42, 0xcf, 0x0e, 0x11
|
||||
db 0x69, 0x7f, 0xc6, 0xdf, 0xc3, 0xbf, 0x04, 0x7f, 0x3a, 0xc6, 0xa1, 0x3d, 0xc6, 0x5b, 0x56, 0x8b
|
||||
db 0x52, 0x23, 0x41, 0xd7, 0x35, 0x7f, 0x86, 0xd2, 0x59, 0xcf, 0xae, 0x28, 0xa3, 0xa2, 0x23, 0x4b
|
||||
db 0x78, 0x78, 0x94, 0x3f, 0x2f, 0xf0, 0xb8, 0x94, 0xa2, 0x62, 0xb9, 0x83, 0xc7, 0x5f, 0x64, 0x45
|
||||
db 0x54, 0xaf, 0x43, 0x93, 0x7f, 0xa1, 0xe8, 0x71, 0x38, 0xc8, 0x21, 0xf4, 0xa6, 0xab, 0x2b, 0xd3
|
||||
db 0x44, 0xa2, 0x74, 0x94, 0x99, 0x3f, 0x56, 0xbc, 0x0a, 0x12, 0xe7, 0x6e, 0x1b, 0x7f, 0x98, 0xad
|
||||
db 0x28, 0xa6, 0xc8, 0x87, 0x7a, 0x88, 0xcb, 0xcf, 0x9f, 0x95, 0xa7, 0xf1, 0x66, 0xfe, 0x43, 0x3d
|
||||
db 0x71, 0x5b, 0x3a, 0xb7, 0xe4, 0xa8, 0x6f, 0x46, 0xa1, 0xaa, 0x66, 0xd2, 0x9e, 0x84, 0xfd, 0x42
|
||||
db 0x98, 0x17, 0x3e, 0xde, 0xaa, 0x18, 0xc9, 0x9c, 0x53, 0x88, 0x2b, 0x92, 0xce, 0x00, 0x8b, 0xb4
|
||||
db 0x15, 0x7a, 0x39, 0xb7, 0x57, 0xf9, 0xf2, 0x17, 0x0a, 0x8c, 0x05, 0x7b, 0x3f, 0x2a, 0xb0, 0xb7
|
||||
db 0x8a, 0xbb, 0x9a, 0x0d, 0xe4, 0x0d, 0x6a, 0xbd, 0x8a, 0xe9, 0xbd, 0xca, 0xb2, 0x6a, 0xbe, 0x76
|
||||
db 0x2c, 0xbe, 0x45, 0x3f, 0x22, 0x03, 0xb1, 0xab, 0x2d, 0xe0, 0x70, 0x52, 0xe5, 0x27, 0x8e, 0xbc
|
||||
db 0xa9, 0x8d, 0x13, 0xf4, 0xe5, 0xd7, 0xeb, 0x4e, 0x30, 0x3f, 0x76, 0x3b, 0x64, 0xad, 0x57, 0x53
|
||||
db 0x91, 0x89, 0xf4, 0x9a, 0xd1, 0x38, 0x3d, 0x58, 0xdc, 0x83, 0x65, 0x4a, 0x36, 0x30, 0x73, 0x92
|
||||
db 0x8c, 0x2f, 0x7d, 0x1e, 0x15, 0x3c, 0xca, 0x54, 0x6f, 0x17, 0xbd, 0xba, 0x97, 0x7e, 0x28, 0x11
|
||||
db 0x8e, 0x96, 0x9f, 0x46, 0x84, 0x69, 0xe3, 0xc2, 0x8e, 0x1e, 0xea, 0x6b, 0x17, 0xa7, 0xf8, 0x17
|
||||
db 0xc3, 0xd9, 0x9c, 0x53, 0x79, 0x95, 0x32, 0xf6, 0x78, 0xcd, 0x5d, 0x2f, 0x30, 0x06, 0xe8, 0x9f
|
||||
db 0x5e, 0xb2, 0x4e, 0x56, 0xf5, 0x31, 0xc3, 0x41, 0xae, 0x4b, 0x0a, 0xbd, 0xdc, 0xce, 0xea, 0xfa
|
||||
db 0x27, 0x09, 0x4e, 0xd1, 0x24, 0x14, 0x33, 0x8b, 0x21, 0x48, 0x99, 0x92, 0x07, 0xa4, 0x1a, 0x87
|
||||
db 0x34, 0x15, 0xa6, 0x12, 0x92, 0x3f, 0xf0, 0x3e, 0x18, 0x3c, 0x65, 0x3a, 0x8b, 0x17, 0x9b, 0xf2
|
||||
db 0xd9, 0x93, 0xa0, 0x19, 0x2b, 0x73, 0x59, 0x29, 0x6f, 0xb7, 0x75, 0x4b, 0x42, 0x24, 0x43, 0xa4
|
||||
db 0x20, 0xd8, 0x59, 0x8d, 0x9f, 0xd6, 0x64, 0xa1, 0xeb, 0xe3, 0x65, 0x82, 0x69, 0x74, 0x1a, 0x2b
|
||||
db 0x8d, 0x9a, 0x59, 0x5d, 0x47, 0x75, 0x63, 0xcd, 0xe4, 0x14, 0x48, 0x5f, 0x67, 0x00, 0x12, 0x3c
|
||||
db 0x58, 0x27, 0x5e, 0x83, 0xde, 0xd8, 0x97, 0xd9, 0x09, 0xd9, 0x06, 0x64, 0x96, 0x67, 0xb4, 0x4f
|
||||
db 0xb9, 0x58, 0x87, 0xc9, 0xb1, 0xdd, 0x64, 0x8f, 0x4e, 0x8f, 0xa9, 0xfa, 0x40, 0xe6, 0x8f, 0xaa
|
||||
db 0x22, 0x26, 0x16, 0x15, 0x6a, 0xa3, 0x88, 0xae, 0xa2, 0xbc, 0xa3, 0xa3, 0x56, 0xa1, 0x74, 0x6c
|
||||
db 0xa2, 0xd0, 0x47, 0x4b, 0x98, 0x0a, 0xea, 0xdd, 0xe8, 0x9c, 0xe1, 0x37, 0x44, 0x1a, 0xc0, 0xc7
|
||||
db 0x83, 0x07, 0x42, 0xca, 0x98, 0x36, 0xd7, 0x43, 0x18, 0x51, 0x32, 0xf6, 0x99, 0x61, 0x73, 0x79
|
||||
db 0x51, 0xc4, 0xe9, 0x5b, 0x9e, 0xa8, 0xb4, 0x28, 0x49, 0xbb, 0x44, 0x90, 0xe2, 0xf7, 0x7e, 0x61
|
||||
db 0x27, 0xbb, 0x85, 0x58, 0xd0, 0xdc, 0x94, 0x53, 0x02, 0x50, 0xfe, 0xc7, 0x37, 0xa2, 0x20, 0x1b
|
||||
db 0x57, 0x00, 0x9b, 0x7c, 0xa4, 0x6c, 0xa6, 0xb1, 0xae, 0xd0, 0x03, 0x67, 0x2b, 0x82, 0xd9, 0x99
|
||||
db 0x76, 0xd0, 0xc7, 0x7d, 0x2d, 0xbd, 0x39, 0x28, 0xcf, 0xe1, 0x13, 0xce, 0x1c, 0xe6, 0x4c, 0xa7
|
||||
db 0x7a, 0x8c, 0x4f, 0xa6, 0x30, 0x77, 0x6b, 0x78, 0x39, 0x6e, 0x10, 0xd1, 0x9c, 0x9a, 0xda, 0x2d
|
||||
db 0xc9, 0xef, 0xd7, 0xb1, 0xb8, 0xdf, 0x21, 0xce, 0x96, 0x53, 0xaa, 0xa6, 0x76, 0x52, 0x56, 0x0e
|
||||
db 0xe6, 0x7f, 0xed, 0x88, 0x15, 0x2a, 0xc1, 0xfe, 0xb3, 0x35, 0x54, 0x09, 0x9b, 0x5d, 0x21, 0x62
|
||||
db 0xc8, 0x6f, 0x2c, 0x6e, 0x56, 0xc8, 0xd9, 0x40, 0x67, 0xeb, 0x26, 0xf5, 0xcb, 0x18, 0xb1, 0x89
|
||||
db 0xfe, 0x58, 0x1a, 0xff, 0x41, 0xb5, 0xd6, 0xe5, 0xb3, 0x82, 0x29, 0x82, 0xee, 0xbb, 0xb2, 0x5a
|
||||
db 0x71, 0xf2, 0xca, 0xf1, 0x2f, 0xa7, 0x4d, 0xb1, 0x5c, 0xbc, 0xc3, 0x1a, 0xb4, 0x20, 0x6a, 0x7e
|
||||
db 0xb9, 0x5e, 0xcb, 0x9b, 0xf3, 0x1c, 0x2b, 0x16, 0xab, 0x15, 0x8d, 0xb5, 0x81, 0xf3, 0xbb, 0xc1
|
||||
db 0x8e, 0x2c, 0xd6, 0xd1, 0xa8, 0x23, 0x3c, 0x98, 0x3f, 0x4e, 0xff, 0x97, 0x77, 0xd1, 0xbd, 0xda
|
||||
db 0xff, 0x9c, 0x55, 0x01, 0x1c, 0x4b, 0x4b, 0x1a, 0xa9, 0x3d, 0xe9, 0xbd, 0x3c, 0x5b, 0xfd, 0x65
|
||||
db 0x34, 0x9c, 0x78, 0x8c, 0x83, 0x46, 0x72, 0xed, 0x66, 0xee, 0x00, 0xac, 0xca, 0x09, 0xaa, 0x3a
|
||||
db 0x2c, 0xc1, 0x7e, 0xde, 0x44, 0xbd, 0xe3, 0x5a, 0x11, 0x41, 0xc7, 0xc8, 0x65, 0x7a, 0xc7, 0xbb
|
||||
db 0x44, 0xad, 0x97, 0x17, 0xe8, 0x9f, 0x29, 0x2b, 0x78, 0x6d, 0x96, 0xb6, 0x9c, 0x3a, 0x6a, 0xc2
|
||||
db 0xab, 0x9a, 0x16, 0x6f, 0x05, 0x78, 0x0d, 0x83, 0xa5, 0x46, 0x8c, 0xd7, 0x57, 0x1e, 0x80, 0x2f
|
||||
db 0x7e, 0x81, 0x68, 0xa4, 0xc4, 0x3d, 0x6c, 0xae, 0x6b, 0x98, 0xb9, 0xe4, 0xb4, 0xfb, 0xf4, 0x19
|
||||
db 0xf9, 0xcd, 0xbb, 0xd0, 0xbc, 0x22, 0xdd, 0x2c, 0xbe, 0x11, 0x01, 0xc2, 0x53, 0xdd, 0xa3, 0x3a
|
||||
db 0xbf, 0x5f, 0x2a, 0x94, 0x8b, 0x58, 0x6e, 0xe3, 0x4e, 0x1b, 0x0d, 0x30, 0x1b, 0x1c, 0x6c, 0x24
|
||||
db 0x0e, 0xd9, 0x1c, 0xe1, 0x4d, 0x42, 0x48, 0xa0, 0x07, 0xb1, 0xe8, 0x10, 0xa1, 0x51, 0x6a, 0x82
|
||||
db 0x2e, 0x99, 0xb3, 0xbf, 0xe3, 0xff, 0x3c, 0x77, 0xf4, 0x0c, 0x1f, 0x22, 0x53, 0xd0, 0x99, 0x60
|
||||
db 0x5d, 0x65, 0x80, 0xb9, 0xa3, 0xb7, 0x25, 0x6d, 0xa6, 0x4f, 0xb5, 0x72, 0xaa, 0x4d, 0x0d, 0x49
|
||||
db 0x4c, 0x34, 0xc5, 0xf4, 0x1b, 0x5c, 0x3f, 0x6c, 0xbb, 0x86, 0xba, 0xc5, 0x32, 0xee, 0x23, 0x95
|
||||
db 0xe5, 0x42, 0x66, 0x92, 0x89, 0x5e, 0xf4, 0xd4, 0x2d, 0x04, 0xf2, 0xbc, 0xd7, 0xc8, 0xc9, 0xd7
|
||||
db 0xe3, 0xdb, 0x4e, 0x4b, 0xda, 0x37, 0x1f, 0xfa, 0x9c, 0xaf, 0x4b, 0x1e, 0xab, 0x64, 0x2a, 0x59
|
||||
db 0x24, 0x0f, 0xb4, 0xaf, 0xd6, 0x32, 0x30, 0xcd, 0x7c, 0xf3, 0x0f, 0xa9, 0xac, 0x3f, 0x55, 0xa2
|
||||
db 0x92, 0x21, 0x58, 0x4e, 0x99, 0xbc, 0x9f, 0xfd, 0x16, 0x7c, 0x4e, 0x5b, 0xb4, 0xc7, 0x5f, 0x8d
|
||||
db 0x0e, 0x26, 0x72, 0x17, 0x02, 0x7d, 0x12, 0xa0, 0xc5, 0xc1, 0x66, 0xd3, 0x19, 0x49, 0x42, 0xfb
|
||||
db 0x18, 0xd7, 0x18, 0x79, 0xd3, 0x32, 0xfc, 0x4a, 0xab, 0x82, 0x72, 0x0a, 0x90, 0xb7, 0xbc, 0x00
|
||||
db 0x16, 0x99, 0xd3, 0x9a, 0x76, 0xc6, 0x44, 0x92, 0x9b, 0x2b, 0x6a, 0x35, 0xca, 0x4e, 0x2e, 0x9c
|
||||
db 0x7f, 0xcb, 0xd3, 0x65, 0x1c, 0xa6, 0x95, 0x2c, 0x3d, 0xe4, 0xd3, 0xe6, 0xe7, 0xe0, 0xde, 0x1e
|
||||
db 0x54, 0xb3, 0x09, 0x3e, 0x34, 0x35, 0x68, 0x53, 0x01, 0x02, 0xf1, 0x4c, 0x89, 0x19, 0xe3, 0xc6
|
||||
db 0x4a, 0x51, 0x49, 0xf5, 0x5f, 0x3e, 0xcd, 0xae, 0x6e, 0xeb, 0x90, 0x1a, 0x53, 0x93, 0x0b, 0xe8
|
||||
db 0xc2, 0x6e, 0xee, 0xf3, 0x38, 0x5d, 0xb8, 0xaf, 0x58, 0x4b, 0xe0, 0xfd, 0x07, 0xcf, 0x15, 0x89
|
||||
db 0x2b, 0x01, 0x35, 0xbb, 0xa0, 0x2f, 0x7e, 0xd3, 0x34, 0x7b, 0x1f, 0x81, 0x12, 0x7f, 0xb0, 0xff
|
||||
db 0xe7, 0xa0, 0xf2, 0xc4, 0x86, 0x98, 0x45, 0xe2, 0xa1, 0x1e, 0x4c, 0xc0, 0x23, 0x05, 0x49, 0x0b
|
||||
db 0x0d, 0xc3, 0x1e, 0x30, 0x20, 0xc6, 0x34, 0xb7, 0xe1, 0x09, 0x84, 0xd5, 0x2a, 0x40, 0x75, 0x9b
|
||||
db 0x46, 0xbb, 0xa5, 0xfe, 0xbd, 0x7d, 0x39, 0xe4, 0x7b, 0x38, 0xdc, 0x9c, 0xaf, 0xc8, 0x12, 0xf4
|
||||
db 0x78, 0xb8, 0x51, 0x4a, 0x21, 0xfe, 0xf9, 0x77, 0xf6, 0xb5, 0xad, 0x69, 0xc9, 0x4d, 0xbf, 0x67
|
||||
db 0xfc, 0x5d, 0x80, 0x7c, 0x76, 0x2c, 0xe5, 0xf2, 0xd7, 0x7f, 0xce, 0xb5, 0x1c, 0x09, 0xa5, 0xc3
|
||||
db 0x98, 0x18, 0x2d, 0x18, 0xfb, 0x61, 0x13, 0xea, 0xbc, 0x87, 0x3a, 0x3f, 0xb4, 0xaf, 0x3c, 0x3b
|
||||
db 0x3b, 0xb6, 0xd2, 0xc7, 0x5c, 0x2c, 0xe1, 0x11, 0xb3, 0x9d, 0xf1, 0x52, 0xba, 0xb5, 0xf0, 0x69
|
||||
db 0xcd, 0xd2, 0x93, 0x9e, 0x80, 0x45, 0x78, 0x17, 0x6d, 0x52, 0x51, 0xad, 0xed, 0x6d, 0x9e, 0x15
|
||||
db 0xca, 0xb1, 0xfe, 0x22, 0x7b, 0x87, 0xb8, 0x40, 0x06, 0x2d, 0xb0, 0xbb, 0x05, 0x7c, 0x52, 0xd2
|
||||
db 0xcd, 0xc8, 0x9c, 0xea, 0xd3, 0x4c, 0xb5, 0x06, 0xb4, 0x70, 0xad, 0x09, 0xa5, 0xb8, 0x66, 0xba
|
||||
db 0x31, 0x0d, 0xe0, 0xe2, 0xcf, 0x62, 0x9f, 0x6d, 0x6d, 0x1a, 0x47, 0x21, 0xd5, 0x33, 0x6b, 0xd7
|
||||
db 0x75, 0xff, 0x98, 0x6c, 0xb2, 0x78, 0x6d, 0x45, 0x50, 0xeb, 0xfb, 0xea, 0xb7, 0x2a, 0x27, 0x02
|
||||
db 0xc4, 0x03, 0xde, 0x56, 0x23, 0x26, 0x10, 0x21, 0x57, 0x9c, 0x3b, 0x4c, 0x79, 0x2c, 0x3e, 0xfe
|
||||
db 0xc8, 0x16, 0xe4, 0xd6, 0x60, 0xb8, 0x46, 0xe3, 0x4b, 0x7e, 0x3d, 0xb3, 0x83, 0x19, 0x54, 0x65
|
||||
db 0x51, 0x7a, 0x81, 0xdd, 0x07, 0x33, 0x92, 0x08, 0x64, 0x0b, 0xc2, 0x06, 0x5c, 0x07, 0x81, 0x40
|
||||
db 0x1b, 0xb4, 0x5a, 0x47, 0x2b, 0xdc, 0x96, 0x98, 0x4c, 0x65, 0xad, 0x8e, 0x8e, 0x77, 0xbe, 0x99
|
||||
db 0x60, 0x4c, 0xb5, 0x6b, 0xed, 0xb7, 0x52, 0x5d, 0x99, 0x2e, 0x93, 0x40, 0xfe, 0x45, 0x83, 0x28
|
||||
db 0x9b, 0x8b, 0x7f, 0x77, 0x2b, 0xdc, 0x61, 0xbe, 0x62, 0x28, 0xe8, 0x23, 0x3f, 0xdb, 0x1d, 0x6d
|
||||
db 0x3b, 0xe8, 0x90, 0x05, 0x12, 0xf2, 0xb4, 0xf0, 0x1b, 0xbb, 0x2f, 0x4b, 0x9e, 0x9f, 0x0e, 0x4e
|
||||
db 0x9e, 0x6a, 0x38, 0x7e, 0x97, 0x13, 0x90, 0x57, 0xb9, 0x49, 0x52, 0xb7, 0x4f, 0xd3, 0xc1, 0x39
|
||||
db 0x95, 0x20, 0xd4, 0x83, 0x48, 0x0e, 0x7a, 0x9d, 0x89, 0x9d, 0xf4, 0xec, 0xe7, 0xcc, 0xde, 0x0a
|
||||
db 0xac, 0xc5, 0xb0, 0x4d, 0xc5, 0x25, 0x74, 0x62, 0x66, 0x51, 0x4f, 0xeb, 0x4e, 0x9d, 0x3d, 0x04
|
||||
db 0x27, 0xec, 0xfe, 0x8d, 0x03, 0x20, 0x38, 0x30, 0x5d, 0xf3, 0xf0, 0x97, 0xbb, 0xa9, 0xd1, 0xea
|
||||
db 0x73, 0x73, 0x40, 0x2c, 0x0b, 0xa7, 0xc9, 0x8d, 0xac, 0x75, 0xc4, 0x46, 0x7c, 0xc2, 0x9a, 0x26
|
||||
db 0x07, 0xae, 0x02, 0x27, 0x42, 0xa8, 0x90, 0xb6, 0x9b, 0x98, 0xec, 0x2e, 0xf6, 0xf6, 0x17, 0xda
|
||||
db 0x9f, 0xfb, 0x54, 0xea, 0xae, 0x96, 0xfe, 0xd6, 0x35, 0x4f, 0x07, 0x9f, 0xf4, 0x57, 0x36, 0xfe
|
||||
db 0xb1, 0x43, 0xee, 0xe3, 0x21, 0x00, 0x43, 0x12, 0xf2, 0xff, 0xa5, 0x37, 0x65, 0x01, 0xf0, 0xb4
|
||||
db 0xe8, 0x68, 0xa3, 0xff, 0x31, 0x5f, 0x3f, 0x56, 0xa5, 0xd2, 0xcc, 0xab, 0xa4, 0x90, 0xf9, 0x98
|
||||
db 0x0b, 0xdc, 0x0d, 0x20, 0x3c, 0x33, 0xda, 0xf1, 0x54, 0xd5, 0x6d, 0xc4, 0xa9, 0xc4, 0x54, 0x29
|
||||
db 0x56, 0x69, 0x96, 0x98, 0x74, 0x13, 0x72, 0x1f, 0x95, 0xe9, 0xe2, 0xab, 0x60, 0x74, 0x91, 0x96
|
||||
db 0xdf, 0xa4, 0xd6, 0x62, 0x3c, 0x35, 0x7e, 0xc4, 0x21, 0x16, 0xa3, 0x32, 0xac, 0x20, 0x52, 0xd4
|
||||
db 0xbb, 0xc2, 0xa5, 0x97, 0x86, 0x4a, 0x55, 0xf4, 0x09, 0xf2, 0x0e, 0xd6, 0x1a, 0xfa, 0x00, 0x67
|
||||
db 0x45, 0x57, 0xb3, 0xaa, 0xe5, 0x7c, 0x17, 0x8d, 0xde, 0x75, 0xd7, 0x49, 0x6e, 0xb0, 0xb2, 0xa0
|
||||
db 0x58, 0xd8, 0x01, 0xf0, 0x22, 0x9c, 0xe4, 0xeb, 0x71, 0x5f, 0x4d, 0x38, 0xf2, 0x7e, 0xee, 0xba
|
||||
db 0xf9, 0x39, 0xff, 0x42, 0x91, 0x00, 0x63, 0x5c, 0x86, 0x02, 0x81, 0x51, 0x10, 0xfb, 0xcf, 0x2a
|
||||
db 0xcf, 0x16, 0xd9, 0x8f, 0x3a, 0xbb, 0x29, 0xcb, 0xe2, 0xc9, 0xd9, 0xe2, 0xd9, 0x05, 0x1b, 0x46
|
||||
db 0x08, 0x2c, 0x6d, 0x5b, 0x1a, 0x7d, 0x5b, 0xca, 0x5b, 0xae, 0x18, 0x48, 0x15, 0x3b, 0x85, 0xd1
|
||||
db 0x29, 0xcf, 0xaf, 0xa5, 0x68, 0xe9, 0x8d, 0x9e, 0x0b, 0xe1, 0x55, 0x54, 0x68, 0x28, 0x9b, 0x4c
|
||||
db 0x94, 0x30, 0x3a, 0xc0, 0xaa, 0xf8, 0xeb, 0x7b, 0x58, 0x53, 0x5f, 0x25, 0x2e, 0xbf, 0x72, 0x26
|
||||
db 0xd8, 0x9c, 0xa9, 0xfe, 0x30, 0xe0, 0x68, 0x25, 0xba, 0x71, 0x1a, 0x82, 0xbb, 0xee, 0x03, 0xc9
|
||||
db 0x4b, 0x0a, 0x22, 0xda, 0x93, 0xa0, 0x72, 0x49, 0x72, 0x3a, 0x8f, 0xbe, 0x39, 0x04, 0x7c, 0x06
|
||||
db 0xa1, 0x50, 0xa1, 0x94, 0xb4, 0x66, 0x91, 0xee, 0x76, 0xa4, 0xbe, 0x21, 0x33, 0xbe, 0xa9, 0x68
|
||||
db 0xe6, 0x03, 0xdd, 0x25, 0x3b, 0x78, 0xe3, 0x5a, 0x0c, 0xcf, 0x2b, 0xa2, 0x03, 0x63, 0x8d, 0xd7
|
||||
db 0xc4, 0xf0, 0x6e, 0xea, 0xe1, 0x76, 0x93, 0x38, 0x7b, 0x85, 0xef, 0xff, 0xce, 0xb0, 0xe1, 0xe3
|
||||
db 0x86, 0x3d, 0xb6, 0xae, 0xee, 0xf7, 0x92, 0x8a, 0x1b, 0x29, 0x00, 0x9b, 0x85, 0xaf, 0xa2, 0x5e
|
||||
db 0x90, 0xd9, 0xdc, 0xca, 0xde, 0xde, 0xab, 0xfe, 0x05, 0x61, 0x3c, 0xb6, 0x2f, 0x40, 0x59, 0x1f
|
||||
db 0x73, 0x80, 0x52, 0xf6, 0x6f, 0x28, 0x30, 0x4b, 0xf2, 0x88, 0x9e, 0x63, 0x84, 0x1b, 0xd2, 0xf4
|
||||
db 0x67, 0x3b, 0xaf, 0x48, 0x27, 0xfd, 0x7e, 0x30, 0x6e, 0xb8, 0x81, 0xbf, 0xe5, 0x4c, 0x19, 0x16
|
||||
db 0x24, 0xd0, 0x8e, 0x3a, 0xc9, 0xcd, 0xc8, 0x6f, 0x2e, 0x99, 0xda, 0xb8, 0x7c, 0xd9, 0xbb, 0x2c
|
||||
db 0xe3, 0xdf, 0xd0, 0x96, 0xe2, 0xcc, 0x99, 0x5b, 0x1d, 0xff, 0x81, 0x74, 0x84, 0x0b, 0x9d, 0x09
|
||||
db 0x3e, 0x1b, 0x0c, 0x42, 0x3d, 0x96, 0x15, 0x44, 0xed, 0x97, 0x9a, 0x99, 0x68, 0x02, 0x2c, 0x79
|
||||
db 0x8f, 0xcc, 0xff, 0x83, 0x5e, 0x6e, 0x97, 0x00, 0x50, 0x83, 0xc2, 0x29, 0x2b, 0x27, 0xe6, 0x4f
|
||||
db 0x18, 0xb0, 0x45, 0xa9, 0xf8, 0x30, 0x35, 0x7f, 0x20, 0xdd, 0xd7, 0x07, 0x32, 0x55, 0x95, 0x4a
|
||||
db 0xf3, 0xf5, 0x35, 0x5b, 0xac, 0xef, 0xfa, 0xbb, 0x54, 0xba, 0x4d, 0x79, 0x66, 0xce, 0x38, 0x5e
|
||||
db 0x23, 0xd7, 0x1b, 0x03, 0x37, 0x74, 0xa7, 0xe0, 0xb1, 0x2c, 0xe5, 0xa4, 0x00, 0x36, 0x9a, 0xe9
|
||||
db 0x36, 0xd4, 0x3e, 0x35, 0x37, 0xb2, 0xc1, 0x71, 0x90, 0x80, 0x3b, 0xd8, 0x6b, 0x7e, 0x79, 0x0a
|
||||
db 0x7d, 0xe3, 0x3d, 0xc8, 0xd3, 0xb3, 0x56, 0xb6, 0xef, 0x73, 0x3d, 0x24, 0x07, 0x0e, 0xeb, 0x8e
|
||||
db 0x9b, 0x25, 0xaf, 0x3b, 0xa3, 0x92, 0xf5, 0x19, 0x16, 0xba, 0x1f, 0x6f, 0x92, 0x4b, 0x3f, 0x3c
|
||||
db 0xc8, 0xac, 0xdd, 0x70, 0xc6, 0x3b, 0x45, 0x0b, 0xa5, 0xe0, 0x8f, 0xa4, 0xd6, 0x56, 0xd8, 0xb9
|
||||
db 0xc1, 0x1a, 0x53, 0x76, 0x37, 0x60, 0xc9, 0xf4, 0xc8, 0x0a, 0x17, 0x6d, 0x1d, 0xb8, 0x8e, 0xec
|
||||
db 0xa8, 0x9c, 0x71, 0x08, 0x1f, 0x45, 0x96, 0xc8, 0xed, 0x1e, 0x47, 0x09, 0xbb, 0xe6, 0xee, 0x36
|
||||
db 0x8e, 0x87, 0xc6, 0xeb, 0xe5, 0x88, 0xd8, 0xab, 0x98, 0x41, 0x4f, 0x2a, 0x49, 0x15, 0x68, 0xf6
|
||||
db 0x51, 0xaf, 0xc7, 0x74, 0x7c, 0xaa, 0x26, 0x1a, 0x2f, 0xe6, 0x96, 0x86, 0x7c, 0x00, 0xa4, 0x57
|
||||
db 0x90, 0x1f, 0x83, 0x02, 0x0c, 0xb2, 0xec, 0x27, 0x7f, 0xbc, 0x78, 0x11, 0x64, 0xbe, 0x34, 0x25
|
||||
db 0xbd, 0xf8, 0x56, 0x00, 0x5f, 0xdd, 0x85, 0x95, 0x23, 0xad, 0xe9, 0x26, 0x1e, 0xd3, 0xfc, 0x22
|
||||
db 0xe6, 0x35, 0x07, 0xbc, 0xf6, 0x88, 0x19, 0x61, 0x2e, 0xd5, 0x0d, 0xc0, 0x98, 0x79, 0x59, 0x0a
|
||||
db 0x33, 0x44, 0xa8, 0x70, 0xd8, 0xda, 0x45, 0x72, 0xdb, 0x83, 0xf7, 0xbe, 0xbb, 0x93, 0xc9, 0xaa
|
||||
db 0xf5, 0xfb, 0xdc, 0x0a, 0x55, 0x54, 0xd1, 0xae, 0x9e, 0x14, 0x38, 0x24, 0x06, 0x6e, 0x4d, 0x17
|
||||
db 0xaa, 0xb1, 0xe4, 0x55, 0x9b, 0x7c, 0xc2, 0xe7, 0xb6, 0x82, 0x1b, 0x5d, 0x21, 0x20, 0xfc, 0x34
|
||||
db 0x51, 0xf7, 0xfd, 0x20, 0x17, 0x4b, 0xd1, 0x9f, 0xc7, 0x2a, 0x57, 0x62, 0x4a, 0x60, 0x3f, 0xfa
|
||||
db 0x70, 0x75, 0x1a, 0x3e, 0x9d, 0xbd, 0x6c, 0xe3, 0x60, 0xc3, 0xd3, 0xa6, 0x3b, 0x73, 0xa5, 0x4f
|
||||
db 0x06, 0x79, 0xf4, 0x6e, 0x3a, 0xae, 0xa4, 0x98, 0x86, 0xb9, 0x1b, 0x8b, 0x66, 0xd9, 0x96, 0xdb
|
||||
db 0xa5, 0x47, 0xd3, 0xa8, 0x05, 0x3c, 0x50, 0x57, 0x8a, 0x8f, 0xe0, 0x7f, 0xaf, 0x75, 0x30, 0x44
|
||||
db 0x01, 0xce, 0x17, 0xb8, 0x89, 0xd4, 0x12, 0xaa, 0xe5, 0x2e, 0xe2, 0x75, 0x70, 0x06, 0x02, 0x5c
|
||||
db 0xbd, 0x85, 0xaa, 0x75, 0x02, 0x98, 0xe0, 0x0f, 0xe9, 0x94, 0x43, 0x84, 0x8c, 0xca, 0xc1, 0x53
|
||||
db 0x2f, 0x5c, 0x9a, 0x04, 0x9c, 0x2c, 0x50, 0xc7, 0x6d, 0x13, 0x70, 0x8f, 0x7d, 0xa5, 0x09, 0xc0
|
||||
db 0x2b, 0x75, 0x55, 0x57, 0xc0, 0x51, 0xad, 0x86, 0x18, 0xc5, 0x9a, 0x9f, 0x1d, 0x99, 0x3e, 0xbd
|
||||
db 0x38, 0x24, 0x33, 0xd6, 0x04, 0x98, 0xde, 0x19, 0xcc, 0xb3, 0x72, 0x53, 0x6b, 0xbb, 0x38, 0x03
|
||||
db 0xdc, 0x86, 0xe3, 0x1b, 0x12, 0x04, 0x86, 0x92, 0x3d, 0x3f, 0xf4, 0x4d, 0x73, 0x8a, 0xe7, 0x67
|
||||
db 0x68, 0xae, 0x63, 0x13, 0x7b, 0x48, 0x90, 0xce, 0x35, 0xfb, 0xf3, 0x46, 0x17, 0xb3, 0xcd, 0x2f
|
||||
db 0xeb, 0xb5, 0x7a, 0x11, 0xa9, 0xe1, 0xa6, 0xab, 0x0c, 0x9e, 0x9f, 0xd1, 0x08, 0xae, 0xc1, 0x68
|
||||
db 0xd2, 0xfc, 0x41, 0x36, 0xa8, 0xf4, 0x97, 0xbf, 0x86, 0x61, 0x90, 0x51, 0x02, 0x2e, 0x9a, 0x64
|
||||
db 0x4e, 0xfb, 0xd1, 0xe5, 0x73, 0x24, 0x07, 0xb5, 0x70, 0xa1, 0xa2, 0xb7, 0xcb, 0x0c, 0xbc, 0x1a
|
||||
db 0x4a, 0x55, 0x9e, 0x3f, 0x3b, 0xdb, 0x33, 0x4c, 0x01, 0x63, 0x1f, 0xbe, 0xae, 0x05, 0x3e, 0x45
|
||||
db 0x9e, 0xcf, 0x2e, 0x5f, 0x3b, 0x83, 0x8a, 0xc7, 0xd7, 0x39, 0x3b, 0xfc, 0x54, 0xf0, 0x10, 0x42
|
||||
db 0x9d, 0x5e, 0x12, 0xc2, 0xb8, 0x8c, 0x4e, 0x26, 0xd7, 0xa0, 0xa1, 0x7a, 0xc0, 0x27, 0x72, 0x52
|
||||
db 0xdb, 0xc5, 0xed, 0xe1, 0x86, 0x19, 0x0a, 0xff, 0x43, 0x3d, 0x1c, 0x12, 0xb2, 0xbe, 0x5c, 0x12
|
||||
db 0x4b, 0xbf, 0xff, 0x20, 0xe3, 0xde, 0x4a, 0x74, 0x89, 0x67, 0x42, 0xc3, 0xaf, 0xe3, 0x8a, 0x8a
|
||||
db 0x57, 0x88, 0xdf, 0xbe, 0x1a, 0x0c, 0x58, 0xa1, 0xfe, 0x21, 0x57, 0x97, 0xf6, 0xef, 0xba, 0x34
|
||||
db 0x54, 0x60, 0x00, 0x71, 0x09, 0x4a, 0x5b, 0x89, 0x61, 0x4a, 0x67, 0x19, 0x34, 0x44, 0x83, 0x21
|
||||
db 0x3d, 0xeb, 0x67, 0xff, 0xf7, 0x68, 0xbb, 0x29, 0xa0, 0x74, 0x5e, 0xad, 0x78, 0xb4, 0x11, 0xc5
|
||||
db 0x5e, 0x0e, 0xc0, 0xd4, 0xe7, 0x50, 0x40, 0xa1, 0xb5, 0x98, 0xdb, 0x75, 0x1f, 0xa5, 0xbc, 0x1b
|
||||
db 0xeb, 0x13, 0x18, 0x0e, 0x92, 0x54, 0x17, 0x2d, 0x5b, 0xf8, 0x09, 0x50, 0x27, 0x49, 0xf5, 0x01
|
||||
db 0xb9, 0x51, 0xd1, 0x85, 0x34, 0x67, 0xd8, 0xb9, 0x5f, 0x01, 0x7b, 0xfc, 0xe7, 0x1e, 0xc8, 0xfc
|
||||
db 0x2f, 0xda, 0x81, 0xfd, 0x76, 0x69, 0x5b, 0x47, 0x98, 0x1b, 0x9b, 0xee, 0x9b, 0x18, 0x8e, 0x30
|
||||
db 0x85, 0x9d, 0x45, 0xde, 0xa8, 0x9b, 0x4e, 0x57, 0x26, 0x90, 0x0b, 0x9a, 0xe0, 0xf7, 0xfa, 0x08
|
||||
db 0x1d, 0xe3, 0xca, 0xb8, 0xaa, 0xda, 0x4e, 0xe3, 0xb6, 0x33, 0x05, 0x9a, 0x75, 0x70, 0x18, 0x86
|
||||
db 0x60, 0x31, 0xc1, 0x05, 0x56, 0x02, 0x30, 0xbd, 0xff, 0x3b, 0xa9, 0xca, 0xe4, 0x84, 0xe6, 0x96
|
||||
db 0x47, 0xcf, 0x8b, 0xa8, 0xd4, 0x63, 0x8f, 0x8f, 0x55, 0x4a, 0xbc, 0x4c, 0x3c, 0x61, 0x96, 0x38
|
||||
db 0xcc, 0x10, 0x7e, 0x4e, 0x5c, 0x97, 0xd3, 0x54, 0x22, 0xde, 0xfb, 0x03, 0x81, 0x4e, 0x6d, 0x76
|
||||
db 0xb5, 0xab, 0x8f, 0xba, 0xf5, 0xf0, 0x1a, 0xf9, 0x69, 0x64, 0x30, 0xb3, 0x19, 0x30, 0x54, 0x97
|
||||
db 0x14, 0x66, 0x5c, 0xcf, 0x48, 0x0f, 0x74, 0xf3, 0xbe, 0x16, 0x10, 0x6c, 0xb4, 0x93, 0x86, 0xd1
|
||||
db 0x21, 0xd0, 0x6a, 0x12, 0x35, 0x03, 0x45, 0x99, 0xaa, 0xe1, 0x0a, 0xd9, 0x58, 0x83, 0x2f, 0x97
|
||||
db 0xcb, 0x0d, 0x81, 0x4b, 0x82, 0x01, 0x6f, 0xd6, 0x20, 0xee, 0xf3, 0xbf, 0xdc, 0x3d, 0x67, 0x6c
|
||||
db 0xa5, 0x7c, 0x6d, 0x21, 0x09, 0x99, 0x2e, 0x0a, 0x98, 0x7c, 0x50, 0x56, 0x19, 0x54, 0xcc, 0x79
|
||||
db 0xe1, 0x84, 0x18, 0x86, 0xf8, 0x5a, 0x1b, 0xf7, 0x1f, 0x38, 0xe0, 0x3a, 0xb9, 0x50, 0xc1, 0xf1
|
||||
db 0xbe, 0x66, 0x89, 0xe2, 0x68, 0x4a, 0x11, 0x0b, 0xfb, 0x84, 0x02, 0x38, 0x31, 0xf4, 0xda, 0x50
|
||||
db 0xb6, 0x5f, 0x27, 0x62, 0xc7, 0x5a, 0x0f, 0x99, 0xb7, 0x7e, 0x4a, 0x49, 0xe9, 0x67, 0xe0, 0xa5
|
||||
db 0x0d, 0x08, 0x95, 0xf0, 0xe4, 0x3b, 0x62, 0x30, 0x2b, 0x89, 0x21, 0xdd, 0x52, 0x99, 0x12, 0x16
|
||||
db 0x83, 0x94, 0x6a, 0x38, 0x1f, 0x8d, 0x81, 0xbf, 0x1f, 0xf9, 0xe0, 0x9c, 0x80, 0xcc, 0x7c, 0xfe
|
||||
db 0x33, 0x35, 0x27, 0x26, 0xca, 0xcc, 0x1f, 0x43, 0xcd, 0xb0, 0x74, 0x0e, 0xff, 0x1c, 0x86, 0x43
|
||||
db 0xab, 0x44, 0xbc, 0x31, 0xff, 0xa4, 0x54, 0x95, 0xd4, 0x79, 0x9e, 0xc0, 0xed, 0x87, 0x1c, 0x2e
|
||||
db 0x50, 0x47, 0xad, 0xc0, 0x2f, 0x5e, 0x8c, 0x15, 0xfb, 0x86, 0x2c, 0xa5, 0x61, 0x2a, 0x60, 0x12
|
||||
db 0xbc, 0x1f, 0x84, 0xe9, 0x75, 0x55, 0x7e, 0x2c, 0x11, 0xd0, 0xfc, 0x66, 0x89, 0x86, 0x2f, 0x26
|
||||
db 0x43, 0x1e, 0xa6, 0x6c, 0xa6, 0x40, 0xa9, 0x37, 0x65, 0x99, 0x72, 0xe1, 0x1a, 0xdc, 0x23, 0x53
|
||||
db 0x09, 0x8e, 0xa1, 0xd6, 0xda, 0xd9, 0x95, 0xaf, 0x58, 0xe0, 0x2a, 0x4a, 0xd3, 0xbd, 0xbd, 0x86
|
||||
|
||||
align 64
|
||||
.xsave_data:
|
||||
times 4096 db 0
|
||||
@@ -0,0 +1,21 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RDX": "0x5152535455565758",
|
||||
"R8": "0x5152535455565758"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
; FEX-Emu had a bug where address size override was overriding destination and source sizes on operations not affecting memory.
|
||||
; This showed up as a bug in OpenSSL where GCC was padding move instructions with the address size prefix, knowing that it wouldn't do anything.
|
||||
; FEX interpreted this address size prefix as making the destination 32-bit resulting in zero-extending the 64-bit source.
|
||||
; Ensure this doesn't happen again.
|
||||
mov rdx, 0x414243444546748
|
||||
mov r8, 0x5152535455565758
|
||||
jmp .test
|
||||
.test:
|
||||
|
||||
; Add a couple address size prefixes
|
||||
db 0x67, 0x67
|
||||
mov rdx, r8
|
||||
hlt
|
||||
@@ -0,0 +1,51 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0x0000000000000078", "0x0000000000000077"],
|
||||
"XMM1": ["0x0000000000000078", "0x0000000000000077"],
|
||||
"XMM2": ["0x0000000000000078", "0x0000000000000077"],
|
||||
"XMM3": ["0x0000000000000078", "0x7800000000000077"],
|
||||
"XMM4": ["0x0000000000000078", "0"],
|
||||
"XMM5": ["0x0000000000000078", "0"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX-Emu had a bug with vector loadstore instructions where 16-bit and 8-bit vector loadstores with reg+reg source would assert in the code emitter.
|
||||
; This affected both vector loads and stores. SSE 8-bit and 16-bit are quite uncommon so this isn't encountered frequently.
|
||||
; Tests a few different instructions that access 16-bit and 8-bit with loads and stores.
|
||||
lea rax, [rel .data]
|
||||
lea rbx, [rel .data_temp]
|
||||
mov rcx, 0
|
||||
jmp .test
|
||||
.test:
|
||||
|
||||
; 16-bit loads
|
||||
pmovzxbq xmm0, [rax + rcx]
|
||||
pmovzxbq xmm1, [rax + rcx*2]
|
||||
pmovzxbq xmm2, [rax + rcx*4]
|
||||
pmovzxbq xmm3, [rax + rcx*8]
|
||||
|
||||
; 8-bit load
|
||||
pinsrb xmm3, [rax + rcx], 1111b
|
||||
|
||||
; 8-bit store
|
||||
pextrb [rbx + rcx], xmm0, 0
|
||||
|
||||
; Load the result back
|
||||
movaps xmm4, [rbx + rcx]
|
||||
|
||||
; 16-bit store
|
||||
pextrb [rbx + rcx], xmm0, 0
|
||||
|
||||
; Load the result back
|
||||
movaps xmm5, [rbx + rcx]
|
||||
|
||||
hlt
|
||||
align 32
|
||||
.data:
|
||||
dq 0x7172737475767778
|
||||
dq 0x4142434445464748
|
||||
|
||||
.data_temp:
|
||||
dq 0,0
|
||||
@@ -0,0 +1,47 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"XMM1": ["0", "0", "0", "0"],
|
||||
"XMM15": ["0xf1cda2562209301d", "0x0f350767409162b7", "0", "0"]
|
||||
},
|
||||
"HostFeatures": ["AVX"]
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX had a bug where VSIB indexing wasn't allow xmm4/ymm4 to be encoded inside of the VSIB due to legacy SIB behaviour
|
||||
; This ensures that VSIB with xmm4 is allowed to work.
|
||||
; 128-bit
|
||||
; 1x displacement
|
||||
; 32-bit indexes
|
||||
|
||||
lea rax, [rel .data_mid]
|
||||
|
||||
vmovapd ymm15, [rel .data]
|
||||
|
||||
; Zero mask
|
||||
vmovaps xmm4, [rel .index_d0]
|
||||
vmovaps xmm1, [rel .mask_00000000]
|
||||
vgatherdps xmm15, [xmm4 * 1 + rax], xmm1
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
|
||||
; Masks only care about the sign bit.
|
||||
.mask_00000000:
|
||||
dd 0, 0, 0, 0, 0, 0, 0, 0
|
||||
|
||||
; Indexing is a signed 32-bit integer.
|
||||
.index_d0:
|
||||
dd 0, 0, 0, 0, 0, 0, 0, 0
|
||||
|
||||
; Random data, 512-byte per line
|
||||
.data:
|
||||
db 0x1d, 0x30, 0x09, 0x22, 0x56, 0xa2, 0xcd, 0xf1, 0xb7, 0x62, 0x91, 0x40, 0x67, 0x07, 0x35, 0x0f, 0x59, 0x33, 0x63, 0x52, 0x26, 0xd2, 0x2f, 0x00, 0x8a, 0x36, 0xb7, 0xf7, 0xaf, 0x4f, 0xa1, 0xc0, 0xe3, 0xbe, 0x99, 0x67, 0x5f, 0x80, 0x60, 0xcb, 0x43, 0xfa, 0x5b, 0x86, 0xb1, 0x11, 0xbc, 0xb3, 0x7b, 0x43, 0x5b, 0x45, 0x9e, 0x33, 0x89, 0xb5, 0x1b, 0xb9, 0x33, 0x4f, 0xdb, 0x5d, 0x93, 0xd6, 0x4f, 0xbc, 0x37, 0xde, 0xeb, 0xdb, 0x43, 0x2b, 0x05, 0x60, 0xb8, 0x98, 0x5c, 0xa3, 0xe3, 0x1b, 0x33, 0x03, 0x29, 0x4b, 0x12, 0x4c, 0x1e, 0xe6, 0x5e, 0x0e, 0x6c, 0xa1, 0xb9, 0x36, 0xfa, 0x6c, 0x7f, 0xc6, 0xa8, 0x38, 0x73, 0x2a, 0x0a, 0x25, 0x69, 0xa5, 0x97, 0x3f, 0x24, 0x00, 0x30, 0x4d, 0x27, 0xb3, 0x94, 0x48, 0xef, 0x47, 0x98, 0x71, 0x0d, 0x56, 0x76, 0xec, 0x41, 0x12, 0x9b, 0x7b, 0x9c, 0xf5, 0x85, 0x07, 0x2d, 0x6b, 0xc6, 0xc1, 0x2e, 0x72, 0x22, 0x5a, 0x43, 0xff, 0x1e, 0xec, 0x67, 0x2b, 0x31, 0x96, 0x14, 0x2c, 0xb1, 0x5f, 0x5d, 0x0c, 0xc9, 0xad, 0x15, 0x5f, 0xab, 0x66, 0x14, 0x1c, 0x72, 0xfa, 0x23, 0xef, 0x9f, 0x77, 0xf6, 0x50, 0xb0, 0x70, 0xb8, 0x3c, 0x85, 0x9e, 0x90, 0x69, 0x17, 0x25, 0xae, 0x6e, 0xe2, 0x16, 0x7d, 0x42, 0x38, 0xdf, 0x74, 0x72, 0x7b, 0x97, 0xa9, 0x9e, 0x40, 0x24, 0x85, 0xdc, 0x64, 0xfa, 0xb1, 0x8b, 0x95, 0xe6, 0xe4, 0x13, 0x72, 0xf1, 0x52, 0x2f, 0xa0, 0xd6, 0x52, 0xc0, 0x11, 0xa7, 0xfe, 0xd5, 0x3b, 0x56, 0xca, 0xbc, 0x01, 0xce, 0x3d, 0xd2, 0x30, 0x97, 0x1d, 0xdc, 0xeb, 0x9d, 0xa9, 0x3e, 0x09, 0xef, 0xee, 0x7f, 0x09, 0x7b, 0x82, 0x43, 0x15, 0x2e, 0xa4, 0x2e, 0x97, 0x21, 0x92, 0x7e, 0x69, 0x21, 0x25, 0xda, 0x46, 0x7c, 0x0c, 0xcd, 0x1d, 0xde, 0x42, 0x11, 0xa2, 0xef, 0xa2, 0xc8, 0x32, 0x9a, 0x82, 0xcf, 0x72, 0x7e, 0x22, 0xa6, 0x11, 0xfa, 0xec, 0x0b, 0x77, 0x99, 0x38, 0x03, 0xf6, 0x80, 0xba, 0xea, 0x75, 0x19, 0xb0, 0x48, 0x02, 0xb2, 0x6b, 0xc0, 0x8c, 0xfb, 0xfe, 0xaf, 0x94, 0x4f, 0x6f, 0xb4, 0xcb, 0x1c, 0x27, 0xf0, 0x41, 0xb6, 0x46, 0x41, 0x68, 0x3d, 0x05, 0x79, 0x6b, 0xcd, 0xb7, 0x20, 0xdc, 0x40, 0x81, 0x58, 0xcb, 0x33, 0xa3, 0xf3, 0x34, 0xdc, 0x63, 0x2d, 0xa5, 0xb5, 0xa1, 0xd1, 0xfd, 0x49, 0x5b, 0x46, 0x94, 0x01, 0xa8, 0xf2, 0xd8, 0x93, 0x2c, 0xbb, 0x57, 0xfe, 0x7c, 0x77, 0x3b, 0x19, 0x6f, 0x3c, 0xaa, 0x23, 0x5b, 0xc0, 0xe7, 0x00, 0x41, 0x97, 0x91, 0xe8, 0x00, 0x12, 0xdf, 0xf6, 0x5c, 0x2e, 0xc6, 0x8e, 0xc6, 0x77, 0x59, 0x78, 0x9b, 0xef, 0x63, 0xb0, 0xd7, 0xbb, 0xc4, 0x0b, 0x60, 0x65, 0x3f, 0xfe, 0xbf, 0x04, 0x3e, 0xae, 0xc2, 0xa5, 0x90, 0xe1, 0x2a, 0x56, 0x3f, 0x4c, 0x3f, 0x7a, 0x7d, 0xda, 0x81, 0x50, 0xea, 0x4c, 0xfe, 0xc3, 0xf8, 0x5c, 0x2b, 0x67, 0xb3, 0x9f, 0x8b, 0x95, 0xda, 0x6f, 0x5d, 0xdd, 0x82, 0x7f, 0x52, 0xa2, 0xcc, 0x57, 0xec, 0xc4, 0x14, 0xd2, 0x4f, 0x1b, 0xcb, 0xea, 0xaf, 0x0e, 0x0f, 0x53, 0xaa, 0x56, 0x63, 0xea, 0x36, 0xa6, 0x89, 0x1a, 0x66, 0xc0, 0x4e, 0xf4, 0x1e, 0x02, 0x43, 0xde, 0xde, 0xc8, 0x9e, 0x88, 0x6e, 0x32, 0xd4, 0xcb, 0x47, 0x24, 0x7c, 0x28, 0x38, 0xd4, 0x95, 0xb6, 0xa3, 0x91, 0x69, 0xc7, 0x8d, 0xfd, 0x15, 0xf5, 0xbf, 0xb1, 0x98, 0x8c, 0x57, 0x51, 0xbf, 0x83, 0x6a, 0x35, 0x10, 0x03, 0x50, 0xe5, 0xf7, 0xfa, 0xf8, 0xa5, 0xb0, 0xdb, 0xfb, 0x42, 0x93, 0xbb, 0x17, 0xf7, 0x36, 0xbe, 0x26, 0x66, 0x61, 0xe2
|
||||
|
||||
db 0xaf, 0xe0, 0x22, 0x25, 0x2a, 0xae, 0x78, 0x67, 0x8f, 0x7e, 0x9e, 0x59, 0xd7, 0xa3, 0x71, 0xcc, 0x43, 0x85, 0x09, 0xf9, 0x18, 0x52, 0x7b, 0x01, 0x73, 0xcb, 0x31, 0x18, 0x66, 0x79, 0x67, 0x10, 0x67, 0xd8, 0xdf, 0x43, 0xaf, 0x2d, 0x9a, 0x09, 0x9c, 0xd1, 0x37, 0x7e, 0xf5, 0x1c, 0x3c, 0x4f, 0x15, 0xe1, 0x6f, 0xfd, 0x13, 0x3d, 0x53, 0x81, 0xa9, 0x93, 0x5f, 0x92, 0x41, 0x48, 0xec, 0x87, 0x87, 0x1d, 0x0b, 0xaa, 0xaa, 0xd3, 0xc2, 0x98, 0x20, 0xce, 0x28, 0xaf, 0x9d, 0x84, 0x69, 0x4a, 0xfd, 0xc0, 0x9c, 0x2e, 0x50, 0x20, 0xb2, 0x00, 0xc1, 0x81, 0x2a, 0x32, 0x8e, 0x95, 0x20, 0xa7, 0xca, 0x39, 0x28, 0x12, 0x23, 0x0e, 0x43, 0xd3, 0x82, 0x76, 0x73, 0x3c, 0xbf, 0xa9, 0x98, 0xf6, 0x39, 0x6d, 0xd9, 0x15, 0x33, 0x1e, 0x07, 0x7c, 0x08, 0x12, 0x23, 0xbd, 0xd3, 0x34, 0x2d, 0x9a, 0x23, 0x21, 0x46, 0xf3, 0x9a, 0x04, 0x25, 0x62, 0xeb, 0x7e, 0x9a, 0xaa, 0xb6, 0x26, 0xaa, 0x85, 0x01, 0x3a, 0xd8, 0xfc, 0x57, 0x98, 0xb9, 0xe4, 0xc4, 0xe9, 0x11, 0x3e, 0x22, 0x95, 0x3b, 0x41, 0x2b, 0x02, 0x04, 0x6c, 0x75, 0xa5, 0xf2, 0xaa, 0x09, 0x9e, 0x6f, 0xab, 0x1d, 0x2a, 0x5c, 0xde, 0x21, 0xb1, 0x96, 0x2d, 0x86, 0x3f, 0xd0, 0x07, 0x18, 0x1f, 0x87, 0xc2, 0x8f, 0xdf, 0x6a, 0x57, 0x6d, 0x3f, 0x80, 0xc5, 0x08, 0x19, 0xa5, 0x09, 0x65, 0x3d, 0xdc, 0x9e, 0x80, 0x3c, 0x2a, 0x0e, 0x7a, 0x40, 0x04, 0x0b, 0xcc, 0x61, 0xdb, 0x73, 0xfc, 0xa5, 0x0a, 0x42, 0x18, 0xc1, 0xd5, 0xbd, 0x18, 0x78, 0xa1, 0xe4, 0xde, 0x44, 0xec, 0x79, 0xb0, 0x27, 0xaa, 0x45, 0x21, 0x57, 0x19, 0x75, 0x09, 0x5c, 0x58, 0xd5, 0xb9, 0x6f, 0x3b, 0x48, 0x59, 0x41, 0x3e, 0xfd, 0x17, 0x43, 0x27, 0xc3, 0x8d, 0x76, 0x8e, 0x38, 0x47, 0xe7, 0xd2, 0xea, 0x54, 0x73, 0x8a, 0x65, 0x4c, 0x49, 0x91, 0xaf, 0x29, 0x65, 0x0d, 0x81, 0xa4, 0x77, 0xd7, 0x32, 0xd0, 0x69, 0xd9, 0x6b, 0xa3, 0x9b, 0x24, 0xd6, 0x0a, 0xd2, 0x77, 0x38, 0x59, 0x0b, 0xc8, 0x5c, 0xc7, 0x0b, 0x1d, 0xd1, 0xfa, 0xa7, 0x45, 0x3c, 0xeb, 0x5c, 0x8e, 0x25, 0x35, 0x81, 0x6d, 0x6d, 0xfe, 0xb4, 0x63, 0x89, 0xe4, 0xf0, 0xa8, 0xda, 0xb7, 0xd4, 0xff, 0x5d, 0x28, 0x97, 0x11, 0xf9, 0x8d, 0xab, 0x29, 0xd5, 0xd3, 0x1c, 0x70, 0x20, 0x4c, 0x41, 0x16, 0x42, 0xfd, 0xfc, 0x62, 0x82, 0x40, 0x59, 0x34, 0x28, 0xd0, 0xd5, 0xfc, 0xac, 0x97, 0xb8, 0x82, 0x0e, 0x4b, 0xae, 0x51, 0x28, 0x1a, 0xf1, 0x87, 0xd3, 0x20, 0xa3, 0xe7, 0x74, 0x69, 0x3c, 0x54, 0x8d, 0xc5, 0x56, 0x1d, 0xcd, 0x75, 0xae, 0x88, 0x17, 0x30, 0xdf, 0x46, 0x4a, 0xbc, 0x64, 0xff, 0xa2, 0xc1, 0x62, 0xbd, 0x88, 0x7b, 0x3e, 0xa1, 0x0c, 0xa9, 0x13, 0x0e, 0xc1, 0xb4, 0x24, 0xe6, 0x96, 0x1b, 0x9c, 0x9b, 0xac, 0x44, 0x33, 0x5b, 0xda, 0xd5, 0x88, 0x4d, 0xfe, 0x81, 0x09, 0x07, 0x17, 0xcf, 0x14, 0x05, 0xaf, 0xf8, 0x72, 0x14, 0x49, 0x5f, 0x06, 0x62, 0xab, 0xe0, 0x42, 0x70, 0x12, 0x59, 0x41, 0x0f, 0x18, 0x83, 0x68, 0x6d, 0xc6, 0x3c, 0xea, 0xe0, 0x6d, 0xd4, 0xae, 0xa6, 0xf1, 0x63, 0x21, 0x7f, 0xb5, 0x9d, 0x22, 0xf4, 0xd2, 0x49, 0x49, 0xed, 0x07, 0xb1, 0x11, 0xf9, 0x2e, 0x74, 0xbe, 0x35, 0x47, 0xdc, 0xef, 0x85, 0x0b, 0x4d, 0x46, 0xe6, 0x1f, 0x60, 0x6a, 0xa1, 0x8a, 0x4d, 0x46, 0x87, 0x30, 0x8e, 0x9a, 0xba, 0x97, 0x3e, 0x15, 0xb7, 0x33, 0x76, 0x81, 0x69, 0xdb, 0x82, 0x5e, 0xe6, 0x7b, 0xec, 0xd2, 0x80, 0x7f, 0x17, 0xf1, 0x73, 0xe2
|
||||
|
||||
.data_mid:
|
||||
db 0xd7, 0xa9, 0x35, 0x61, 0x5f, 0xb8, 0xb6, 0x2d, 0x29, 0x34, 0x63, 0xbf, 0xe2, 0x1c, 0x34, 0xe9, 0xf5, 0xff, 0x34, 0x8a, 0x2f, 0xea, 0xd4, 0x3f, 0x3b, 0xfe, 0x6e, 0xdf, 0xa6, 0xd8, 0xc6, 0xb8, 0xc5, 0xff, 0x12, 0x97, 0x53, 0xda, 0x86, 0xa8, 0x0a, 0x12, 0x4e, 0x5d, 0x96, 0x65, 0x51, 0x22, 0xe2, 0x9d, 0x08, 0x71, 0x84, 0x19, 0x8b, 0xbf, 0x29, 0xd3, 0x3f, 0xab, 0xde, 0xe4, 0x27, 0x8b, 0x99, 0xcc, 0xb1, 0x7c, 0xa5, 0x71, 0x91, 0x9a, 0x0b, 0xad, 0x75, 0x86, 0xe3, 0x9c, 0x4e, 0x0c, 0x01, 0xb3, 0x12, 0x33, 0x90, 0x81, 0x7c, 0x71, 0x2c, 0x70, 0x61, 0xd5, 0x39, 0x0c, 0x45, 0xfc, 0x27, 0xaf, 0xbb, 0xd9, 0x26, 0x1b, 0x33, 0xb4, 0x0d, 0xf8, 0xd6, 0x2d, 0x09, 0xc7, 0x8c, 0xbf, 0x48, 0x53, 0x14, 0x94, 0x76, 0x25, 0xc7, 0x0c, 0x69, 0x49, 0x82, 0xb4, 0x2f, 0x48, 0x38, 0x44, 0x9d, 0x90, 0x6d, 0x66, 0x35, 0xe9, 0x3e, 0x2f, 0x2a, 0xb7, 0xe1, 0xb1, 0x2b, 0x99, 0x08, 0x6f, 0x5c, 0x6c, 0xdf, 0xdb, 0x10, 0xe2, 0xaa, 0x86, 0xe7, 0xf8, 0x9e, 0x62, 0xde, 0xa5, 0x81, 0x6b, 0x20, 0x47, 0xa9, 0x06, 0x49, 0xc0, 0x78, 0x8c, 0x70, 0x93, 0x7e, 0xda, 0xda, 0x5e, 0x3b, 0x23, 0xf9, 0xcc, 0x87, 0xdf, 0x48, 0x4f, 0xd6, 0x77, 0xce, 0x45, 0xe1, 0xdc, 0x0c, 0x7a, 0x0c, 0x50, 0x15, 0x63, 0x8c, 0x48, 0xd3, 0x8e, 0xfa, 0xcc, 0xac, 0x1a, 0x83, 0xde, 0xb1, 0x87, 0x2a, 0x58, 0x5c, 0xa5, 0x20, 0x3d, 0xaa, 0x1e, 0x5d, 0x71, 0xa6, 0x57, 0x75, 0x82, 0xb7, 0x33, 0x9e, 0x6b, 0xf3, 0x35, 0x02, 0x98, 0x03, 0xe1, 0x3b, 0xd2, 0x9f, 0x7a, 0x06, 0x85, 0xef, 0x7d, 0xd9, 0xf2, 0x0c, 0x9e, 0xce, 0xb9, 0xce, 0x13, 0x4a, 0x9e, 0x8a, 0x29, 0xe6, 0xe5, 0xe4, 0x39, 0xba, 0xfd, 0xa3, 0x33, 0xa8, 0x13, 0x9e, 0xa5, 0x11, 0x37, 0x69, 0xbc, 0xda, 0x11, 0x49, 0x2d, 0x4a, 0xef, 0x20, 0x8b, 0x7a, 0xb8, 0x9c, 0xc3, 0xaf, 0x26, 0x71, 0xd9, 0xa2, 0xf6, 0x0f, 0x85, 0x87, 0xa8, 0x6c, 0xf9, 0x99, 0xa2, 0xb2, 0x36, 0x2d, 0x78, 0x10, 0xe4, 0x33, 0x8d, 0xa4, 0x63, 0xea, 0x02, 0xb9, 0xac, 0x2f, 0x90, 0x39, 0x2d, 0x0e, 0x2e, 0xf5, 0x08, 0xa5, 0x5c, 0x8e, 0x71, 0x30, 0x0d, 0x1b, 0x84, 0x7a, 0xd7, 0xd4, 0xab, 0x81, 0x82, 0x18, 0x37, 0xf3, 0x28, 0x6f, 0x4e, 0x28, 0x71, 0xda, 0xc9, 0x99, 0x46, 0x14, 0x46, 0x77, 0x01, 0x16, 0x21, 0xae, 0x83, 0x93, 0x86, 0x7f, 0x5a, 0xee, 0xd5, 0xdf, 0x48, 0x5b, 0x15, 0xc8, 0x09, 0x30, 0x8f, 0x01, 0xcc, 0x95, 0x30, 0xd9, 0xf7, 0x72, 0x97, 0xfd, 0x9d, 0xec, 0x9f, 0xbf, 0x5c, 0xbf, 0x4f, 0xca, 0x33, 0xb4, 0xd2, 0xa2, 0xb9, 0x08, 0x9c, 0x40, 0x25, 0x3f, 0x86, 0xdc, 0x83, 0x70, 0x2f, 0xfb, 0x2a, 0xf8, 0x61, 0x1f, 0xa1, 0x1f, 0x36, 0x04, 0xe2, 0xef, 0x1c, 0xa4, 0xcd, 0x3c, 0x7f, 0xc5, 0x73, 0x9c, 0x2e, 0xeb, 0x03, 0x79, 0xd1, 0x02, 0xfc, 0x6f, 0xbd, 0x5a, 0x95, 0xb2, 0xf6, 0x25, 0x96, 0xe6, 0x80, 0x0a, 0xc5, 0xc7, 0xca, 0x8d, 0x31, 0xae, 0xf0, 0x49, 0xcf, 0x43, 0x06, 0x27, 0x7f, 0x25, 0xc7, 0x4c, 0xb7, 0xfc, 0x73, 0xd3, 0x04, 0xd3, 0xb9, 0x9f, 0x74, 0xed, 0x9e, 0x3c, 0xf0, 0xcf, 0x26, 0x2b, 0xd9, 0xcb, 0x78, 0x2a, 0xef, 0x72, 0xf7, 0xb6, 0x78, 0x30, 0x2d, 0x8c, 0x83, 0x73, 0x66, 0x74, 0x3d, 0x66, 0x0a, 0x74, 0x5a, 0x3f, 0x9f, 0x6e, 0x56, 0x68, 0x01, 0xc2, 0xca, 0x2b, 0xa1, 0x25, 0x36, 0x9c, 0x3b, 0xa4, 0x5e, 0x44, 0xf1, 0x18, 0x1d, 0xb6, 0x1a, 0x3a, 0xee, 0x8d, 0x67, 0x34, 0x9c
|
||||
|
||||
db 0xdd, 0x48, 0x14, 0xc2, 0x5f, 0xd8, 0xe5, 0x71, 0x22, 0xbf, 0xbc, 0x84, 0xda, 0xc1, 0xb1, 0x22, 0x55, 0xa4, 0x63, 0x41, 0x77, 0xac, 0x40, 0x2d, 0x44, 0x73, 0x8c, 0x14, 0xba, 0x5e, 0x63, 0x68, 0x65, 0x61, 0x6d, 0xec, 0xe2, 0x6d, 0x37, 0x22, 0x04, 0xeb, 0xc7, 0xd4, 0xc9, 0x62, 0x56, 0x13, 0x96, 0x29, 0x03, 0xf4, 0x55, 0xe2, 0x58, 0x7d, 0xda, 0x52, 0x2e, 0x94, 0x07, 0xe6, 0xef, 0xc0, 0xee, 0x9e, 0x0b, 0xf7, 0xcd, 0x13, 0x8b, 0x7d, 0xea, 0xdc, 0xf8, 0xf1, 0xcb, 0xad, 0x49, 0x97, 0xc9, 0x98, 0x0b, 0xcf, 0x84, 0x8e, 0x8e, 0xbb, 0x06, 0x2e, 0x54, 0xf5, 0xa7, 0xbd, 0x70, 0x7e, 0x38, 0x69, 0x8d, 0xb0, 0x01, 0x7b, 0x41, 0x80, 0x09, 0x44, 0xfd, 0x7e, 0x21, 0xb4, 0xbe, 0x6b, 0x4a, 0xb7, 0xca, 0x2d, 0x19, 0xfe, 0x6d, 0xd6, 0x11, 0x29, 0xbb, 0xb2, 0x16, 0xf1, 0xe7, 0x92, 0x71, 0xda, 0x7e, 0x68, 0x3a, 0xe0, 0xea, 0x89, 0x8d, 0xe0, 0x44, 0x48, 0x25, 0x92, 0x37, 0x54, 0x26, 0xf2, 0xab, 0xb3, 0x3b, 0xdb, 0xbb, 0x2b, 0x5c, 0xf5, 0xbc, 0xc7, 0x97, 0xdb, 0xc7, 0x49, 0x25, 0x7c, 0xc2, 0x80, 0x02, 0x69, 0xd4, 0xda, 0xda, 0xe1, 0x04, 0xf3, 0x19, 0xb8, 0xc9, 0xb2, 0xfb, 0x1e, 0x47, 0xa9, 0x0c, 0xa3, 0x48, 0xce, 0xc2, 0x9e, 0x3b, 0x28, 0x23, 0x5a, 0x20, 0x44, 0x77, 0x40, 0xe2, 0xd7, 0x20, 0xd5, 0x71, 0x6f, 0xd4, 0x3c, 0x68, 0x38, 0x9b, 0x89, 0x2e, 0x2d, 0xa8, 0x1f, 0x99, 0xb5, 0x8a, 0x66, 0x07, 0x59, 0x75, 0x9e, 0xf8, 0xd9, 0xbe, 0x85, 0x6a, 0x20, 0x92, 0x9d, 0xd2, 0x5e, 0x45, 0xc0, 0x60, 0xbe, 0x85, 0x0b, 0x84, 0x47, 0xf5, 0xa8, 0x43, 0x87, 0xf1, 0x21, 0x21, 0xb0, 0x3b, 0x04, 0x13, 0x16, 0x3e, 0xdf, 0xc3, 0xc6, 0x04, 0x73, 0xcd, 0x92, 0x76, 0xfb, 0xe7, 0x9c, 0xd3, 0x46, 0x11, 0x78, 0xca, 0x12, 0xd9, 0x4a, 0x35, 0xf1, 0x6e, 0x89, 0x8b, 0xe9, 0x7a, 0x04, 0xba, 0x18, 0x25, 0x7c, 0x9e, 0xe6, 0x4f, 0xc2, 0x56, 0x05, 0x72, 0xc3, 0x76, 0xee, 0x7d, 0x77, 0x19, 0x7a, 0x73, 0x2c, 0x81, 0xb8, 0xc7, 0xd9, 0x7f, 0x17, 0x5d, 0x30, 0xda, 0x77, 0x3c, 0x14, 0x88, 0xe8, 0xe4, 0xbf, 0xee, 0x21, 0x1c, 0x29, 0x4e, 0x58, 0xa8, 0x8a, 0x5c, 0xae, 0xa2, 0x1c, 0x7c, 0x25, 0x7c, 0x1c, 0x39, 0xa4, 0x28, 0x4b, 0x78, 0x52, 0xae, 0x2c, 0xbb, 0x5f, 0xbf, 0x51, 0x09, 0x20, 0x76, 0xb2, 0x7d, 0xb1, 0x63, 0x84, 0xc5, 0x49, 0x8a, 0x73, 0xdb, 0x76, 0x1d, 0x25, 0x31, 0xf2, 0x1e, 0x19, 0x38, 0xc8, 0x3b, 0x51, 0x3c, 0x13, 0x52, 0x84, 0xae, 0xc2, 0xe4, 0x8a, 0x57, 0x0d, 0xde, 0x8d, 0x18, 0x48, 0x9a, 0xbd, 0xbf, 0xf3, 0xea, 0x79, 0x17, 0x06, 0x96, 0x72, 0x08, 0x60, 0x95, 0xf9, 0x6f, 0x25, 0x0c, 0xb7, 0x9d, 0x98, 0x23, 0x01, 0xc8, 0x7a, 0xdb, 0x75, 0x63, 0x64, 0x14, 0x5e, 0x10, 0xf5, 0x16, 0x48, 0xbc, 0xc6, 0x7e, 0x24, 0xf3, 0xad, 0x57, 0x3f, 0x7d, 0x6c, 0xab, 0x18, 0x8c, 0x12, 0xc5, 0x0c, 0xd8, 0xb5, 0x1e, 0x43, 0x7c, 0x23, 0x17, 0x48, 0xba, 0x76, 0x3b, 0xd9, 0x2b, 0xae, 0x1b, 0xef, 0x58, 0xfa, 0x87, 0xad, 0x9b, 0x6d, 0xf9, 0xab, 0xa8, 0x3c, 0xfc, 0x59, 0x67, 0xa6, 0x2c, 0xc7, 0x75, 0xa4, 0x97, 0xca, 0x18, 0x18, 0x04, 0x2c, 0xb3, 0x0e, 0xa9, 0x69, 0x33, 0x67, 0xa2, 0xc6, 0xbc, 0x98, 0x48, 0x71, 0x11, 0x05, 0x30, 0xf6, 0xa9, 0x61, 0x40, 0x46, 0xf1, 0x41, 0x37, 0xd0, 0x6b, 0x7c, 0x1f, 0x03, 0x5c, 0xe9, 0xf4, 0x59, 0x1d, 0x35, 0xf0, 0x98, 0x42, 0x4a, 0x92, 0x2a, 0xc3, 0x9a, 0xb8, 0xa5
|
||||
@@ -3,7 +3,8 @@
|
||||
"RegData": {
|
||||
"XMM0": ["0x4142434445464748", "0x5152535455565758"],
|
||||
"XMM1": ["0x4142434445464748", "0x5758575857585758"],
|
||||
"XMM2": ["0x6162636465666768", "0x7172717271727172"]
|
||||
"XMM2": ["0x6162636465666768", "0x7172717271727172"],
|
||||
"XMM3": ["0x4142434445464748", "0x5556555657585758"]
|
||||
},
|
||||
"MemoryRegions": {
|
||||
"0x100000000": "4096"
|
||||
@@ -27,4 +28,7 @@ movapd xmm0, [rdx]
|
||||
pshufhw xmm1, xmm0, 0x0
|
||||
pshufhw xmm2, [rdx + 8 * 2], 0xFF
|
||||
|
||||
; Top bit different from low bits
|
||||
pshufhw xmm3, [rdx], 80
|
||||
|
||||
hlt
|
||||
@@ -4,7 +4,8 @@
|
||||
"XMM0": ["0x4142434445464748", "0x5152535455565758"],
|
||||
"XMM1": ["0x6162636465666768", "0x7172737475767778"],
|
||||
"XMM2": ["0x4748474847484748", "0x5152535455565758"],
|
||||
"XMM3": ["0x6162616261626162", "0x7172737475767778"]
|
||||
"XMM3": ["0x6162616261626162", "0x7172737475767778"],
|
||||
"XMM4": ["0x4546454647484748", "0x5152535455565758"]
|
||||
},
|
||||
"MemoryRegions": {
|
||||
"0x100000000": "4096"
|
||||
@@ -29,4 +30,7 @@ movapd xmm1, [rdx + 8 * 2]
|
||||
pshuflw xmm2, xmm0, 0x0
|
||||
pshuflw xmm3, xmm1, 0xFF
|
||||
|
||||
; Top bit different from low bits
|
||||
pshuflw xmm4, [rdx], 80
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,44 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"HostFeatures": ["AVX"],
|
||||
"RegData": {
|
||||
"XMM0": ["0xc01c000000000000", "0xc035800000000000", "0xc073000000000000", "0xc083f00000000000"],
|
||||
"XMM1": ["0x4045c00000000000", "0x4063400000000000", "0xc0b7cf8000000000", "0xc0d2b66000000000"],
|
||||
"XMM2": ["0x4040800000000000", "0x4053b00000000000", "0xc0959e0000000000", "0xc0b5448000000000"],
|
||||
"XMM3": ["0xc01c000000000000", "0xc035800000000000", "0", "0"],
|
||||
"XMM4": ["0x4045c00000000000", "0x4063400000000000", "0", "0"],
|
||||
"XMM5": ["0x4040800000000000", "0x4053b00000000000", "0", "0"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
vmovups ymm0, [rel .data]
|
||||
vmovups ymm1, [rel .data2]
|
||||
vmovups ymm2, [rel .data3]
|
||||
|
||||
vmovups ymm3, [rel .data]
|
||||
vmovups ymm4, [rel .data2]
|
||||
vmovups ymm5, [rel .data3]
|
||||
|
||||
vfmadd231pd ymm0, ymm1, ymm2
|
||||
vfmadd213pd ymm1, ymm0, ymm2
|
||||
vfmadd132pd ymm2, ymm1, ymm0
|
||||
|
||||
vfmadd231pd xmm3, xmm4, xmm5
|
||||
vfmadd213pd xmm4, xmm3, xmm5
|
||||
vfmadd132pd xmm5, xmm4, xmm3
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
.data:
|
||||
dq 2.0, 3.0
|
||||
dq 6.0, 7.0
|
||||
|
||||
.data2:
|
||||
dq -6.0, -7.0
|
||||
dq 20.0, 30.0
|
||||
|
||||
.data3:
|
||||
dq 1.5, 3.5
|
||||
dq -15.5, -21.5
|
||||
@@ -0,0 +1,44 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"HostFeatures": ["AVX"],
|
||||
"RegData": {
|
||||
"XMM0": ["0xc1ac0000c0e00000", "0x4294999942400000", "0xc41f8000c3980000", "0x44bd4000446d0000"],
|
||||
"XMM1": ["0x431a0000422e0000", "0xc4291999c3c2c000", "0xc695b300c5be7c00", "0x4793e90d47143780"],
|
||||
"XMM2": ["0x429d800042040000", "0xc49c1051c4236000", "0xc5aa2400c4acf000", "0x47eceac0476b3d80"],
|
||||
"XMM3": ["0xc1ac0000c0e00000", "0x4294999942400000", "0", "0"],
|
||||
"XMM4": ["0x431a0000422e0000", "0xc4291999c3c2c000", "0", "0"],
|
||||
"XMM5": ["0x429d800042040000", "0xc49c1051c4236000", "0", "0"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
vmovups ymm0, [rel .data]
|
||||
vmovups ymm1, [rel .data2]
|
||||
vmovups ymm2, [rel .data3]
|
||||
|
||||
vmovups ymm3, [rel .data]
|
||||
vmovups ymm4, [rel .data2]
|
||||
vmovups ymm5, [rel .data3]
|
||||
|
||||
vfmadd231ps ymm0, ymm1, ymm2
|
||||
vfmadd213ps ymm1, ymm0, ymm2
|
||||
vfmadd132ps ymm2, ymm1, ymm0
|
||||
|
||||
vfmadd231ps xmm3, xmm4, xmm5
|
||||
vfmadd213ps xmm4, xmm3, xmm5
|
||||
vfmadd132ps xmm5, xmm4, xmm3
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
.data:
|
||||
dd 2.0, 3.0, 4.0, 5.0
|
||||
dd 6.0, 7.0, 8.0, 9.0
|
||||
|
||||
.data2:
|
||||
dd -6.0, -7.0, -8.0, -9.0
|
||||
dd 20.0, 30.0, 40.0, 50.0
|
||||
|
||||
.data3:
|
||||
dd 1.5, 3.5, -5.5, -7.7
|
||||
dd -15.5, -21.5, 23.5, 30.1
|
||||
@@ -0,0 +1,33 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"HostFeatures": ["AVX"],
|
||||
"RegData": {
|
||||
"XMM0": ["0xc01c000000000000", "0x4008000000000000", "0", "0"],
|
||||
"XMM1": ["0x4045c00000000000", "0xc01c000000000000", "0", "0"],
|
||||
"XMM2": ["0x4040800000000000", "0x400c000000000000", "0", "0"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
vmovups ymm0, [rel .data]
|
||||
vmovups ymm1, [rel .data2]
|
||||
vmovups ymm2, [rel .data3]
|
||||
|
||||
vfmadd231sd xmm0, xmm1, xmm2
|
||||
vfmadd213sd xmm1, xmm0, xmm2
|
||||
vfmadd132sd xmm2, xmm1, xmm0
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
.data:
|
||||
dq 2.0, 3.0
|
||||
dq 6.0, 7.0
|
||||
|
||||
.data2:
|
||||
dq -6.0, -7.0
|
||||
dq 20.0, 30.0
|
||||
|
||||
.data3:
|
||||
dq 1.5, 3.5
|
||||
dq -15.5, -21.5
|
||||
Loaded 100 of 323 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user