mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-09 07:00:18 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ba1b4744c5 | ||
|
|
bd13c02451 | ||
|
|
5902b175f9 | ||
|
|
d0e47f9073 | ||
|
|
bd7215d36f | ||
|
|
f3f134f9de | ||
|
|
cdbf5d57bd | ||
|
|
1ae50bd670 | ||
|
|
d28c9f9843 | ||
|
|
fe7d52aa78 | ||
|
|
fc0907f8c1 | ||
|
|
e57678d7a6 | ||
|
|
45e594e806 | ||
|
|
87e7a0effa | ||
|
|
4fd1a35b2a | ||
|
|
c460cf0678 | ||
|
|
983802da61 | ||
|
|
49273e0d59 | ||
|
|
76b8459cdc | ||
|
|
e269eb6f65 | ||
|
|
7b4774f375 | ||
|
|
70e9a25112 | ||
|
|
9fb83ea56a | ||
|
|
8bb3398376 | ||
|
|
d42fbb3d4d | ||
|
|
53b2245dc1 | ||
|
|
db14975828 | ||
|
|
a5139d2710 | ||
|
|
9f584c8014 | ||
|
|
6f1b98fb52 | ||
|
|
c258a90505 | ||
|
|
eb47ef43a7 | ||
|
|
1876d6b923 | ||
|
|
90c8fcf393 | ||
|
|
6af90575e9 | ||
|
|
994613260c | ||
|
|
1430fa8220 | ||
|
|
33e06058c6 | ||
|
|
b032d1e1f7 | ||
|
|
d6b43b1fe6 | ||
|
|
fc771c8683 | ||
|
|
f91ac09f87 | ||
|
|
e4fa399412 | ||
|
|
952e949e10 | ||
|
|
3d093d66fb | ||
|
|
05fe2893c7 | ||
|
|
6fc17294b6 | ||
|
|
6607921bee | ||
|
|
0b0793438f | ||
|
|
3dd591e760 | ||
|
|
f290d2f899 | ||
|
|
a676ad7193 | ||
|
|
096c408ef6 | ||
|
|
00b65f76b6 | ||
|
|
39fb266282 | ||
|
|
3969d0ac78 | ||
|
|
439c6bb3c0 | ||
|
|
5cedbf9d34 | ||
|
|
427b235eb5 | ||
|
|
92d5ba580f | ||
|
|
bc2f331c8b | ||
|
|
ca58aef676 | ||
|
|
379dc405f6 | ||
|
|
d74b5c42da | ||
|
|
6004971439 | ||
|
|
a12b8927bc | ||
|
|
57e23b289a | ||
|
|
8214ffccf0 | ||
|
|
547135dc2d | ||
|
|
9c72113161 | ||
|
|
8cc967fa22 | ||
|
|
4bd30bb72d | ||
|
|
5eeb4dabbd | ||
|
|
a27c4b3860 | ||
|
|
dfee08f74f | ||
|
|
0e2629bdd4 | ||
|
|
05c8630b07 | ||
|
|
1cffe618d2 | ||
|
|
98c7bb23b5 | ||
|
|
922853cee1 | ||
|
|
a251e61859 | ||
|
|
9c19799023 | ||
|
|
f423b110a8 | ||
|
|
6fd471e652 | ||
|
|
3d69029d33 | ||
|
|
2258f2e424 | ||
|
|
cca5a68e20 | ||
|
|
da5c9bff68 | ||
|
|
5b87f0699b | ||
|
|
a67fe561a1 | ||
|
|
e2f4065376 | ||
|
|
d3bf87f4f4 | ||
|
|
6d351ec47f | ||
|
|
4dc1dd2511 | ||
|
|
5205ae40fa | ||
|
|
32f1dcde7e | ||
|
|
6772581c53 | ||
|
|
387201815b | ||
|
|
24d61e1125 | ||
|
|
04259f031d | ||
|
|
6403da3715 | ||
|
|
90cb76312c | ||
|
|
b6cff01abb | ||
|
|
b34b711161 | ||
|
|
709d767d61 | ||
|
|
e075916154 | ||
|
|
de10154f29 | ||
|
|
ba71e79e54 | ||
|
|
2cc70b8051 | ||
|
|
9d965f94de | ||
|
|
40c2db4744 | ||
|
|
9a7285dca4 | ||
|
|
8c00ac78b1 | ||
|
|
cf4478eeee | ||
|
|
85c8e7f1bb | ||
|
|
efd95efb40 | ||
|
|
99ad7ea45c | ||
|
|
aba0c57f73 | ||
|
|
8c4f6b648e | ||
|
|
3b83bdd88d | ||
|
|
8d71e08b44 | ||
|
|
c31063a8ef | ||
|
|
28d101f1dd | ||
|
|
aaef344ae3 | ||
|
|
42d0324304 | ||
|
|
2a0019347a | ||
|
|
e0305ea1b9 | ||
|
|
105ff47ae3 | ||
|
|
5ee190a41e | ||
|
|
cf37617c25 | ||
|
|
2e9c8f0f51 | ||
|
|
06c2319851 | ||
|
|
da0668c7cc | ||
|
|
b34df334cb | ||
|
|
9e9f2ccae1 | ||
|
|
15b8f75730 | ||
|
|
9d6b9aa574 | ||
|
|
64724886af | ||
|
|
c088369f4a | ||
|
|
39dbf46422 | ||
|
|
3c1b0bb917 | ||
|
|
2227170dbb | ||
|
|
e08f421e1c | ||
|
|
42c58c5420 | ||
|
|
4afbdd9afb | ||
|
|
06b9e13904 | ||
|
|
29473b43cb | ||
|
|
0427d48b98 | ||
|
|
b228746f1d | ||
|
|
0be8485116 | ||
|
|
5d0279ff08 | ||
|
|
5d908d902c | ||
|
|
3a014f80f2 | ||
|
|
fbefd7855c | ||
|
|
11f9135be6 | ||
|
|
7bb0ce810e | ||
|
|
993b832771 | ||
|
|
013ac1e627 | ||
|
|
d7977a02fa | ||
|
|
73a32ff22c | ||
|
|
ee4ae5390b | ||
|
|
f71db11035 | ||
|
|
9497288b97 | ||
|
|
a00260d801 | ||
|
|
cad48e07e4 | ||
|
|
d91e8a4278 | ||
|
|
faf74eee90 | ||
|
|
0b52e1cd14 | ||
|
|
b62890f136 | ||
|
|
de1d37eef8 | ||
|
|
a57c557485 | ||
|
|
3c9f6c845b | ||
|
|
ff25e9a92e | ||
|
|
6a60f72a9e | ||
|
|
94b690df43 | ||
|
|
94edbc3436 | ||
|
|
5eab1e559a | ||
|
|
53db3ad6f2 | ||
|
|
581f3263ed | ||
|
|
1e3c642be6 | ||
|
|
22c3cd553f | ||
|
|
e5743f8dae | ||
|
|
bddc2f227d | ||
|
|
686c04ea93 | ||
|
|
b38369199e | ||
|
|
e862c904a9 | ||
|
|
43d9384b1c | ||
|
|
cb9af0b86a | ||
|
|
b8c17a843c | ||
|
|
7ad7f181d7 | ||
|
|
eb0bf55033 | ||
|
|
f4e3e4ad30 | ||
|
|
5ae82410cc | ||
|
|
2febb524e9 | ||
|
|
b9e452133c | ||
|
|
747ea0a1f7 | ||
|
|
f8c52ca34a | ||
|
|
663fd5a98b | ||
|
|
93e58bc15e | ||
|
|
3ccdf6508e | ||
|
|
fd33cf1ce5 | ||
|
|
2bb64ad1c6 | ||
|
|
e3de62058b | ||
|
|
6ee9984280 | ||
|
|
e1df548ae9 | ||
|
|
79a685c15e | ||
|
|
a61ab2803c | ||
|
|
209ad27332 | ||
|
|
df08981475 | ||
|
|
5ce6039a02 | ||
|
|
6c3fdf723a | ||
|
|
1a617c1eb2 | ||
|
|
3a9b801400 | ||
|
|
6e663acdad | ||
|
|
ae1023bb7a | ||
|
|
9cf25e276d | ||
|
|
8db3670ecc | ||
|
|
baee367532 | ||
|
|
8da4e72d87 | ||
|
|
b4a84a2317 | ||
|
|
43d6347212 | ||
|
|
438501e49c | ||
|
|
c034e99aaf | ||
|
|
d2d0ca2de9 | ||
|
|
cbe2b442b2 | ||
|
|
c5b1cd6e7d | ||
|
|
8d20d1dae3 | ||
|
|
1f2d702c4e | ||
|
|
d2d35d0553 | ||
|
|
1e7f54dd7e | ||
|
|
8430a2f7e6 | ||
|
|
326cde78e6 | ||
|
|
bbc2b0b42f | ||
|
|
231a2c54aa | ||
|
|
36ae4cee73 | ||
|
|
62e5ee2201 | ||
|
|
b89ebd931e | ||
|
|
d1b4ddaf61 | ||
|
|
734a0b236b | ||
|
|
2229c04d4d | ||
|
|
07cff27fa2 | ||
|
|
2cf86998bc | ||
|
|
6ed15a6fd6 | ||
|
|
7610243b0c | ||
|
|
e11349b577 | ||
|
|
9b425697cb | ||
|
|
42596ff91e | ||
|
|
b3c2ff47f3 | ||
|
|
75a0bc79be | ||
|
|
5a002ad08d | ||
|
|
fbac6f86d1 | ||
|
|
ca18bf2a3d | ||
|
|
199effdff7 | ||
|
|
8212f4b7fb | ||
|
|
e1a45a2720 | ||
|
|
bd7edd8651 | ||
|
|
4e1d10a46f | ||
|
|
70b6bc2bae | ||
|
|
46d019fe02 | ||
|
|
d56f689e15 | ||
|
|
90702b4102 | ||
|
|
0cf105b64a | ||
|
|
5c74d9458c | ||
|
|
75391bf834 | ||
|
|
63304a1d88 | ||
|
|
1ab79bd72e | ||
|
|
985bdf2b6c | ||
|
|
b57ea83aea | ||
|
|
d716e22476 | ||
|
|
bbb8e1ccab | ||
|
|
96f20779c6 | ||
|
|
908313e378 | ||
|
|
12e5c60633 | ||
|
|
11946ffc4c | ||
|
|
f44cd9c545 | ||
|
|
9c5ccb13de | ||
|
|
e9bcfd4784 | ||
|
|
fd1e8d4566 | ||
|
|
68dcce0739 | ||
|
|
3c554cd787 | ||
|
|
9e5f2269d9 | ||
|
|
81fc502c6c | ||
|
|
eda8ca5449 | ||
|
|
35dd8972e2 | ||
|
|
643dd56b74 | ||
|
|
b748eab4ed | ||
|
|
f4eaab6977 | ||
|
|
900c114831 | ||
|
|
7937b7e52d | ||
|
|
b40e707771 | ||
|
|
edde5c8516 | ||
|
|
197facb845 | ||
|
|
ca697d0d5d | ||
|
|
32ddf790d5 | ||
|
|
2cfba8c6d0 | ||
|
|
cf6c0765fa | ||
|
|
ad93f27271 | ||
|
|
92428e5cbc | ||
|
|
02c00a87dd | ||
|
|
6c77bc12c1 | ||
|
|
fbb428c249 | ||
|
|
854a741ea4 | ||
|
|
ed1952a79a | ||
|
|
39e8f5122f | ||
|
|
6ee77eb1a1 | ||
|
|
46fc45b952 | ||
|
|
79d90f3c7f | ||
|
|
eb41cb2261 | ||
|
|
87e8a9b6aa | ||
|
|
69f5aa8e35 | ||
|
|
d654f55c3c | ||
|
|
eb1689e79f | ||
|
|
cf6472df90 | ||
|
|
bc6295a78d | ||
|
|
ed5774b88f | ||
|
|
40a29ca9a7 | ||
|
|
a1e1838b11 | ||
|
|
4731dbab2f | ||
|
|
f2841ccb5e | ||
|
|
81a474fe12 | ||
|
|
5af2477d00 | ||
|
|
4cbba94a29 | ||
|
|
2a57428314 | ||
|
|
f9ac57205b | ||
|
|
3c6246f99a | ||
|
|
072e7bd241 | ||
|
|
4080dca816 | ||
|
|
c22fb80129 | ||
|
|
123fcd4bea | ||
|
|
e0c17672f2 | ||
|
|
53d484c5f4 | ||
|
|
a2ea767b77 | ||
|
|
4bd51a4be0 | ||
|
|
474f2dc267 | ||
|
|
033310234e | ||
|
|
10835c4b62 | ||
|
|
3e741df77b | ||
|
|
31d4b4a201 | ||
|
|
678985cbf5 | ||
|
|
e83e9cb6b3 | ||
|
|
48a51679b8 | ||
|
|
987fc925ee | ||
|
|
b6cbb23781 | ||
|
|
cbb9017c12 | ||
|
|
832d1e8d2c | ||
|
|
34af7f942b | ||
|
|
5f390c16be | ||
|
|
f36ac0498d | ||
|
|
18bdd9b665 | ||
|
|
5fd3852fd3 | ||
|
|
764432c77c | ||
|
|
55c43229ca | ||
|
|
dbc4daaaa4 | ||
|
|
acdbbec78d | ||
|
|
3247e777c0 | ||
|
|
c6e60ff3f5 | ||
|
|
b7df102919 | ||
|
|
f5d450b95c | ||
|
|
972aaf9f5f | ||
|
|
b98d5f30c0 | ||
|
|
d2e2e189d8 | ||
|
|
17d110575c | ||
|
|
a967976420 | ||
|
|
7e1e1ccccc | ||
|
|
502452602e | ||
|
|
f8ff46f3e3 | ||
|
|
8f572deaed | ||
|
|
601dd1d2b5 | ||
|
|
673e826e46 | ||
|
|
f1f81f9de2 | ||
|
|
8513c02ec1 | ||
|
|
282f091e85 | ||
|
|
8647033029 |
No files matched your search
@@ -0,0 +1,79 @@
|
||||
name: steamrt4 build
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
|
||||
jobs:
|
||||
steamrt4_build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
arch: [[self-hosted, ARM64, distrobox]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name : submodule checkout
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: |
|
||||
rm -Rf ${{runner.workspace}}/build
|
||||
cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
# Setup everything required.
|
||||
- name : distrobox setup
|
||||
run: |
|
||||
distrobox create -Y -i registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306 steamrt4 || true
|
||||
distrobox upgrade steamrt4
|
||||
distrobox enter --name steamrt4 -- sudo apt-get install -y \
|
||||
git cmake ninja-build ccache \
|
||||
lld clang \
|
||||
libclang-dev llvm-dev \
|
||||
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross \
|
||||
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
|
||||
|
||||
- name: Create Build Environment
|
||||
run: distrobox enter --name steamrt4 -- cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: distrobox enter --name steamrt4 -- cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: distrobox enter --name steamrt4 -- cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: install
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DESTDIR: ${{runner.workspace}}/install
|
||||
run: distrobox enter --name steamrt4 -- cmake --build . --config $BUILD_TYPE -t install
|
||||
- name: Upload libraries
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
overwrite: true
|
||||
name: steamrt4_steampipe_depot
|
||||
path: ${{runner.workspace}}/install/*
|
||||
retention-days: 1
|
||||
compression-level: 9
|
||||
+14
-2
@@ -33,6 +33,7 @@ option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling ca
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for the FEXCore profiler (gpuvis, tracy)")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
option(BUILD_STEAM_SUPPORT "Builds FEX for integration into Steam" FALSE)
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
@@ -64,6 +65,10 @@ if (NOT MINGW_BUILD)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (BUILD_STEAM_SUPPORT)
|
||||
add_definitions(-DFEX_STEAM_SUPPORT=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
@@ -476,13 +481,16 @@ add_subdirectory(FEXHeaderUtils/)
|
||||
add_subdirectory(CodeEmitter/)
|
||||
add_subdirectory(FEXCore/)
|
||||
|
||||
if (_M_ARM_64 AND NOT MINGW_BUILD)
|
||||
if (_M_ARM_64 AND NOT MINGW_BUILD AND NOT BUILD_STEAM_SUPPORT)
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
endif()
|
||||
|
||||
# Install the ThunksDB file
|
||||
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
|
||||
@@ -582,6 +590,10 @@ if (BUILD_THUNKS)
|
||||
add_dependencies(uninstall uninstall_guest-libs-32)
|
||||
endif()
|
||||
|
||||
if (BUILD_STEAM_SUPPORT)
|
||||
add_subdirectory(Source/Steam/)
|
||||
endif()
|
||||
|
||||
set(FEX_VERSION_MAJOR "0")
|
||||
set(FEX_VERSION_MINOR "0")
|
||||
set(FEX_VERSION_PATCH "0")
|
||||
|
||||
@@ -36,24 +36,31 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
if (IsADRRange(Imm)) {
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void adr(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADR});
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adr(rd, &Label->Backward);
|
||||
return adr(rd, &Label->Backward);
|
||||
} else {
|
||||
adr(rd, &Label->Forward);
|
||||
return adr(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -62,38 +69,53 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) {
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADRP});
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adrp(rd, &Label->Backward);
|
||||
return adrp(rd, &Label->Backward);
|
||||
} else {
|
||||
adrp(rd, &Label->Forward);
|
||||
return adrp(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
const auto SLocation = reinterpret_cast<int64_t>(Label->Location);
|
||||
const auto ULocation = std::bit_cast<uint64_t>(SLocation);
|
||||
|
||||
const int64_t Imm = SLocation - (GetCursorAddress<int64_t>());
|
||||
const auto UImm = std::bit_cast<uint64_t>(Imm);
|
||||
|
||||
if (IsADRRange(Imm)) {
|
||||
// If the range is in ADR range then we can just use ADR.
|
||||
adr(rd, Label);
|
||||
} else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
return adr(rd, Label);
|
||||
}
|
||||
if (IsADRPRange(Imm)) {
|
||||
const int64_t ADRPImm = (SLocation & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
|
||||
// If the range is in the ADRP range then we can use ADRP.
|
||||
bool NeedsOffset = !IsADRPAligned(reinterpret_cast<uint64_t>(Label->Location));
|
||||
uint64_t AlignedOffset = reinterpret_cast<uint64_t>(Label->Location) & 0xFFFULL;
|
||||
const bool NeedsOffset = !IsADRPAligned(ULocation);
|
||||
const uint64_t AlignedOffset = ULocation & 0xFFFULL;
|
||||
|
||||
// First emit ADRP
|
||||
adrp(rd, ADRPImm >> 12);
|
||||
@@ -102,23 +124,33 @@ public:
|
||||
// Now even an add
|
||||
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset too large");
|
||||
FEX_UNREACHABLE;
|
||||
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Stinky path, we need to load the address as a sequence of movz+movk+movk
|
||||
movz(ARMEmitter::Size::i64Bit, rd, (UImm >> 32) & 0xFFFF, 32);
|
||||
movk(ARMEmitter::Size::i64Bit, rd, (UImm >> 16) & 0xFFFF, 16);
|
||||
movk(ARMEmitter::Size::i64Bit, rd, UImm & 0xFFFF);
|
||||
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
void LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
|
||||
// Emit a register index and a nop. These will be backpatched.
|
||||
// Emit a register index and two nops. These will be backpatched.
|
||||
dc32(rd.Idx());
|
||||
nop();
|
||||
nop();
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
LongAddressGen(rd, &Label->Backward);
|
||||
return LongAddressGen(rd, &Label->Backward);
|
||||
} else {
|
||||
LongAddressGen(rd, &Label->Forward);
|
||||
return LongAddressGen(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -862,12 +894,6 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b001'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
|
||||
@@ -20,23 +20,31 @@ public:
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm);
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(Cond, &Label->Backward);
|
||||
return b(Cond, &Label->Backward);
|
||||
} else {
|
||||
b(Cond, &Label->Forward);
|
||||
return b(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,24 +53,32 @@ public:
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm);
|
||||
}
|
||||
void bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bc(Cond, &Label->Backward);
|
||||
return bc(Cond, &Label->Backward);
|
||||
} else {
|
||||
bc(Cond, &Label->Forward);
|
||||
return bc(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -98,25 +114,32 @@ public:
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
void b(const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void b(ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void b(BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(&Label->Backward);
|
||||
return b(&Label->Backward);
|
||||
} else {
|
||||
b(&Label->Forward);
|
||||
return b(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -126,25 +149,33 @@ public:
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
|
||||
void bl(const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void bl(ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void bl(BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bl(&Label->Backward);
|
||||
return bl(&Label->Backward);
|
||||
} else {
|
||||
bl(&Label->Forward);
|
||||
return bl(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -155,28 +186,35 @@ public:
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbz(s, rt, &Label->Backward);
|
||||
return cbz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbz(s, rt, &Label->Forward);
|
||||
return cbz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -186,28 +224,35 @@ public:
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbnz(s, rt, &Label->Backward);
|
||||
return cbnz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbnz(s, rt, &Label->Forward);
|
||||
return cbnz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -217,28 +262,35 @@ public:
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbz(rt, Bit, &Label->Backward);
|
||||
return tbz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbz(rt, Bit, &Label->Forward);
|
||||
return tbz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -247,27 +299,34 @@ public:
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbnz(rt, Bit, &Label->Backward);
|
||||
return tbnz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbnz(rt, Bit, &Label->Forward);
|
||||
return tbnz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -586,6 +586,15 @@ concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegi
|
||||
template<typename T>
|
||||
concept IsQOrDRegister = std::is_same_v<T, QRegister> || std::is_same_v<T, DRegister>;
|
||||
|
||||
template<typename T>
|
||||
concept IsLabel = std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>;
|
||||
|
||||
enum class BranchEncodeSucceeded {
|
||||
Success,
|
||||
Failure,
|
||||
};
|
||||
|
||||
// Whether or not a given set of vector registers are sequential
|
||||
// in increasing order as far as the register file is concerned (modulo its size)
|
||||
//
|
||||
@@ -638,19 +647,25 @@ public:
|
||||
|
||||
// Bind a backward label to an address.
|
||||
// Address that is bound is the current emitter location.
|
||||
void Bind(BackwardLabel* Label) {
|
||||
[[nodiscard]] bool Bind(BackwardLabel* Label) {
|
||||
LOGMAN_THROW_A_FMT(Label->Location == nullptr, "Trying to bind a label twice");
|
||||
Label->Location = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Always binds because it is only storing a location.
|
||||
return true;
|
||||
}
|
||||
|
||||
void Bind(const ForwardLabel::Reference* Label) {
|
||||
[[nodiscard]] bool Bind(const ForwardLabel::Reference* Label) {
|
||||
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
|
||||
// Patch up the instructions
|
||||
switch (Label->Type) {
|
||||
case ForwardLabel::InstType::ADR: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
if (!IsADRRange(Imm)) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
@@ -662,7 +677,12 @@ public:
|
||||
case ForwardLabel::InstType::ADRP: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
|
||||
Imm >>= 12;
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
@@ -672,11 +692,13 @@ public:
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::InstType::B: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FF'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
@@ -686,11 +708,13 @@ public:
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::InstType::TEST_BRANCH: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
@@ -704,7 +728,10 @@ public:
|
||||
case ForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x7'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
@@ -714,38 +741,44 @@ public:
|
||||
break;
|
||||
}
|
||||
case ForwardLabel::InstType::LONG_ADDRESS_GEN: {
|
||||
uint32_t* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
auto OriginalOffset = GetCursorOffset();
|
||||
const auto* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
const auto ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
const auto ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
const auto ImmInstThree = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[2]);
|
||||
const auto OriginalOffset = GetCursorOffset();
|
||||
|
||||
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
|
||||
const auto InstOffset = GetCursorOffsetFromAddress(Instructions);
|
||||
SetCursorOffset(InstOffset);
|
||||
|
||||
// We encoded the destination register in to the first instruction space.
|
||||
// Read it back.
|
||||
ARMEmitter::Register DestReg(Instructions[0]);
|
||||
|
||||
if (IsADRRange(ImmInstTwo)) {
|
||||
// If within ADR range from the second instruction, then we can emit NOP+ADR
|
||||
if (IsADRRange(ImmInstThree)) {
|
||||
// If within ADR range from the third instruction, then we can emit NOP+NOP+ADR
|
||||
nop();
|
||||
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
|
||||
} else if (IsADRPRange(ImmInstOne)) {
|
||||
nop();
|
||||
adr(DestReg, static_cast<uint32_t>(ImmInstThree) & 0x7FFF);
|
||||
} else if (IsADRPRange(ImmInstTwo)) {
|
||||
|
||||
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
|
||||
// First check if we are in non-offset range for second instruction.
|
||||
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
|
||||
// We can emit nop + adrp
|
||||
// We can emit nop + nop + adrp
|
||||
nop();
|
||||
nop();
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstThree >> 12) & 0x7FFF);
|
||||
} else {
|
||||
// Not aligned, need nop + adrp + add
|
||||
nop();
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
|
||||
} else {
|
||||
// Not aligned, need adrp + add
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
|
||||
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
|
||||
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstTwo & 0xFFF);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
|
||||
FEX_UNREACHABLE;
|
||||
// Stinky path, we need to emit a movz+movk+movk sequence.
|
||||
movz(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 32) & 0x7FFF, 32);
|
||||
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 16) & 0xFFFF, 16);
|
||||
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne) & 0xFFFF);
|
||||
}
|
||||
|
||||
SetCursorOffset(OriginalOffset);
|
||||
@@ -753,27 +786,41 @@ public:
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
// Bind a forward label to a location.
|
||||
// This walks all the instructions in the label's vector.
|
||||
// Then backpatching all instructions that have used the label.
|
||||
void Bind(ForwardLabel* Label) {
|
||||
[[nodiscard]] bool Bind(ForwardLabel* Label) {
|
||||
bool Bound = true;
|
||||
if (Label->FirstInst.Location) {
|
||||
Bind(&Label->FirstInst);
|
||||
Bound &= Bind(&Label->FirstInst);
|
||||
}
|
||||
for (auto& Inst : Label->Insts) {
|
||||
Bind(&Inst);
|
||||
Bound &= Bind(&Inst);
|
||||
}
|
||||
|
||||
return Bound;
|
||||
}
|
||||
|
||||
// Bind a bidirectional location to a location.
|
||||
// Binds both forwards and backwards depending on how the label was used.
|
||||
void Bind(BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] bool Bind(BiDirectionalLabel* Label) {
|
||||
bool Bound = true;
|
||||
if (!Label->Backward.Location) {
|
||||
Bind(&Label->Backward);
|
||||
Bound &= Bind(&Label->Backward);
|
||||
}
|
||||
Bind(&Label->Forward);
|
||||
Bound &= Bind(&Label->Forward);
|
||||
|
||||
return Bound;
|
||||
}
|
||||
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
#include <CodeEmitter/VixlUtils.inl>
|
||||
|
||||
Vendored
+1
-1
Submodule External/Catch2 updated: 8ac8190e49...b3fb4b9fea.
Vendored
+1
-1
Submodule External/drm-headers updated: 0675d2f291...3e49836995.
Vendored
+1
-1
Submodule External/fmt updated: e424e3f2e6...407c905e45.
Vendored
+1
-1
Submodule External/xxhash updated: bbb27a5efb...e626a72bc2.
@@ -74,9 +74,11 @@ add_compile_options($<$<COMPILE_LANGUAGE:CXX>:-fno-strict-aliasing> $<$<COMPILE_
|
||||
|
||||
add_subdirectory(Source/)
|
||||
|
||||
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
endif()
|
||||
|
||||
if (BUILD_TESTING)
|
||||
add_subdirectory(unittests/)
|
||||
|
||||
@@ -200,6 +200,15 @@ def print_man_environment_tail():
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"APP_CACHE_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX stores and loads cache files",
|
||||
"By default FEX will look in $XDG_CACHE_HOME/fex-emu/ or $HOME/.cache/fex-emu/",
|
||||
"This will override the full path, trailing forward-slash is expected to exist",
|
||||
],
|
||||
"''", True)
|
||||
|
||||
def print_man_header():
|
||||
header ='''.Dd {0}
|
||||
.Dt FEX
|
||||
|
||||
@@ -31,7 +31,6 @@ set (SRCS
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
@@ -66,6 +65,7 @@ set (SRCS
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/x87StackOptimizationPass.cpp
|
||||
Utils/LongJump.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
@@ -201,8 +201,10 @@ add_custom_target(CONFIG_INC
|
||||
DEPENDS "${OUTPUT_MAN_NAME}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
|
||||
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
|
||||
endif()
|
||||
|
||||
# Add in diagnostic colours if the option is available.
|
||||
# Ninja code generator will kill colours if this isn't here
|
||||
@@ -289,7 +291,7 @@ AddObject(${PROJECT_NAME}_object OBJECT)
|
||||
AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
if (NOT MINGW_BUILD AND NOT BUILD_STEAM_SUPPORT)
|
||||
install(TARGETS ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
@@ -12,7 +13,7 @@ namespace FEXCore {
|
||||
// Buffered JIT symbol tracking.
|
||||
struct JITSymbolBuffer {
|
||||
// Maximum buffer size to ensure we are a page in size.
|
||||
constexpr static size_t BUFFER_SIZE = 4096 - (8 * 2);
|
||||
constexpr static size_t BUFFER_SIZE = FEXCore::Utils::FEX_PAGE_SIZE - (8 * 2);
|
||||
// Maximum distance until the end of the buffer to do a write.
|
||||
constexpr static size_t NEEDS_WRITE_DISTANCE = BUFFER_SIZE - 64;
|
||||
// Maximum time threshhold to wait before a buffer write occurs.
|
||||
@@ -27,7 +28,7 @@ struct JITSymbolBuffer {
|
||||
size_t Offset {};
|
||||
char Buffer[BUFFER_SIZE] {};
|
||||
};
|
||||
static_assert(sizeof(JITSymbolBuffer) == 4096, "Ensure this is one page in size");
|
||||
static_assert(sizeof(JITSymbolBuffer) == FEXCore::Utils::FEX_PAGE_SIZE, "Ensure this is one page in size");
|
||||
|
||||
class JITSymbols final {
|
||||
public:
|
||||
|
||||
@@ -504,47 +504,6 @@ struct FEX_PACKED X80SoftFloat {
|
||||
return std::bit_cast<float>(Result);
|
||||
}
|
||||
|
||||
bool IsSignalingNaN() const {
|
||||
return (Exponent == 0x7FFF) && (Significand & 0x8000000000000000ULL) && !(Significand & 0x4000000000000000ULL) && // Bit 62 clear (signaling)
|
||||
(Significand & 0x3FFFFFFFFFFFFFFFULL);
|
||||
}
|
||||
|
||||
bool IsQuietNaN() const {
|
||||
return (Exponent == 0x7FFF) && (Significand & 0x8000000000000000ULL) && (Significand & 0x4000000000000000ULL); // Bit 62 set (quiet)
|
||||
}
|
||||
|
||||
// Helper to detect if this is any NaN
|
||||
bool IsNaN() const {
|
||||
return IsSignalingNaN() || IsQuietNaN();
|
||||
}
|
||||
|
||||
// X87 value to F64 while preserving signaling nan property
|
||||
double ToF64_PreserveNan(softfloat_state* state) const {
|
||||
if (IsSignalingNaN()) {
|
||||
// we keep it as a signaling nan in ieee754 in 64bits
|
||||
uint64_t sign_bit = Sign ? 0x8000000000000000ULL : 0;
|
||||
uint64_t exp_bits = 0x7FF0000000000000ULL;
|
||||
uint64_t x87_frac = Significand & 0x3FFFFFFFFFFFFFFFULL;
|
||||
uint64_t ieee_frac = (x87_frac >> 11) & 0x0007FFFFFFFFFFFFULL;
|
||||
|
||||
if (ieee_frac == 0) {
|
||||
ieee_frac = 1;
|
||||
}
|
||||
ieee_frac &= ~0x0008000000000000ULL;
|
||||
|
||||
uint64_t result_bits = sign_bit | exp_bits | ieee_frac;
|
||||
return std::bit_cast<double>(result_bits);
|
||||
} else if (IsQuietNaN()) {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
uint64_t result_bits = std::bit_cast<uint64_t>(Result);
|
||||
result_bits |= 0x0008000000000000ULL;
|
||||
return std::bit_cast<double>(result_bits);
|
||||
} else {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
return std::bit_cast<double>(Result);
|
||||
}
|
||||
}
|
||||
|
||||
double ToF64(softfloat_state* state) const {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
return std::bit_cast<double>(Result);
|
||||
@@ -625,39 +584,6 @@ struct FEX_PACKED X80SoftFloat {
|
||||
*this = f64_to_extF80(state, std::bit_cast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
// Create X80SoftFloat from double while preserving NaN signaling properties
|
||||
static X80SoftFloat FromF64_PreserveNaN(softfloat_state* state, double value) {
|
||||
uint64_t bits = std::bit_cast<uint64_t>(value);
|
||||
|
||||
// Check if it's a nan
|
||||
if ((bits & 0x7FF0000000000000ULL) == 0x7FF0000000000000ULL && (bits & 0x000FFFFFFFFFFFFFULL) != 0) {
|
||||
|
||||
X80SoftFloat result;
|
||||
result.Sign = (bits >> 63) & 1;
|
||||
result.Exponent = 0x7FFF;
|
||||
|
||||
bool is_signaling = !(bits & 0x0008000000000000ULL);
|
||||
uint64_t ieee_payload = bits & 0x0007FFFFFFFFFFFFULL;
|
||||
|
||||
// set bit 63 required for x87
|
||||
result.Significand = 0x8000000000000000ULL;
|
||||
|
||||
if (is_signaling) { // clear bit 62 for signaling nan
|
||||
result.Significand &= ~0x4000000000000000ULL;
|
||||
} else { // clear bit 62 for quiet nan
|
||||
result.Significand |= 0x4000000000000000ULL;
|
||||
}
|
||||
|
||||
// ieee754 51-bit payload -> x87 62-bit payload
|
||||
result.Significand |= (ieee_payload << 11) & 0x3FFFFFFFFFFFFFFFULL;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// For non-NaN values, use standard conversion
|
||||
return X80SoftFloat(state, value);
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(state, std::bit_cast<float128_t>(rhs));
|
||||
|
||||
@@ -4,6 +4,8 @@
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#include <immintrin.h>
|
||||
#else
|
||||
#include <cstdint>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
@@ -30,14 +30,14 @@ class Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Config {
|
||||
namespace DefaultValues {
|
||||
namespace detail {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
|
||||
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
} // namespace DefaultValues
|
||||
} // namespace detail
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR_LOCAL = 0,
|
||||
@@ -134,7 +134,7 @@ public:
|
||||
void Load();
|
||||
|
||||
template<typename T>
|
||||
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<DefaultValues::Type::StringArrayType, T>)
|
||||
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<StringArrayType, T>)
|
||||
std::optional<T> GetConv(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
@@ -142,7 +142,7 @@ public:
|
||||
}
|
||||
|
||||
const auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(!std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
LOGMAN_THROW_A_FMT(!std::holds_alternative<StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
if (std::holds_alternative<T>(Value)) [[likely]] {
|
||||
return std::get<T>(Value);
|
||||
@@ -165,7 +165,7 @@ public:
|
||||
|
||||
private:
|
||||
void MergeConfigMap(const LayerOptions& Options);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value);
|
||||
};
|
||||
|
||||
void MetaLayer::Load() {
|
||||
@@ -181,7 +181,7 @@ void MetaLayer::Load() {
|
||||
}
|
||||
|
||||
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value) {
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value) {
|
||||
// Environment variables need a bit of additional work
|
||||
// We want to merge the arrays rather than overwrite entirely
|
||||
auto MetaEnvironment = OptionMap.find(Option);
|
||||
@@ -193,7 +193,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Defa
|
||||
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](const DefaultValues::Type::StringArrayType& Value) {
|
||||
const auto AddToMap = [&LookupMap](const StringArrayType& Value) {
|
||||
for (const auto& EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == fextl::string::npos) {
|
||||
@@ -209,7 +209,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Defa
|
||||
}
|
||||
};
|
||||
|
||||
AddToMap(std::get<DefaultValues::Type::StringArrayType>(MetaEnvironment->second));
|
||||
AddToMap(std::get<StringArrayType>(MetaEnvironment->second));
|
||||
AddToMap(Value);
|
||||
|
||||
// Now with the two layers merged in the map
|
||||
@@ -225,8 +225,8 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto& it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(it.second), "Tried to get config of invalid type!");
|
||||
MergeEnvironmentVariables(it.first, std::get<DefaultValues::Type::StringArrayType>(it.second));
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<StringArrayType>(it.second), "Tried to get config of invalid type!");
|
||||
MergeEnvironmentVariables(it.first, std::get<StringArrayType>(it.second));
|
||||
} else {
|
||||
OptionMap.insert_or_assign(it.first, it.second);
|
||||
}
|
||||
@@ -423,7 +423,7 @@ bool Exists(ConfigOption Option) {
|
||||
return Meta->OptionExists(Option);
|
||||
}
|
||||
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
std::optional<StringArrayType*> All(ConfigOption Option) {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
@@ -436,6 +436,12 @@ std::optional<T> GetConv(ConfigOption Option) {
|
||||
return Meta->GetConv<T>(Option);
|
||||
}
|
||||
|
||||
template std::optional<bool> GetConv(ConfigOption Option);
|
||||
template std::optional<uint8_t> GetConv(ConfigOption Option);
|
||||
template std::optional<int32_t> GetConv(ConfigOption Option);
|
||||
template std::optional<uint32_t> GetConv(ConfigOption Option);
|
||||
template std::optional<uint64_t> GetConv(ConfigOption Option);
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
@@ -491,13 +497,12 @@ template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t De
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List) {
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<DefaultValues::Type::StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option,
|
||||
DefaultValues::Type::StringArrayType* List);
|
||||
template void Value<StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
|
||||
} // namespace FEXCore::Config
|
||||
@@ -16,6 +16,13 @@
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"EnableCodeCachingWIP": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enable the code caching subsystem"
|
||||
]
|
||||
},
|
||||
"HostFeatures": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::HostFeatures::OFF",
|
||||
@@ -94,6 +101,13 @@
|
||||
"Desc": [
|
||||
"Scales the cycle counter on systems that have low frequencies."
|
||||
]
|
||||
},
|
||||
"CPUFeatureRegisters": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Allows overriding cpu feature flags for manual testing"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
@@ -161,6 +175,44 @@
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
},
|
||||
"DisableL2Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
]
|
||||
},
|
||||
"DynamicL1Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
]
|
||||
},
|
||||
"DynamicL1CacheIncreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "250",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should increase its size.",
|
||||
"Lower numbers means more aggressive scaling upward to the maximum size.",
|
||||
"Higher numbers means more conservative scaling, using less memory.",
|
||||
"Can potentially introduce stutters, more likely the higher the number.",
|
||||
"Don't have this number smaller than the decrease count!"
|
||||
]
|
||||
},
|
||||
"DynamicL1CacheDecreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "50",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should decrease its size.",
|
||||
"The higher the number, the more aggressively it reduces the L1 cache size.",
|
||||
"Lower numbers means more conservative memory savings.",
|
||||
"Can potentially introduce more stutters, more likely the higher the number.",
|
||||
"Don't have this number larger than the increase count!"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
@@ -330,6 +382,13 @@
|
||||
"Enables FEX's low-overhead sampling profile statistics.",
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
]
|
||||
},
|
||||
"EnableGpuvisProfiling": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enables profiling when FEX was built with the gpuvis profiler backend."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
@@ -384,12 +443,19 @@
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
]
|
||||
},
|
||||
"KernelUnalignedAtomicBackpatching": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"When the kernel unaligned atomic handler is enabled, use backpatching to reduce kernel context switches."
|
||||
]
|
||||
},
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Use volatile metadata in PE files to inform TSO instructions when available.",
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
@@ -399,31 +465,6 @@
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
},
|
||||
"X87StrictReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enables stricter X87 floating point behavior when X87ReducedPrecision is enabled.",
|
||||
"Adds additional checks and implementations like NaN propagation for better compatibility."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When enabled enables an optimization around flags.",
|
||||
"Assumes flags are not used across cals.",
|
||||
"Hand-written assembly can violate this assumption."
|
||||
]
|
||||
},
|
||||
"ParanoidTSO": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Makes TSO operations even more strict.",
|
||||
"Forces vector loadstores to also become atomic."
|
||||
]
|
||||
},
|
||||
"StallProcess": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include <Interface/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -29,6 +28,7 @@
|
||||
namespace FEXCore {
|
||||
class SignalDelegator;
|
||||
class ThunkHandler;
|
||||
struct LookupCacheWriteLockToken;
|
||||
|
||||
namespace Core {
|
||||
struct DebugData;
|
||||
@@ -61,7 +61,7 @@ struct CustomIRResult {
|
||||
, Data(Data) {}
|
||||
};
|
||||
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class CodeCache : public AbstractCodeCache {
|
||||
@@ -72,12 +72,34 @@ public:
|
||||
ContextImpl& CTX;
|
||||
bool IsGeneratingCache = false;
|
||||
|
||||
uint64_t ComputeCodeMapId(std::string_view Filename, int FD) override;
|
||||
|
||||
void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
|
||||
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
|
||||
|
||||
void InitiateCacheGeneration() override {
|
||||
IsGeneratingCache = true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies a set of FEX relocations to the given code section.
|
||||
*
|
||||
* FEX relocations describe runtime-dependencies of FEX-generated code.
|
||||
* When loading a code cache, they are used to move cached code to the
|
||||
* dynamically chosen base address of the guest binary.
|
||||
*
|
||||
* Conversely, relocations are applied in reverse when writing code caches
|
||||
* to ensure consistency across generation runs.
|
||||
*
|
||||
* Note that FEX relocations are unrelated to ELF/PE relocations.
|
||||
*
|
||||
* @param GuestDelta Guest address offset to apply to RIP-relative data
|
||||
* @param ForStorage True for serializing data (producing deterministic output); false for de-serializing it (resolving dynamic symbols)
|
||||
*
|
||||
* @return Returns true on success
|
||||
*/
|
||||
[[nodiscard]]
|
||||
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations, bool ForStorage);
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
|
||||
@@ -154,10 +176,20 @@ public:
|
||||
return CodeCache;
|
||||
}
|
||||
|
||||
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
|
||||
void SetCodeMapWriter(fextl::unique_ptr<CodeMapWriter> Writer) override {
|
||||
CodeMapWriter = std::move(Writer);
|
||||
}
|
||||
|
||||
void FlushAndCloseCodeMap() override {
|
||||
if (CodeMapWriter) {
|
||||
CodeMapWriter.reset();
|
||||
}
|
||||
}
|
||||
|
||||
void OnCodeBufferAllocated(const std::shared_ptr<CPU::CodeBuffer>&) override;
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start,
|
||||
uint64_t Length) override;
|
||||
void InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
|
||||
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
@@ -197,7 +229,6 @@ public:
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
@@ -205,9 +236,7 @@ public:
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(x87StrictReducedPrecision, X87STRICTREDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
@@ -227,14 +256,12 @@ public:
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
CodeCache CodeCache;
|
||||
fextl::unique_ptr<CodeMapWriter> CodeMapWriter;
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
ContextImpl(const FEXCore::HostFeatures& Features);
|
||||
|
||||
static bool ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
|
||||
|
||||
// This is used as a replacement for the SMC writes in the mono callsite backpatcher that avoids atomic operations
|
||||
@@ -269,9 +296,9 @@ public:
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual CPUBackendAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator {"FEXMem_OpDispatcher"};
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator {"FEXMem_Frontend"};
|
||||
FEXCore::Utils::PooledAllocatorVirtualWithGuard CPUBackendAllocator {"FEXMem_CPUBackend"};
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const {
|
||||
@@ -312,10 +339,6 @@ protected:
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
VectorAtomicTSOEmulationEnabled = false;
|
||||
MemcpyAtomicTSOEmulationEnabled = false;
|
||||
} else if (Config.ParanoidTSO) {
|
||||
AtomicTSOEmulationEnabled = true;
|
||||
VectorAtomicTSOEmulationEnabled = true;
|
||||
MemcpyAtomicTSOEmulationEnabled = true;
|
||||
} else {
|
||||
AtomicTSOEmulationEnabled = Config.TSOEnabled;
|
||||
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
@@ -323,13 +346,6 @@ protected:
|
||||
}
|
||||
}
|
||||
|
||||
void UpdateX87PrecisionConfig() {
|
||||
// If strict reduced precision is enabled, automatically enable reduced precision
|
||||
if (Config.x87StrictReducedPrecision() && !Config.x87ReducedPrecision()) {
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_X87REDUCEDPRECISION, "1");
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
@@ -361,5 +377,8 @@ private:
|
||||
|
||||
bool MonoDetected = false;
|
||||
std::atomic<uint64_t> MonoBackpatcherBlock;
|
||||
|
||||
std::mutex CodeBufferListLock;
|
||||
fextl::vector<std::weak_ptr<CPU::CodeBuffer>> CodeBufferList;
|
||||
};
|
||||
} // namespace FEXCore::Context
|
||||
@@ -105,9 +105,12 @@ constexpr ARMEmitter::PRegister PRED_TMP_32B = ARMEmitter::PReg::p7;
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public ARMEmitter::Emitter {
|
||||
protected:
|
||||
public:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
@@ -117,8 +120,6 @@ protected:
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
|
||||
|
||||
// Correlate an ARM register back to an x86 register index.
|
||||
|
||||
@@ -360,9 +360,7 @@ namespace CPU {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, "FEXMemJIT");
|
||||
#endif
|
||||
FEXCore::Allocator::VirtualName("FEXMemJIT", reinterpret_cast<void*>(Ptr), Size);
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
@@ -402,7 +400,7 @@ namespace CPU {
|
||||
Latest = Buffer;
|
||||
LatestOffset = 0;
|
||||
|
||||
OnCodeBufferAllocated(*Buffer);
|
||||
OnCodeBufferAllocated(Buffer);
|
||||
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
@@ -81,7 +81,7 @@ namespace CPU {
|
||||
// Protects writes to the latest CodeBuffer and changes to LatestOffset
|
||||
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
|
||||
|
||||
virtual void OnCodeBufferAllocated(CodeBuffer&) {};
|
||||
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
|
||||
|
||||
private:
|
||||
fextl::shared_ptr<CodeBuffer> Latest;
|
||||
@@ -161,7 +161,7 @@ namespace CPU {
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() = 0;
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@ $end_info$
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
@@ -88,6 +89,7 @@ namespace ProductNames {
|
||||
static const char ARM_Blizzard_M2Pro[] = "Apple Blizzard (M2 Pro)";
|
||||
static const char ARM_Avalanche_M2Max[] = "Apple Avalanche (M2 Max)";
|
||||
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
|
||||
static const char ARM_AppleSilicon[] = "Apple Silicon";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
static const char ARM_Ampere_1[] = "AmpereOne";
|
||||
@@ -188,6 +190,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x61, 0x029, 1, ProductNames::ARM_Firestorm_M1Max}, // Apple Firestorm (M1 Max)
|
||||
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
|
||||
{0x61, 0, 1, ProductNames::ARM_AppleSilicon}, // QEmu Apple Silicon
|
||||
|
||||
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
|
||||
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
|
||||
@@ -441,10 +444,10 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(0 << 24); // Local APIC ID
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(GetCPUID() << 24); // Local APIC ID
|
||||
|
||||
Res.ecx = (1 << 0) | // SSE3
|
||||
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
|
||||
@@ -507,7 +510,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
(1 << 25) | // SSE
|
||||
(1 << 26) | // SSE2
|
||||
(0 << 27) | // Self Snoop
|
||||
(1 << 28) | // Max APIC IDs reserved field is valid
|
||||
(0 << 28) | // (HTT) Max APIC IDs reserved field is valid
|
||||
(1 << 29) | // Thermal monitor
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Pending break enable
|
||||
@@ -1094,9 +1097,9 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) con
|
||||
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
|
||||
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
(CoreCount << 0); // Count count subtract one
|
||||
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
(std::bit_ceil(Cores) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
(CoreCount << 0); // Count count subtract one
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
@@ -277,7 +277,7 @@ private:
|
||||
// 0: Highest function parameter and ID
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 1: Processor info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 2: Cache and TLB info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 3: Serial Number(previously), now reserved
|
||||
|
||||
@@ -1,12 +1,213 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <Interface/Context/Context.h>
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
#include <Interface/Context/Context.h>
|
||||
#include <Interface/Core/ArchHelpers/Arm64Emitter.h>
|
||||
#include <Interface/Core/JIT/Relocations.h>
|
||||
#include <Interface/Core/LookupCache.h>
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <git_version.h>
|
||||
|
||||
#include <xxhash.h>
|
||||
|
||||
#include <fstream>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
#if __clang_major__ < 16
|
||||
ExecutableFileInfo::ExecutableFileInfo(fextl::unique_ptr<HLE::SourcecodeMap> Map, uint64_t FileId, fextl::string Filename)
|
||||
: SourcecodeMap(std::move(Map))
|
||||
, FileId(FileId)
|
||||
, Filename(Filename) {}
|
||||
#endif
|
||||
ExecutableFileInfo::~ExecutableFileInfo() = default;
|
||||
|
||||
fextl::string CodeMap::GetBaseFilename(const ExecutableFileInfo& MainExecutable, bool AddNombSuffix) {
|
||||
auto FileId = MainExecutable.FileId;
|
||||
|
||||
std::string_view base_filename = FHU::Filesystem::GetFilename(std::string_view {MainExecutable.Filename});
|
||||
if (FileId != 0xffff'ffff'ffff'ffff) {
|
||||
return fextl::fmt::format("{}-{:016x}{}", base_filename, MainExecutable.FileId, AddNombSuffix ? "-nomb" : "");
|
||||
}
|
||||
|
||||
return "";
|
||||
}
|
||||
|
||||
fextl::map<CodeMapFileId, CodeMap::ParsedContents> CodeMap::ParseCodeMap(std::ifstream& File) {
|
||||
fextl::map<CodeMapFileId, CodeMap::ParsedContents> Ret;
|
||||
while (true) {
|
||||
Entry Entry;
|
||||
File.read(reinterpret_cast<char*>(&Entry), sizeof(Entry));
|
||||
if (!File) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (Entry.FileId == LoadExternalLibrary.FileId && Entry.BlockOffset == LoadExternalLibrary.BlockOffset) {
|
||||
ExternalLibraryInfo Info;
|
||||
File.read(reinterpret_cast<char*>(&Info), sizeof(Info));
|
||||
|
||||
fextl::string Filename;
|
||||
std::getline(File, Filename, '\0');
|
||||
|
||||
// Align to 4-byte boundary
|
||||
char Null[4];
|
||||
File.read(Null, AlignUp(Filename.size() + 1, 4) - Filename.size() - 1);
|
||||
if (!File) {
|
||||
break;
|
||||
}
|
||||
Ret[Info.ExternalFileId].Filename = std::move(Filename);
|
||||
} else if (Entry.FileId == SetExecutableFileId {}.Marker.FileId && Entry.BlockOffset == SetExecutableFileId {}.Marker.BlockOffset) {
|
||||
CodeMapFileId ExecutableFileId;
|
||||
File.read(reinterpret_cast<char*>(&ExecutableFileId), sizeof(ExecutableFileId));
|
||||
if (!File) {
|
||||
break;
|
||||
}
|
||||
Ret[ExecutableFileId].IsExecutable = true;
|
||||
} else {
|
||||
if (!Ret.contains(Entry.FileId)) {
|
||||
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
|
||||
} else {
|
||||
Ret[Entry.FileId].Blocks.insert(Entry.BlockOffset);
|
||||
}
|
||||
}
|
||||
|
||||
if (!File) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
return Ret;
|
||||
}
|
||||
|
||||
CodeMapWriter::CodeMapWriter(CodeMapOpener& Opener, bool OpenEagerly)
|
||||
: Buffer(4096)
|
||||
, FileOpener(Opener) {
|
||||
if (OpenEagerly) {
|
||||
CodeMapFD = FileOpener.OpenCodeMapFile();
|
||||
}
|
||||
}
|
||||
|
||||
CodeMapWriter::~CodeMapWriter() {
|
||||
if (CodeMapFD.value_or(-1) != -1) {
|
||||
Flush(BufferOffset);
|
||||
close(*CodeMapFD);
|
||||
}
|
||||
}
|
||||
|
||||
bool CodeMapWriter::IsWriteEnabled(const ExecutableFileSectionInfo& Section) {
|
||||
if (CodeMapFD == -1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// PV libraries can't yet be read by FEXServer, so skip dumping them
|
||||
if (Section.FileInfo.Filename.starts_with("/run/pressure-vessel")) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (CodeMapFD) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Acquire mutex and re-check CodeMapFD to avoid race conditions
|
||||
auto lk = std::unique_lock {Mutex};
|
||||
if (!CodeMapFD) {
|
||||
CodeMapFD = FileOpener.OpenCodeMapFile();
|
||||
}
|
||||
|
||||
return CodeMapFD != -1;
|
||||
}
|
||||
|
||||
void CodeMapWriter::Flush(size_t Offset) {
|
||||
// Acquire exclusive lock and flush circular buffer
|
||||
std::unique_lock Lock {Mutex};
|
||||
Flush(Offset, Lock);
|
||||
}
|
||||
|
||||
void CodeMapWriter::Flush(size_t Offset, std::unique_lock<std::shared_mutex>&) {
|
||||
write(*CodeMapFD, Buffer.data(), Offset);
|
||||
BufferOffset = 0;
|
||||
}
|
||||
|
||||
void CodeMapWriter::AppendBlock(const FEXCore::ExecutableFileSectionInfo& SectionInfo, uint64_t BlockEntry) {
|
||||
if (!IsWriteEnabled(SectionInfo)) {
|
||||
return;
|
||||
}
|
||||
|
||||
BlockEntry -= SectionInfo.FileStartVA;
|
||||
if (BlockEntry > std::numeric_limits<uint32_t>::max()) {
|
||||
ERROR_AND_DIE_FMT("Cannot write code map");
|
||||
}
|
||||
|
||||
// Register new library if not already known
|
||||
bool NewLibraryLoad = false;
|
||||
{
|
||||
// Check prior registration with shared lock
|
||||
std::shared_lock Lock {Mutex};
|
||||
NewLibraryLoad = !KnownFileIds.contains(SectionInfo.FileInfo.FileId);
|
||||
}
|
||||
if (NewLibraryLoad) {
|
||||
// Register to map with exclusive lock
|
||||
std::unique_lock Lock {Mutex};
|
||||
NewLibraryLoad &= KnownFileIds.insert(SectionInfo.FileInfo.FileId).second;
|
||||
}
|
||||
if (NewLibraryLoad) {
|
||||
// Add entry to code map
|
||||
AppendLibraryLoad(SectionInfo.FileInfo);
|
||||
}
|
||||
|
||||
// Register the actual code block
|
||||
CodeMap::Entry DataEntry {SectionInfo.FileInfo.FileId, static_cast<uint32_t>(BlockEntry)};
|
||||
AppendData(std::as_bytes(std::span {&DataEntry, 1}));
|
||||
}
|
||||
|
||||
void CodeMapWriter::AppendLibraryLoad(const FEXCore::ExecutableFileInfo& FileInfo) {
|
||||
// See CodeMap::ExternalLibraryInfo
|
||||
auto ExternalFileId = FileInfo.FileId;
|
||||
auto TotalSize = AlignUp(sizeof(CodeMap::LoadExternalLibrary) + sizeof(ExternalFileId) + FileInfo.Filename.size() + 1, 4);
|
||||
const auto Data = reinterpret_cast<char*>(alloca(TotalSize));
|
||||
auto WritePtr = std::copy_n(reinterpret_cast<const char*>(&CodeMap::LoadExternalLibrary), sizeof(CodeMap::LoadExternalLibrary), Data);
|
||||
WritePtr = std::copy_n(reinterpret_cast<const char*>(&ExternalFileId), sizeof(ExternalFileId), WritePtr);
|
||||
WritePtr = std::copy(FileInfo.Filename.begin(), FileInfo.Filename.end(), WritePtr);
|
||||
std::fill(WritePtr, Data + TotalSize, 0);
|
||||
AppendData(std::as_bytes(std::span {Data, TotalSize}));
|
||||
}
|
||||
|
||||
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo) {
|
||||
CodeMap::SetExecutableFileId Data {.ExecutableFileId = FileInfo.FileId};
|
||||
AppendData(std::span {reinterpret_cast<const std::byte*>(&Data), sizeof(Data)});
|
||||
}
|
||||
|
||||
void CodeMapWriter::AppendData(std::span<const std::byte> Data) {
|
||||
std::shared_lock Lock {Mutex};
|
||||
auto Offset = BufferOffset.fetch_add(Data.size_bytes());
|
||||
if (Offset + Data.size_bytes() > Buffer.size()) {
|
||||
// Acquire exclusive lock and flush the buffer.
|
||||
// Under heavy pressure, multiple threads may observe an exhausted buffer simultaneously.
|
||||
// The thread with the last in-bounds Offset is responsible for flushing the buffer.
|
||||
Lock.unlock();
|
||||
bool IsResponsibleForFlush = false;
|
||||
{
|
||||
std::unique_lock ExclusiveLock {Mutex};
|
||||
IsResponsibleForFlush = (Offset <= Buffer.size());
|
||||
if (IsResponsibleForFlush) {
|
||||
Flush(Offset, ExclusiveLock);
|
||||
}
|
||||
}
|
||||
if (!IsResponsibleForFlush) {
|
||||
// Wait for the buffer to be flushed on the responsible thread
|
||||
Utils::SpinWaitLock::WaitPred<std::less_equal<>, size_t>(reinterpret_cast<size_t*>(&BufferOffset), Buffer.size());
|
||||
}
|
||||
AppendData(Data);
|
||||
return;
|
||||
}
|
||||
|
||||
memcpy(&Buffer.at(Offset), Data.data(), Data.size_bytes());
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Context {
|
||||
@@ -15,12 +216,156 @@ CodeCache::CodeCache(ContextImpl& CTX_)
|
||||
: CTX(CTX_) {}
|
||||
CodeCache::~CodeCache() = default;
|
||||
|
||||
uint64_t CodeCache::ComputeCodeMapId(std::string_view Filename, int FD) {
|
||||
if (Filename.empty()) {
|
||||
return 0xffff'ffff'ffff'ffff;
|
||||
}
|
||||
|
||||
// For now, we just use the file path as an identifier.
|
||||
// TODO: Ensure the hash is unique enough to distinguish executables while remaining independent of the installation location
|
||||
return XXH3_64bits(Filename.data(), Filename.size());
|
||||
}
|
||||
|
||||
struct CodeCacheHeader {
|
||||
char Magic[4] = {'F', 'X', 'C', 'C'};
|
||||
uint32_t FormatVersion = 1;
|
||||
char FEXVersion[8] = {};
|
||||
uint32_t NumBlocks;
|
||||
uint32_t NumCodePages;
|
||||
uint32_t CodeBufferSize;
|
||||
uint32_t NumRelocations;
|
||||
uint64_t SerializedBaseAddress;
|
||||
// TODO: Consider including information from LookupCache.BlockLinks
|
||||
};
|
||||
|
||||
void CodeCache::LoadData(Core::InternalThreadState& Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& GuestRIPLookup) {
|
||||
// TODO
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static constexpr auto IsOrderedContainer(const T&) -> std::false_type;
|
||||
template<typename... T>
|
||||
static constexpr auto IsOrderedContainer(const std::map<T...>&) -> std::true_type;
|
||||
template<typename... T>
|
||||
static constexpr auto IsOrderedContainer(const std::set<T...>&) -> std::true_type;
|
||||
|
||||
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
|
||||
// TODO
|
||||
auto CodeBuffer = CTX.GetLatest();
|
||||
auto& LookupCache = *Thread.LookupCache->Shared;
|
||||
|
||||
auto Relocations = Thread.CPUBackend->TakeRelocations(SourceBinary.FileStartVA);
|
||||
|
||||
// Write file header
|
||||
CodeCacheHeader header;
|
||||
memcpy(&header.FEXVersion[0], GIT_SHORT_HASH, strlen(GIT_SHORT_HASH));
|
||||
header.NumBlocks = LookupCache.BlockList.size();
|
||||
header.NumCodePages = LookupCache.CodePages.size();
|
||||
header.CodeBufferSize = CTX.LatestOffset;
|
||||
header.NumRelocations = Relocations.size();
|
||||
header.SerializedBaseAddress = SerializedBaseAddress;
|
||||
::write(fd, &header, sizeof(header));
|
||||
|
||||
// Dump guest<->host block mappings
|
||||
{
|
||||
// Cache contents must be deterministic, so copy the unordered block list and then sort by key
|
||||
static_assert(!decltype(IsOrderedContainer(LookupCache.BlockList))::value, "Already deterministic; drop temporary container");
|
||||
fextl::vector<std::pair<uint64_t, const GuestToHostMap::BlockEntry*>> BlockList;
|
||||
BlockList.reserve(LookupCache.BlockList.size());
|
||||
for (auto& [Guest, BlockEntry] : LookupCache.BlockList) {
|
||||
static_assert(sizeof(Guest) == 8, "Breaking change in code cache data layout");
|
||||
BlockList.emplace_back(Guest, &BlockEntry);
|
||||
}
|
||||
std::ranges::sort(BlockList);
|
||||
|
||||
for (auto [Guest, Host] : BlockList) {
|
||||
static_assert(sizeof(Host->HostCode) == 8, "Breaking change in code cache data layout");
|
||||
static_assert(sizeof(Host->CodePages[0]) == 8, "Breaking change in code cache data layout");
|
||||
|
||||
Guest -= SourceBinary.FileStartVA;
|
||||
::write(fd, &Guest, sizeof(Guest));
|
||||
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->Ptr);
|
||||
::write(fd, &HostCode, sizeof(HostCode));
|
||||
uint64_t NumCodePages = Host->CodePages.size();
|
||||
::write(fd, &NumCodePages, sizeof(NumCodePages));
|
||||
LOGMAN_THROW_A_FMT(std::ranges::is_sorted(Host->CodePages), "Code pages aren't sorted");
|
||||
for (auto CodePage : Host->CodePages) {
|
||||
CodePage -= SourceBinary.FileStartVA;
|
||||
::write(fd, &CodePage, sizeof(CodePage));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Dump relocations
|
||||
static_assert(sizeof(Relocations[0]) == 48, "Breaking change in code cache data layout");
|
||||
::write(fd, Relocations.data(), Relocations.size() * sizeof(Relocations[0]));
|
||||
|
||||
// Pad to next page in file so that the CodeBuffer can be mmap'ed into process on load
|
||||
char Zero[64] {};
|
||||
auto Off = lseek(fd, 0, SEEK_CUR);
|
||||
while (Off != AlignUp(Off, Utils::FEX_PAGE_SIZE)) {
|
||||
auto BytesToWrite = std::min(AlignUp(Off, Utils::FEX_PAGE_SIZE) - Off, sizeof(Zero));
|
||||
::write(fd, Zero, BytesToWrite);
|
||||
Off += BytesToWrite;
|
||||
}
|
||||
|
||||
// Dump the host code (relocated for position-independent serialization)
|
||||
std::vector CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->Ptr), reinterpret_cast<std::byte*>(CodeBuffer->Ptr) + CTX.LatestOffset);
|
||||
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, true)) {
|
||||
LOGMAN_THROW_A_FMT(false, "Failed to apply code relocations");
|
||||
return false;
|
||||
}
|
||||
::write(fd, CodeBufferData.data(), CodeBufferData.size());
|
||||
|
||||
// Dump code pages
|
||||
static_assert(decltype(IsOrderedContainer(LookupCache.CodePages))::value, "Non-deterministic data source");
|
||||
for (auto& [Page, Entrypoints] : LookupCache.CodePages) {
|
||||
static_assert(sizeof(Page) == 8, "Breaking change in code cache data layout");
|
||||
::write(fd, &Page, sizeof(Page));
|
||||
uint64_t NumEntrypoints = Entrypoints.size();
|
||||
::write(fd, &NumEntrypoints, sizeof(NumEntrypoints));
|
||||
::write(fd, Entrypoints.data(), Entrypoints.size() * sizeof(Entrypoints[0]));
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
|
||||
std::span<const FEXCore::CPU::Relocation> EntryRelocations, bool ForStorage) {
|
||||
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
|
||||
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
|
||||
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
|
||||
Emitter.SetCursorOffset(Reloc.Header.Offset);
|
||||
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
// Generate a literal so we can place it
|
||||
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Reloc.NamedSymbolLiteral.Symbol);
|
||||
Emitter.dc64(Pointer);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
Emitter.dc64(GuestEntry + Reloc.GuestRIP.GuestRIP);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
uint64_t Pointer = Reloc.GuestRIP.GuestRIP + GuestEntry;
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIP.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
|
||||
default: ERROR_AND_DIE_FMT("Unknown relocation type {}", ToUnderlying(Reloc.Header.Type));
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -57,13 +57,20 @@ $end_info$
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <condition_variable>
|
||||
#include <fcntl.h>
|
||||
#include <functional>
|
||||
#include <mutex>
|
||||
#include <queue>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <string_view>
|
||||
#include <sys/stat.h>
|
||||
#include <type_traits>
|
||||
#include <unistd.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
|
||||
@@ -93,8 +100,6 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
// Ensure X87 precision constraints are respected.
|
||||
UpdateX87PrecisionConfig();
|
||||
}
|
||||
|
||||
struct GetFrameBlockInfoResult {
|
||||
@@ -371,7 +376,9 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
|
||||
|
||||
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->State.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->State.L1Mask = Thread->LookupCache->GetScaledL1PointerMask();
|
||||
|
||||
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
|
||||
|
||||
Dispatcher->InitThreadPointers(Thread);
|
||||
@@ -393,6 +400,7 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXC
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
|
||||
.CTX = this,
|
||||
};
|
||||
FEXCore::Allocator::VirtualName("FEXMem_ThreadState", Thread, sizeof(*Thread));
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
@@ -429,6 +437,10 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
|
||||
Profiler::PostForkAction(Child);
|
||||
if (Child) {
|
||||
if (CodeMapWriter) {
|
||||
CodeMapWriter->ResetAfterFork();
|
||||
}
|
||||
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
StrictSplitLockMutex = 0;
|
||||
@@ -451,9 +463,14 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
|
||||
void ContextImpl::OnCodeBufferAllocated(const fextl::shared_ptr<CPU::CodeBuffer>& Buffer) {
|
||||
if (Config.GlobalJITNaming()) {
|
||||
Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
Symbols.RegisterJITSpace(Buffer->Ptr, Buffer->Size);
|
||||
}
|
||||
|
||||
{
|
||||
std::scoped_lock lk {CodeBufferListLock};
|
||||
CodeBufferList.emplace_back(Buffer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -465,7 +482,8 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, boo
|
||||
Thread->CPUBackend->ClearCache();
|
||||
} else {
|
||||
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
|
||||
Thread->LookupCache->ClearCache();
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
Thread->LookupCache->ClearCache(lk);
|
||||
}
|
||||
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
}
|
||||
@@ -637,10 +655,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
|
||||
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::NOEXEC_INST) {
|
||||
Thread->OpDispatcher->NoExecOp(DecodedInfo);
|
||||
} else {
|
||||
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::INVALID_INST) {
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
} else {
|
||||
Thread->OpDispatcher->NoExecOp(DecodedInfo);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -706,7 +724,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
MappedSection->FileInfo.SourcecodeMap = SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, MappedSection->FileInfo.FileId);
|
||||
MappedSection->FileInfo.SourcecodeMap =
|
||||
SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, CodeMap::GetBaseFilename(MappedSection->FileInfo, false));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -723,7 +742,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
// but this would increase lock contention. Redundant frontend runs aren't
|
||||
// as expensive and are easily reverted.
|
||||
if (MaxInst != 1) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
|
||||
.DebugData = nullptr,
|
||||
@@ -764,10 +783,13 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
// Accumulate a JIT count now, as even if another thread raced us, it should count as a compile.
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedJITCount, 1);
|
||||
|
||||
auto [CompiledCode, DebugData, StartAddr, Length, NeedsAddGuestCodeRanges] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
|
||||
if (CodePtr == nullptr) {
|
||||
@@ -821,20 +843,32 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
fextl::vector<uint64_t> CodePages;
|
||||
|
||||
if (NeedsAddGuestCodeRanges) {
|
||||
// Track in the guest to host map all entrypoints for all pages the compiled block touches, if any page didn't previously
|
||||
// contain code, inform the frontend so it can setup SMC detection.
|
||||
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
CodePages.reserve(BlockInfo->CodePages.size());
|
||||
CodePages.insert(CodePages.end(), BlockInfo->CodePages.begin(), BlockInfo->CodePages.end());
|
||||
for (auto CodePage : BlockInfo->CodePages) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(Thread, BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
|
||||
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
|
||||
Thread->LookupCache->AddBlockMapping(GuestAddr, HostAddr);
|
||||
Thread->LookupCache->AddBlockMapping(Thread, GuestAddr, CodePages, HostAddr);
|
||||
}
|
||||
|
||||
if (CodeMapWriter) {
|
||||
auto Region = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (Region && Region->FileStartVA != 0) {
|
||||
CodeMapWriter->AppendBlock(*Region, GuestRIP);
|
||||
}
|
||||
}
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
@@ -861,49 +895,37 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
|
||||
uint64_t Start, uint64_t Length) {
|
||||
void ContextImpl::InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) {
|
||||
FEXCORE_PROFILE_SCOPED("InvalidateCodeBuffersCodeRange");
|
||||
|
||||
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
std::scoped_lock lk {CodeBufferListLock};
|
||||
auto it = CodeBufferList.begin();
|
||||
while (it != CodeBufferList.end()) {
|
||||
if (auto Strong = it->lock()) {
|
||||
Strong->LookupCache->InvalidateRange(Start, Length);
|
||||
it++;
|
||||
} else {
|
||||
it = CodeBufferList.erase(it);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
|
||||
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
// Ensures now-modified mappings aren't cached as being in their previous non-executable state.
|
||||
// Accessing FrontendDecoder is safe as the thread's code invalidation mutex must be locked here.
|
||||
Thread->FrontendDecoder->ResetExecutableRangeCache();
|
||||
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
auto& CodePages = Thread->LookupCache->Shared->CodePages;
|
||||
if (Thread->LookupCache->InvalidateCacheRange(Start, Length)) {
|
||||
FEXCORE_PROFILE_SCOPED("InvalidateCallRet");
|
||||
|
||||
auto lower = CodePages.lower_bound(Start >> 12);
|
||||
auto upper = CodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
Accumulator.emplace_back(std::move(it->second));
|
||||
}
|
||||
|
||||
bool InvalidatedAnyEntries = false;
|
||||
for (const auto& PageEntries : Accumulator) {
|
||||
for (const auto& Entry : PageEntries) {
|
||||
if (ContextImpl::ThreadRemoveCodeEntry(Thread, Entry)) {
|
||||
InvalidatedAnyEntries = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (InvalidatedAnyEntries) {
|
||||
// This may cause access violations in the thread on Windows as zeroing is not atomic, this is handled by the frontend
|
||||
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
|
||||
uint64_t Start, uint64_t Length) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
|
||||
}
|
||||
|
||||
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
|
||||
"be unique_locked here");
|
||||
|
||||
return Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
static_cast<ContextImpl*>(Frame->Thread->CTX)->SyscallHandler->InvalidateGuestCodeRange(Frame->Thread, GuestRIP, 1);
|
||||
}
|
||||
@@ -971,7 +993,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
void ContextImpl::AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
ForceTSOValidRanges.Insert(ValidRanges);
|
||||
ForceTSOInstructions.merge(Instructions);
|
||||
ForceTSOInstructions.merge(std::move(Instructions));
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
|
||||
|
||||
@@ -1,11 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/VectorRegType.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -26,9 +25,7 @@
|
||||
#endif
|
||||
|
||||
#include <array>
|
||||
#include <atomic>
|
||||
#include <bit>
|
||||
#include <condition_variable>
|
||||
#include <csignal>
|
||||
#include <cstring>
|
||||
|
||||
@@ -38,12 +35,14 @@ static void SleepThread(FEXCore::Context::ContextImpl* CTX, FEXCore::Core::CpuSt
|
||||
CTX->SyscallHandler->SleepThread(CTX, Frame);
|
||||
}
|
||||
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 4;
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = FEXCore::Utils::FEX_PAGE_SIZE * 4;
|
||||
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl* ctx)
|
||||
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
|
||||
, CTX {ctx} {
|
||||
EmitDispatcher();
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(GetBufferBase()), MAX_DISPATCHER_CODE_SIZE);
|
||||
}
|
||||
|
||||
Dispatcher::~Dispatcher() {
|
||||
@@ -93,12 +92,12 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
FillStaticRegs();
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
|
||||
ARMEmitter::BiDirectionalLabel LoopTop {};
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
b(&LoopTop);
|
||||
(void)b(&LoopTop);
|
||||
|
||||
AbsoluteLoopTopAddressEnterECFillSRA = GetCursorAddress<uint64_t>();
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
@@ -106,10 +105,10 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
// Force a single instruction block if ENTRY_FILL_SRA_SINGLE_INST_REG is nonzero entering the JIT, used for inline SMC handling.
|
||||
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
|
||||
// Enter JIT
|
||||
b(&LoopTop);
|
||||
(void)b(&LoopTop);
|
||||
|
||||
AbsoluteLoopTopAddressEnterEC = GetCursorAddress<uint64_t>();
|
||||
// Load ThreadState and write the target PC there
|
||||
@@ -130,7 +129,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, REG_CALLRET_SP);
|
||||
// EC_CALL_CHECKER_PC_REG is REG_PF which isn't touched by any of the above
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, EC_CALL_CHECKER_PC_REG, TMP1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
|
||||
|
||||
// If the entry at the TOS is for the target address, pop it and return to the JIT code
|
||||
add(ARMEmitter::Size::i64Bit, REG_CALLRET_SP, REG_CALLRET_SP, 0x10);
|
||||
@@ -142,7 +141,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
// We want to ensure that we are 16 byte aligned at the top of this loop
|
||||
Align16B();
|
||||
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
@@ -169,66 +168,73 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
|
||||
Bind(&l_NotECCode);
|
||||
(void)Bind(&l_NotECCode);
|
||||
#endif
|
||||
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
|
||||
}
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
ARMEmitter::ForwardLabel NoBlock;
|
||||
|
||||
{
|
||||
// Offset the address and add to our page pointer
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
|
||||
if (DisableL2Cache()) {
|
||||
(void)b(&NoBlock);
|
||||
} else {
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Load the pointer from the offset
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
|
||||
}
|
||||
|
||||
// If page pointer is zero then we have no block
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
|
||||
// The the full LookupCacheEntry with a single LDP.
|
||||
// Check the guest address first to ensure it maps to the address we are currently at.
|
||||
// This fixes aliasing problems
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
sub(TMP2, TMP2, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
// Offset the address and add to our page pointer
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
|
||||
// Load the pointer from the offset
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
|
||||
// Jump to the block
|
||||
br(TMP4);
|
||||
// If page pointer is zero then we have no block
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
|
||||
// The the full LookupCacheEntry with a single LDP.
|
||||
// Check the guest address first to ensure it maps to the address we are currently at.
|
||||
// This fixes aliasing problems
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
sub(TMP2, TMP2, RipReg);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
|
||||
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// L1Mask is pre-shifted.
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg.R(), ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
add(TMP1, TMP1, TMP2);
|
||||
|
||||
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
|
||||
|
||||
// Jump to the block
|
||||
br(TMP4);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -304,7 +310,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
(void)Bind(&NoBlock);
|
||||
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
@@ -338,7 +344,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
|
||||
{
|
||||
Bind(&CompileSingleStep);
|
||||
(void)Bind(&CompileSingleStep);
|
||||
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
@@ -482,7 +488,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->X86CodeGen.CallbackReturn);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->SignalDelegation->GetThunkCallbackRET());
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, CTX->Config.Is64BitMode ? 16 : 12);
|
||||
@@ -500,7 +506,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
b(&LoopTop);
|
||||
(void)b(&LoopTop);
|
||||
}
|
||||
|
||||
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
|
||||
@@ -569,14 +575,15 @@ void Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&l_CTX);
|
||||
(void)Bind(&l_CTX);
|
||||
dc64(reinterpret_cast<uintptr_t>(CTX));
|
||||
Bind(&l_Sleep);
|
||||
(void)Bind(&l_Sleep);
|
||||
dc64(reinterpret_cast<uint64_t>(SleepThread));
|
||||
Bind(&l_CompileBlock);
|
||||
(void)Bind(&l_CompileBlock);
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMFCompileBlock.GetConvertedPointer());
|
||||
Bind(&l_CompileSingleStep);
|
||||
(void)Bind(&l_CompileSingleStep);
|
||||
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
|
||||
dc64(PMFCompileSingleStep.GetConvertedPointer());
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#include <array>
|
||||
@@ -50,6 +51,10 @@ public:
|
||||
}
|
||||
#endif
|
||||
|
||||
uint64_t GetExitFunctionLinkerAddress() const {
|
||||
return ExitFunctionLinkerAddress;
|
||||
}
|
||||
|
||||
SignalDelegatorConfig MakeSignalDelegatorConfig() const;
|
||||
|
||||
protected:
|
||||
@@ -92,6 +97,8 @@ private:
|
||||
|
||||
void EmitDispatcher();
|
||||
uint64_t GenerateABICall(FallbackABI ABI);
|
||||
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -9,7 +9,6 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include <array>
|
||||
@@ -90,11 +89,6 @@ Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
}
|
||||
|
||||
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
// Treat FEX-internal X86 callbacks as always executable
|
||||
if (EntryPoint == CTX->X86CodeGen.CallbackReturn) {
|
||||
return true;
|
||||
}
|
||||
|
||||
while (Address < ExecutableRangeBase || Address + Size > ExecutableRangeEnd) {
|
||||
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, Address);
|
||||
ExecutableRangeBase = RangeInfo.Base;
|
||||
@@ -1047,8 +1041,11 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
auto Result = ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST :
|
||||
DecodeInst->InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST :
|
||||
DecodedBlockStatus::NOEXEC_INST;
|
||||
DecodeInst->InstSize = 0;
|
||||
return ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST : DecodedBlockStatus::NOEXEC_INST;
|
||||
return Result;
|
||||
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
@@ -1450,7 +1447,10 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
EraseBlock = true;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" : "NoExec", OpAddress);
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::NOEXEC_INST ? "NoExec" :
|
||||
"PartialDecode",
|
||||
OpAddress);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -27,6 +27,7 @@ public:
|
||||
SUCCESS,
|
||||
INVALID_INST,
|
||||
NOEXEC_INST,
|
||||
PARTIAL_DECODE_INST,
|
||||
};
|
||||
|
||||
// New Frontend decoding
|
||||
|
||||
@@ -2,13 +2,11 @@
|
||||
#pragma once
|
||||
#include "Common/SoftFloat.h"
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/SHMStats.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
|
||||
@@ -79,12 +77,6 @@ struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle8(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
ScopedSoftFloatState State {FCW, Frame};
|
||||
auto Context = static_cast<Context::ContextImpl*>(Frame->Thread->CTX);
|
||||
auto ReducedPrecisionMode = Context->Config.x87ReducedPrecision;
|
||||
auto StrictReducedPrecisionMode = Context->Config.x87StrictReducedPrecision;
|
||||
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
|
||||
return X80SoftFloat::FromF64_PreserveNaN(&State.State, src);
|
||||
}
|
||||
return X80SoftFloat(&State.State, src);
|
||||
}
|
||||
};
|
||||
@@ -123,12 +115,6 @@ struct OpHandlers<IR::OP_F80CVT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
ScopedSoftFloatState State {FCW, Frame};
|
||||
auto Context = static_cast<Context::ContextImpl*>(Frame->Thread->CTX);
|
||||
auto ReducedPrecisionMode = Context->Config.x87ReducedPrecision;
|
||||
auto StrictReducedPrecisionMode = Context->Config.x87StrictReducedPrecision;
|
||||
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
|
||||
return X80SoftFloat(src).ToF64_PreserveNan(&State.State);
|
||||
}
|
||||
return X80SoftFloat(src).ToF64(&State.State);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -50,14 +50,13 @@ DEF_OP(EntrypointOffset) {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
|
||||
auto Constant = Entry + Op->Offset;
|
||||
auto Dst = GetReg(Node);
|
||||
uint64_t Mask = ~0ULL;
|
||||
const auto OpSize = IROp->Size;
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Constant & Mask);
|
||||
InsertGuestRIPMove(GetReg(Node), Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
@@ -588,7 +587,7 @@ DEF_OP(ShiftFlags) {
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
|
||||
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, TMP1, &Done);
|
||||
(void)cbz(EmitSize, TMP1, &Done);
|
||||
{
|
||||
// PF/SF/ZF/OF
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
@@ -652,7 +651,7 @@ DEF_OP(ShiftFlags) {
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP2);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
|
||||
// TODO: Make RA less dumb so this can't happen (e.g. with late-kill).
|
||||
if (PFOutput != PFTemp) {
|
||||
@@ -669,7 +668,7 @@ DEF_OP(RotateFlags) {
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
(void)cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto BitSize = IR::OpSizeToSize(Op->Size) * 8;
|
||||
@@ -701,7 +700,7 @@ DEF_OP(RotateFlags) {
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(Extr) {
|
||||
@@ -767,14 +766,14 @@ DEF_OP(PDep) {
|
||||
// Now, they're copied, so we can start setting Dest (even if it overlaps with
|
||||
// one of them). Handle early exit case
|
||||
mov(EmitSize, Dest, 0);
|
||||
cbz(EmitSize, OrigMask, &Done);
|
||||
(void)cbz(EmitSize, OrigMask, &Done);
|
||||
|
||||
// Setup for first iteration
|
||||
neg(EmitSize, T0, Mask);
|
||||
and_(EmitSize, T0, T0, Mask);
|
||||
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
(void)Bind(&NextBit);
|
||||
sbfx(EmitSize, T1, Input, 0, 1);
|
||||
eor(EmitSize, Mask, Mask, T0);
|
||||
and_(EmitSize, T0, T1, T0);
|
||||
@@ -782,10 +781,10 @@ DEF_OP(PDep) {
|
||||
orr(EmitSize, Dest, Dest, T0);
|
||||
lsr(EmitSize, Input, Input, 1);
|
||||
and_(EmitSize, T0, Mask, T1);
|
||||
cbnz(EmitSize, T0, &NextBit);
|
||||
(void)cbnz(EmitSize, T0, &NextBit);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -821,27 +820,27 @@ DEF_OP(PExt) {
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
|
||||
cbz(EmitSize, Mask, &EarlyExit);
|
||||
(void)cbz(EmitSize, Mask, &EarlyExit);
|
||||
mov(EmitSize, MaskReg, Mask);
|
||||
mov(EmitSize, ValueReg, Input);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
cbz(EmitSize, MaskReg, &Done);
|
||||
(void)Bind(&NextBit);
|
||||
(void)cbz(EmitSize, MaskReg, &Done);
|
||||
clz(EmitSize, BitReg, MaskReg);
|
||||
lslv(EmitSize, ValueReg, ValueReg, BitReg);
|
||||
lslv(EmitSize, MaskReg, MaskReg, BitReg);
|
||||
extr(EmitSize, Dest, Dest, ValueReg, OpSizeBitsM1);
|
||||
bfc(EmitSize, MaskReg, OpSizeBitsM1, 1);
|
||||
b(&NextBit);
|
||||
(void)b(&NextBit);
|
||||
|
||||
// Early exit
|
||||
Bind(&EarlyExit);
|
||||
(void)Bind(&EarlyExit);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -909,7 +908,7 @@ DEF_OP(Div) {
|
||||
eor(EmitSize, TMP1, TMP1, Upper);
|
||||
|
||||
// If the sign bit matches then the result is zero
|
||||
cbz(EmitSize, TMP1, &Only64Bit);
|
||||
(void)cbz(EmitSize, TMP1, &Only64Bit);
|
||||
|
||||
// Long divide
|
||||
{
|
||||
@@ -928,17 +927,17 @@ DEF_OP(Div) {
|
||||
mov(EmitSize, Remainder, TMP2);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
(void)b(&LongDIVRet);
|
||||
}
|
||||
|
||||
Bind(&Only64Bit);
|
||||
(void)Bind(&Only64Bit);
|
||||
// 64-Bit only
|
||||
{
|
||||
sdiv(EmitSize, Quotient, Lower, Divisor);
|
||||
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
|
||||
}
|
||||
|
||||
Bind(&LongDIVRet);
|
||||
(void)Bind(&LongDIVRet);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown DIV Size: {}", OpSize); break;
|
||||
@@ -992,7 +991,7 @@ DEF_OP(UDiv) {
|
||||
|
||||
// Check the upper bits for zero
|
||||
// If the upper bits are zero then we can do a 64-bit divide
|
||||
cbz(EmitSize, Upper, &Only64Bit);
|
||||
(void)cbz(EmitSize, Upper, &Only64Bit);
|
||||
|
||||
// Long divide
|
||||
{
|
||||
@@ -1011,17 +1010,17 @@ DEF_OP(UDiv) {
|
||||
mov(EmitSize, Remainder, TMP2);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
(void)b(&LongDIVRet);
|
||||
}
|
||||
|
||||
Bind(&Only64Bit);
|
||||
(void)Bind(&Only64Bit);
|
||||
// 64-Bit only
|
||||
{
|
||||
udiv(EmitSize, Quotient, Lower, Divisor);
|
||||
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
|
||||
}
|
||||
|
||||
Bind(&LongDIVRet);
|
||||
(void)Bind(&LongDIVRet);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown LUDIV Size: {}", OpSize); break;
|
||||
|
||||
@@ -11,23 +11,18 @@ $end_info$
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl& CTX, FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op)); break;
|
||||
return CTX.Dispatcher->GetExitFunctionLinkerAddress();
|
||||
|
||||
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum) {
|
||||
Relocation MoveABI {};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.NamedThunkMove.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE};
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
@@ -38,9 +33,9 @@ void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR
|
||||
}
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(*CTX, Op);
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Lit {
|
||||
NamedSymbolLiteralPair Lit {
|
||||
.Lit = Pointer,
|
||||
.MoveABI =
|
||||
{
|
||||
@@ -48,92 +43,72 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
|
||||
{
|
||||
.Header =
|
||||
{
|
||||
.Offset = 0, // Set by PlaceNamedSymbolLiteral
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit) {
|
||||
switch (Lit.MoveABI.Header.Type) {
|
||||
case RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL:
|
||||
case RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
Lit.MoveABI.Header.Offset = GetCursorOffset();
|
||||
break;
|
||||
}
|
||||
|
||||
Bind(&Lit.Loc);
|
||||
default: ERROR_AND_DIE_FMT("Unknown relocation type for {}", __FUNCTION__);
|
||||
}
|
||||
|
||||
BindOrRestart(&Lit.Loc);
|
||||
dc64(Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
auto Arm64JITCore::InsertGuestRIPLiteral(uint64_t GuestRIP) -> NamedSymbolLiteralPair {
|
||||
return {
|
||||
.Lit = GuestRIP,
|
||||
.MoveABI =
|
||||
{
|
||||
.GuestRIP = {.Header =
|
||||
{
|
||||
.Offset = 0, // Set by PlaceNamedSymbolLiteral
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL,
|
||||
},
|
||||
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
|
||||
.GuestRIP = GuestRIP},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI {};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestRIP.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE};
|
||||
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
|
||||
MoveABI.GuestRIP.GuestRIP = Constant;
|
||||
MoveABI.GuestRIP.RegisterIndex = Reg.Idx();
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, false);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation> Relocations) {
|
||||
const auto OrigBase = GetBufferBase();
|
||||
const auto OrigSize = GetBufferSize();
|
||||
const auto OrigOffset = GetCursorOffset();
|
||||
|
||||
SetBuffer(reinterpret_cast<std::uint8_t*>(Code.data()), Code.size_bytes());
|
||||
for (auto& Reloc : Relocations) {
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc.NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
SetCursorOffset(Reloc.NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
dc64(Pointer);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(Reloc.NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(Reloc.GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations(uint64_t GuestBaseAddress) {
|
||||
// Rebase relocations to library base address
|
||||
for (auto& Relocation : Relocations) {
|
||||
switch (Relocation.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE:
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
Relocation.GuestRIP.GuestRIP -= GuestBaseAddress;
|
||||
break;
|
||||
}
|
||||
default:;
|
||||
}
|
||||
}
|
||||
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations() {
|
||||
return std::move(Relocations);
|
||||
}
|
||||
|
||||
|
||||
@@ -62,27 +62,27 @@ DEF_OP(CASPair) {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
|
||||
// This instruction sequence must be synced with HandleCASPAL_Armv8.
|
||||
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
|
||||
cmp(EmitSize, TMP2, Expected0);
|
||||
ccmp(EmitSize, TMP3, Expected1, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxp(EmitSize, TMP2, Desired0, Desired1, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP2, &LoopTop);
|
||||
mov(EmitSize, Dst0, Expected0);
|
||||
mov(EmitSize, Dst1, Expected1);
|
||||
|
||||
b(&LoopExpected);
|
||||
(void)b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
(void)Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst0, TMP2.R());
|
||||
mov(EmitSize, Dst1, TMP3.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopExpected);
|
||||
(void)Bind(&LoopExpected);
|
||||
|
||||
// Restore
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
@@ -114,7 +114,7 @@ DEF_OP(CAS) {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (IROp->Size == IR::OpSize::i8Bit) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
|
||||
@@ -123,38 +123,18 @@ DEF_OP(CAS) {
|
||||
} else {
|
||||
cmp(EmitSize, TMP2, Expected);
|
||||
}
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxr(SubEmitSize, TMP3, Desired, MemSrc);
|
||||
cbnz(EmitSize, TMP3, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP3, &LoopTop);
|
||||
mov(EmitSize, Dst, Expected);
|
||||
b(&LoopExpected);
|
||||
(void)b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
(void)Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst, TMP2.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopExpected);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
steorl(SubEmitSize, Src, MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
eor(EmitSize, TMP2, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
(void)Bind(&LoopExpected);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -179,10 +159,10 @@ DEF_OP(AtomicSwap) {
|
||||
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
stlxr(SubEmitSize, TMP4, Src, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
ubfm(EmitSize, GetReg(Node), TMP2, 0, IR::OpSizeAsBits(OpSize) - 1);
|
||||
}
|
||||
}
|
||||
@@ -199,11 +179,11 @@ DEF_OP(AtomicFetchAdd) {
|
||||
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
add(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -221,11 +201,11 @@ DEF_OP(AtomicFetchSub) {
|
||||
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
sub(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -243,11 +223,11 @@ DEF_OP(AtomicFetchAnd) {
|
||||
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
and_(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -264,11 +244,11 @@ DEF_OP(AtomicFetchCLR) {
|
||||
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
bic(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -285,11 +265,11 @@ DEF_OP(AtomicFetchOr) {
|
||||
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
orr(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -306,11 +286,11 @@ DEF_OP(AtomicFetchXor) {
|
||||
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
eor(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -326,20 +306,20 @@ DEF_OP(AtomicFetchNeg) {
|
||||
// Use a CAS loop to avoid needing to emulate unaligned LLSC atomics
|
||||
ldr(SubEmitSize, TMP2, MemSrc);
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
mov(EmitSize, TMP4, TMP2);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
casal(SubEmitSize, TMP2, TMP3, MemSrc);
|
||||
sub(EmitSize, TMP3, TMP2, TMP4);
|
||||
cbnz(EmitSize, TMP3, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP3, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -359,11 +339,11 @@ DEF_OP(TelemetrySetValue) {
|
||||
stsetl(ARMEmitter::SubRegSize::i64Bit, TMP1, TMP2);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
|
||||
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
|
||||
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
|
||||
cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -141,7 +141,7 @@ DEF_OP(ExitFunction) {
|
||||
if (!Op->CallReturnBlock.IsInvalid()) {
|
||||
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
|
||||
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
|
||||
adr(TMP1, &l_CallReturn);
|
||||
(void)adr(TMP1, &l_CallReturn);
|
||||
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
|
||||
} else {
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
|
||||
@@ -149,16 +149,16 @@ DEF_OP(ExitFunction) {
|
||||
} else if (Op->Hint == IR::BranchHint::CheckTF) {
|
||||
ARMEmitter::ForwardLabel TFUnset;
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, NewRIP);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
blr(TMP2);
|
||||
Bind(&TFUnset);
|
||||
(void)Bind(&TFUnset);
|
||||
}
|
||||
|
||||
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
|
||||
Bind(&l_CallReturn);
|
||||
(void)Bind(&l_CallReturn);
|
||||
#ifdef _M_ARM_64EC
|
||||
}
|
||||
#endif
|
||||
@@ -170,40 +170,38 @@ DEF_OP(ExitFunction) {
|
||||
// First try to pop from the call-ret stack, otherwise follow the normal path (but ending in a ret)
|
||||
ldp<ARMEmitter::IndexType::POST>(TMP1, TMP2, REG_CALLRET_SP, 0x10);
|
||||
sub(TMP1, TMP1, RipReg.X());
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
}
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
|
||||
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// arithmetic. ubfiz+add is marginally faster on Firestorm than
|
||||
// and+add(shift). Same performance on Cortex.
|
||||
static_assert(LookupCache::L1_ENTRIES_MASK == ((1u << 20) - 1));
|
||||
ubfiz(ARMEmitter::Size::i64Bit, TMP4, RipReg, 4, 20);
|
||||
add(TMP1, TMP1, TMP4);
|
||||
// L1Mask is pre-shifted.
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
add(TMP1, TMP1, TMP2);
|
||||
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
|
||||
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
sub(TMP1, TMP1, RipReg.X());
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
|
||||
Bind(&SkipFullLookup);
|
||||
(void)Bind(&SkipFullLookup);
|
||||
if (Op->Hint == IR::BranchHint::Call) {
|
||||
ARMEmitter::ForwardLabel l_CallReturn;
|
||||
if (!Op->CallReturnBlock.IsInvalid()) {
|
||||
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
|
||||
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
|
||||
adr(TMP1, &l_CallReturn);
|
||||
(void)adr(TMP1, &l_CallReturn);
|
||||
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
|
||||
} else {
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
|
||||
}
|
||||
blr(TMP2);
|
||||
Bind(&l_CallReturn);
|
||||
(void)Bind(&l_CallReturn);
|
||||
} else if (Op->Hint == IR::BranchHint::Return) {
|
||||
ret(TMP2);
|
||||
} else {
|
||||
@@ -224,7 +222,7 @@ DEF_OP(CondJump) {
|
||||
auto TrueTargetLabel = JumpTarget(Op->TrueBlock);
|
||||
|
||||
if (Op->FromNZCV) {
|
||||
b(MapCC(Op->Cond), TrueTargetLabel);
|
||||
b_OrRestart(MapCC(Op->Cond), TrueTargetLabel);
|
||||
} else {
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
@@ -237,16 +235,16 @@ DEF_OP(CondJump) {
|
||||
|
||||
if (Op->Cond == IR::CondClass::EQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbz(Size, Reg, TrueTargetLabel);
|
||||
cbz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond == IR::CondClass::NEQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbnz(Size, Reg, TrueTargetLabel);
|
||||
cbnz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond == IR::CondClass::TSTZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbz(Reg, Const, TrueTargetLabel);
|
||||
tbz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond == IR::CondClass::TSTNZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbnz(Reg, Const, TrueTargetLabel);
|
||||
tbnz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "CondJump expected simple condition");
|
||||
}
|
||||
@@ -262,16 +260,10 @@ DEF_OP(Syscall) {
|
||||
// X1: ThreadState
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) == FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
|
||||
// Need to spill all caller saved registers still
|
||||
GPRSpillMask = CALLER_GPR_MASK;
|
||||
FPRSpillMask = CALLER_FPR_MASK;
|
||||
}
|
||||
|
||||
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
|
||||
|
||||
@@ -305,117 +297,22 @@ DEF_OP(Syscall) {
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegs();
|
||||
PopDynamicRegs();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
}
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
|
||||
DEF_OP(InlineSyscall) {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
// Arguments are passed as follows:
|
||||
// X8: SyscallNumber - RA INTERSECT
|
||||
// X0: Arg0 & Return
|
||||
// X1: Arg1
|
||||
// X2: Arg2
|
||||
// X3: Arg3
|
||||
// X4: Arg4 - RA INTERSECT
|
||||
// X5: Arg5 - RA INTERSECT
|
||||
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
|
||||
|
||||
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
|
||||
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS - 1> RegArgs = {
|
||||
{ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5}};
|
||||
|
||||
bool Intersects {};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
|
||||
|
||||
SpillMask |= (1U << Reg.Idx());
|
||||
Intersects = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Now that we have claimed to be a syscall we can set up the arguments
|
||||
const auto EmitSize = CTX->Config.Is64BitMode() ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSubSize = CTX->Config.Is64BitMode() ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
|
||||
if (Intersects) {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (SpillMask & (1U << Reg.Idx())) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RDX, and RSP. Which have just been spilled
|
||||
// Just load back from the context.
|
||||
auto Correlation = GetX86RegRelationToARMReg(Reg);
|
||||
LOGMAN_THROW_A_FMT(Correlation != X86State::REG_INVALID, "Invalid register mapping");
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[Correlation]));
|
||||
} else {
|
||||
mov(EmitSize, RegArgs[i].R(), Reg);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i]));
|
||||
}
|
||||
}
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, Op->HostSyscallNumber);
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask, ~0U, ARMEmitter::Reg::r8, ARMEmitter::Reg::r1);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Result is now in x0
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -431,8 +328,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr));
|
||||
|
||||
auto thunkFn = static_cast<Context::ContextImpl*>(ThreadState->CTX)->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
|
||||
InsertNamedThunkRelocation(ARMEmitter::Reg::r2, Op->ThunkNameHash);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
@@ -458,7 +354,7 @@ DEF_OP(ValidateCode) {
|
||||
while (len >= Size) {
|
||||
LoadData();
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &Fail);
|
||||
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
|
||||
len -= Size;
|
||||
Offset += Size;
|
||||
}
|
||||
@@ -486,10 +382,10 @@ DEF_OP(ValidateCode) {
|
||||
|
||||
ARMEmitter::ForwardLabel End;
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
|
||||
b(&End);
|
||||
Bind(&Fail);
|
||||
b_OrRestart(&End);
|
||||
BindOrRestart(&Fail);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
|
||||
Bind(&End);
|
||||
BindOrRestart(&End);
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
|
||||
@@ -11,8 +11,6 @@ desc: Main glue logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
@@ -30,6 +28,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
@@ -37,7 +36,6 @@ $end_info$
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace {
|
||||
@@ -495,7 +493,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
}
|
||||
}
|
||||
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record, bool Call) {
|
||||
static void DirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record, bool Call) {
|
||||
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
|
||||
uintptr_t CallerAddress = JumpThunkStartAddress + Record->CallerOffset;
|
||||
auto BranchOffset = JumpThunkStartAddress / 4 - CallerAddress / 4;
|
||||
@@ -513,11 +511,12 @@ static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Co
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(CallerAddress), 4);
|
||||
}
|
||||
|
||||
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
static void IndirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
|
||||
uint32_t BranchInst = 0;
|
||||
ARMEmitter::Emitter BranchEmit(reinterpret_cast<uint8_t*>(&BranchInst), 4);
|
||||
BranchEmit.b(0x8);
|
||||
// Restore branch +2 instructions to jump to the linker block
|
||||
BranchEmit.b(0x2);
|
||||
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(BranchInst, std::memory_order::relaxed);
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
|
||||
@@ -540,7 +539,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
|
||||
auto lk_inval =
|
||||
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
HostCode = Thread->LookupCache->FindBlock(Thread, GuestRip);
|
||||
}
|
||||
if (!HostCode) {
|
||||
// Hold a reference to the code buffer, to avoid linking unmapped code if compilation triggers a recreation.
|
||||
@@ -565,7 +564,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
// Lock here is necessary to prevent simultaneous linking and delinking
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
|
||||
// For non-calls, this would extend into the block's code, however that's fine as an out-of-range adr would never
|
||||
// be generated avoiding any false positives.
|
||||
@@ -578,14 +577,12 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
|
||||
if (KnownCallMarkerInst == ExpectedKnownCallMarkerInst) {
|
||||
BranchEmit.bl(BranchOffset);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
DirectBlockDelinker(Frame, Record, true);
|
||||
});
|
||||
Thread->LookupCache->AddBlockLink(
|
||||
GuestRip, Record, [](FEXCore::Context::ExitFunctionLinkData* Record) { DirectBlockDelinker(Record, true); }, lk);
|
||||
} else {
|
||||
BranchEmit.b(BranchOffset);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
DirectBlockDelinker(Frame, Record, false);
|
||||
});
|
||||
Thread->LookupCache->AddBlockLink(
|
||||
GuestRip, Record, [](FEXCore::Context::ExitFunctionLinkData* Record) { DirectBlockDelinker(Record, false); }, lk);
|
||||
}
|
||||
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(CallerAddress)).store(BranchInst, std::memory_order::relaxed);
|
||||
@@ -604,7 +601,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(LdrInst, std::memory_order::relaxed);
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker, lk);
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
@@ -669,15 +666,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
|
||||
CurrentCodeBuffer = CodeBuffers.GetLatest();
|
||||
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (ParanoidTSO()) {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
|
||||
} else {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -689,13 +677,13 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
std::lock_guard lk(PrevCodeBuffer->LookupCache->WriteLock);
|
||||
auto lk = PrevCodeBuffer->LookupCache->AcquireWriteLock();
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache);
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
@@ -748,11 +736,11 @@ void Arm64JITCore::EmitTFCheck() {
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
tbz(TMP1, 1, &l_TFBlocked);
|
||||
(void)tbz(TMP1, 1, &l_TFBlocked);
|
||||
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
@@ -775,11 +763,11 @@ void Arm64JITCore::EmitTFCheck() {
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
|
||||
Bind(&l_TFBlocked);
|
||||
(void)Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
Bind(&l_TFUnset);
|
||||
(void)Bind(&l_TFUnset);
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
@@ -795,16 +783,16 @@ void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::ForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
(void)Bind(&l_NoSuspend);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF) {
|
||||
// Get the address of the JITCodeHeader and store in to the core state.
|
||||
// Two instruction cost, each 1 cycle.
|
||||
adr(TMP1, &HeaderLabel);
|
||||
adr_OrRestart(TMP1, &HeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
if (CheckTF) {
|
||||
@@ -821,36 +809,65 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
CallReturnTargets.clear();
|
||||
PendingJumpThunks.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
|
||||
const auto PrevNumAllocations = Relocations.size();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->DebugData = DebugData;
|
||||
this->IR = IR;
|
||||
RequiresFarARM64Jumps = false;
|
||||
SSANodeMultiplier = 24;
|
||||
|
||||
// Prepare restart via long jump in case branch encoding fails.
|
||||
// This uses UncheckedLongJump since we don't implement std::longjmp in WoA setups
|
||||
switch (static_cast<RestartOptions::Control>(FEXCore::UncheckedLongJump::SetJump(ThreadState->RestartJump))) {
|
||||
case RestartOptions::Control::Incoming:
|
||||
// Nothing
|
||||
break;
|
||||
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
|
||||
case RestartOptions::Control::NeedsLargerJITSpace:
|
||||
// Get rid of the claimed buffer immediately, we can't fit in it at all.
|
||||
TempAllocator.UnclaimBuffer();
|
||||
SSANodeMultiplier *= 2;
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Arm64 restart condition!");
|
||||
}
|
||||
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
JumpTargets.clear();
|
||||
CallReturnTargets.clear();
|
||||
PendingJumpThunks.clear();
|
||||
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
|
||||
|
||||
CodeData.EntryPoints.clear();
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = 0x1000 + SSACount * 24;
|
||||
// One page baseline, plus SSANodeMultipler bytes, plus another page for guard page.
|
||||
const uint32_t DesiredBufferRange = AlignUp(FEXCore::Utils::FEX_PAGE_SIZE * 2 + SSACount * SSANodeMultiplier, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
|
||||
// This minimizes lock contention of CodeBufferWriteMutex.
|
||||
auto TempCodeBuffer = TempAllocator.ReownOrClaimBuffer(BufferRange);
|
||||
SetBuffer(TempCodeBuffer, BufferRange);
|
||||
auto TempCodeBufferInfo = TempAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
|
||||
auto TempCodeBuffer = TempCodeBufferInfo.Ptr;
|
||||
const uint32_t UsableBufferRange = TempCodeBufferInfo.Size - FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
|
||||
SetBuffer(TempCodeBuffer, UsableBufferRange);
|
||||
|
||||
ThreadState->JITGuardPage = reinterpret_cast<uintptr_t>(TempCodeBuffer) + UsableBufferRange;
|
||||
ThreadState->JITGuardOverflowArgument = FEXCore::ToUnderlying(RestartOptions::Control::NeedsLargerJITSpace);
|
||||
|
||||
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Put the code header at the start of the data block.
|
||||
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
|
||||
Bind(&JITCodeHeaderLabel);
|
||||
(void)Bind(&JITCodeHeaderLabel);
|
||||
JITCodeHeader* CodeHeader = GetCursorAddress<JITCodeHeader*>();
|
||||
CursorIncrement(sizeof(JITCodeHeader));
|
||||
|
||||
@@ -898,7 +915,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
if (PendingTargetLabel->Backward.Location) {
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
b(PendingTargetLabel);
|
||||
b_OrRestart(PendingTargetLabel);
|
||||
PendingTargetLabel = nullptr;
|
||||
}
|
||||
|
||||
@@ -908,14 +925,14 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
const auto IsReturnTarget = CallReturnTargets.try_emplace(Node).first;
|
||||
if (PendingTargetLabel) {
|
||||
// If there is a fallthrough branch to this block, skip over the entrypoint code.
|
||||
b(Target);
|
||||
b_OrRestart(Target);
|
||||
} else if (PendingCallReturnTargetLabel && PendingCallReturnTargetLabel != &IsReturnTarget->second) {
|
||||
// If we just emitted a call, but the block we're now emitting is not the return block so don't fallthrough.
|
||||
b(PendingCallReturnTargetLabel);
|
||||
b_OrRestart(PendingCallReturnTargetLabel);
|
||||
}
|
||||
PendingCallReturnTargetLabel = nullptr;
|
||||
|
||||
Bind(&IsReturnTarget->second);
|
||||
BindOrRestart(&IsReturnTarget->second);
|
||||
CodeData.EntryPoints.emplace(BlockStartRIP, GetCursorAddress<uint8_t*>());
|
||||
DebugData->GuestOpcodes.push_back({BlockIROp->GuestEntryOffset, GetCursorAddress<uint8_t*>() - CodeData.BlockBegin});
|
||||
|
||||
@@ -924,18 +941,16 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
if (PendingCallReturnTargetLabel) {
|
||||
// If there is still a pending call return target, then the block we're emitting is not the return block so don't fallthrough.
|
||||
b(PendingCallReturnTargetLabel);
|
||||
b_OrRestart(PendingCallReturnTargetLabel);
|
||||
PendingCallReturnTargetLabel = nullptr;
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
Bind(Target);
|
||||
BindOrRestart(Target);
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, CodeNode); break
|
||||
#define REGISTER_OP(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, CodeNode); break
|
||||
|
||||
@@ -956,7 +971,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
if (PendingTargetLabel->Backward.Location) {
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
b(PendingTargetLabel);
|
||||
b_OrRestart(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
@@ -967,31 +982,37 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
ARMEmitter::ForwardLabel l_DoLink;
|
||||
uint64_t ThunkAddress = GetCursorAddress<uint64_t>();
|
||||
Bind(&PendingJumpThunk.Label);
|
||||
b(&l_DoLink);
|
||||
BindOrRestart(&PendingJumpThunk.Label);
|
||||
b_OrRestart(&l_DoLink);
|
||||
br(TMP1);
|
||||
Bind(&l_DoLink);
|
||||
BindOrRestart(&l_DoLink);
|
||||
ldr(TMP1, &l_ExitLink);
|
||||
blr(TMP1);
|
||||
|
||||
// This is a ExitFunctionLinkData struct
|
||||
Bind(&l_ExitLink);
|
||||
dc64(0); // HostCode
|
||||
dc64(PendingJumpThunk.GuestRIP); // GuestRIP
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(0); // HostCode
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
}
|
||||
|
||||
Bind(&l_ExitLink);
|
||||
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
BindOrRestart(&l_ExitLink);
|
||||
PlaceNamedSymbolLiteral(InsertNamedSymbolLiteral(RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER));
|
||||
|
||||
// CodeSize not including the header or tail data.
|
||||
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeBegin;
|
||||
|
||||
// Add the JitCodeTail
|
||||
// Add the JitCodeTail (written later)
|
||||
Align(alignof(JITCodeTail));
|
||||
auto JITBlockTailLocation = GetCursorAddress<uint8_t*>();
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
const auto JITBlockTailLocation = GetCursorAddress<uint8_t*>();
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
JITCodeTail JITBlockTail {
|
||||
.RIP = Entry,
|
||||
.GuestSize = Size,
|
||||
.SpinLockFutex = 0,
|
||||
.SingleInst = SingleInst,
|
||||
};
|
||||
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
@@ -1009,23 +1030,13 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
// FEXCore::Utils::vl64 GuestRIPOffset;
|
||||
// };
|
||||
|
||||
auto JITRIPEntriesBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
JITBlockTail->GuestSize = Size;
|
||||
JITBlockTail->SingleInst = SingleInst;
|
||||
JITBlockTail->SpinLockFutex = 0;
|
||||
|
||||
const auto JITRIPEntriesBegin = JITBlockTailLocation + sizeof(JITBlockTail);
|
||||
auto JITRIPEntriesLocation = JITRIPEntriesBegin;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
|
||||
JITBlockTail.NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail.OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
|
||||
@@ -1041,14 +1052,20 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
}
|
||||
|
||||
CursorIncrement(JITRIPEntriesLocation - JITRIPEntriesBegin);
|
||||
SetCursorOffset(JITRIPEntriesLocation - CodeData.BlockBegin);
|
||||
Align();
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
// Finalize and write block tail data
|
||||
JITBlockTail.Size = CodeData.Size;
|
||||
{
|
||||
auto PrevCur = GetCursorOffset();
|
||||
memcpy(JITBlockTailLocation, &JITBlockTail, sizeof(JITBlockTail));
|
||||
SetCursorOffset(JITBlockTailLocation - CodeData.BlockBegin + offsetof(JITCodeTail, RIP));
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(JITBlockTail.RIP));
|
||||
SetCursorOffset(PrevCur);
|
||||
}
|
||||
|
||||
// Migrate the compile output from temporary storage to the actual CodeBuffer.
|
||||
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
|
||||
@@ -1057,7 +1074,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// Query size of generated code
|
||||
const auto TempSize = GetCursorOffset();
|
||||
LOGMAN_THROW_A_FMT(TempSize <= BufferRange, "Exceeded bounds of temporary buffer ({:#x} vs {:#x})", TempSize, BufferRange);
|
||||
|
||||
// Bring CodeBuffer up to date
|
||||
{
|
||||
@@ -1065,7 +1081,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
"doesn't match up!\n");
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
|
||||
auto lk = ThreadState->LookupCache->AcquireWriteLock();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
@@ -1088,6 +1105,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
CodeBegin += Delta;
|
||||
|
||||
for (std::size_t Idx = PrevNumAllocations; Idx != Relocations.size(); ++Idx) {
|
||||
Relocations[Idx].Header.Offset += CodeBuffers.LatestOffset;
|
||||
}
|
||||
|
||||
// Copy over CodeBuffer contents
|
||||
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
|
||||
SetCursorOffset(CodeBuffers.LatestOffset + TempSize);
|
||||
|
||||
@@ -23,6 +23,7 @@ $end_info$
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
@@ -60,14 +61,27 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
const bool HostSupportsAVX256 {};
|
||||
const bool HostSupportsRPRES {};
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
struct RestartOptions {
|
||||
enum class Control : uint64_t {
|
||||
Incoming = 0,
|
||||
EnableFarARM64Jumps = 1,
|
||||
NeedsLargerJITSpace = 2,
|
||||
};
|
||||
};
|
||||
|
||||
// FEXCore makes assumptions in the JIT about certain conditions being true.
|
||||
// In the rare case when those assumptions are broken, FEX needs to safely restart the JIT.
|
||||
RestartOptions RestartControl {};
|
||||
bool RequiresFarARM64Jumps {};
|
||||
// Default to 6 instructions per SSA node.
|
||||
uint32_t SSANodeMultiplier {24};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
|
||||
ARMEmitter::BiDirectionalLabel* PendingCallReturnTargetLabel {};
|
||||
FEXCore::Context::ContextImpl* CTX {};
|
||||
@@ -331,14 +345,187 @@ private:
|
||||
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
|
||||
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
|
||||
auto& Thunk = PendingJumpThunks.back();
|
||||
Bind(&Thunk.Label);
|
||||
BindOrRestart(&Thunk.Label);
|
||||
if (Call) {
|
||||
bl(&Thunk.Label);
|
||||
bl_OrRestart(&Thunk.Label);
|
||||
} else {
|
||||
b(&Thunk.Label);
|
||||
b_OrRestart(&Thunk.Label);
|
||||
}
|
||||
}
|
||||
|
||||
// Restart helpers
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void bl_OrRestart(T* Label) {
|
||||
if (bl(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void b_OrRestart(T* Label) {
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void b_OrRestart(ARMEmitter::Condition Cond, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)b(InvertCondition(Cond), &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (b(Cond, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void cbz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)cbnz(s, rt, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (cbz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void cbnz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)cbz(s, rt, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (cbnz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void tbz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)tbnz(rt, Bit, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (tbz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void tbnz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)tbz(rt, Bit, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (tbnz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void adr_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
if (LongAddressGen(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Unable to encode long ADR.");
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (adr(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void adrp_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
if (LongAddressGen(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Unable to encode long ADRP.");
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (adrp(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void BindOrRestart(T* Label) {
|
||||
if (Bind(Label)) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (RequiresFarARM64Jumps) {
|
||||
// This should have been caught before this point.
|
||||
ERROR_AND_DIE_FMT("Unhandled long bind");
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
@@ -349,8 +536,6 @@ private:
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
@@ -387,19 +572,30 @@ private:
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Inserts a relocation for a constant value relative to the guest entrypoint
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertGuestRIPLiteral(uint64_t GuestRIP);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit);
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation>);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() override;
|
||||
/**
|
||||
* Returns any relocations generated since the last call to TakeRelocations.
|
||||
*
|
||||
* GuestBaseAddress must match the base virtual address to which the
|
||||
* input x86 binary is mapped.
|
||||
*/
|
||||
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) override;
|
||||
|
||||
/** @} */
|
||||
|
||||
@@ -430,17 +626,8 @@ private:
|
||||
|
||||
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
// Dynamic Dispatcher supporting operations
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
|
||||
|
||||
@@ -912,7 +912,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: ld1<ARMEmitter::SubRegSize::i8Bit>(TempDst.Q(), i, TempMemReg); break;
|
||||
@@ -923,7 +923,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
(void)Bind(&Skip);
|
||||
|
||||
if ((i + 1) != NumElements) {
|
||||
// Handle register rename to save a move.
|
||||
@@ -1013,7 +1013,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: st1<ARMEmitter::SubRegSize::i8Bit>(RegData.Q(), i, TempMemReg); break;
|
||||
@@ -1024,7 +1024,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
(void)Bind(&Skip);
|
||||
|
||||
if ((i + 1) != NumElements) {
|
||||
// Handle register rename to save a move.
|
||||
@@ -1102,7 +1102,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
PerformMove(ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// Skip if the mask's sign bit isn't set
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
|
||||
// Extract Index Element
|
||||
if ((IndexElement * IR::OpSizeToSize(VectorIndexSize)) >= 16) {
|
||||
@@ -1140,7 +1140,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, ElementSize); FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
(void)Bind(&Skip);
|
||||
}
|
||||
|
||||
if (NeedsDestTmp) {
|
||||
@@ -1874,7 +1874,7 @@ DEF_OP(MemSet) {
|
||||
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
(void)tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
|
||||
@@ -1922,7 +1922,7 @@ DEF_OP(MemSet) {
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
|
||||
// Early exit if zero count.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
ARMEmitter::ForwardLabel AgainInternal256Exit {};
|
||||
@@ -1939,50 +1939,50 @@ DEF_OP(MemSet) {
|
||||
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
|
||||
// single copy loop if size < 64.
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
|
||||
// Fill VTMP2 with the set pattern
|
||||
dup(SubRegSize, VTMP2.Q(), Value);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
|
||||
Bind(&AgainInternal256);
|
||||
(void)Bind(&AgainInternal256);
|
||||
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
|
||||
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal256);
|
||||
(void)tbz(TMP1, 63, &AgainInternal256);
|
||||
|
||||
Bind(&AgainInternal256Exit);
|
||||
(void)Bind(&AgainInternal256Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
Bind(&AgainInternal128);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128);
|
||||
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal128);
|
||||
(void)tbz(TMP1, 63, &AgainInternal128);
|
||||
|
||||
Bind(&AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&AgainInternal);
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemStoreTSO(Value, OpSize, SizeDirection);
|
||||
} else {
|
||||
MemStore(Value, OpSize, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
|
||||
Bind(&DoneInternal);
|
||||
(void)Bind(&DoneInternal);
|
||||
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
@@ -2012,12 +2012,12 @@ DEF_OP(MemSet) {
|
||||
EmitMemset(Direction);
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
(void)b(&Done);
|
||||
(void)Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
}
|
||||
@@ -2067,7 +2067,7 @@ DEF_OP(MemCpy) {
|
||||
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
(void)tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
|
||||
@@ -2164,7 +2164,7 @@ DEF_OP(MemCpy) {
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
|
||||
// Early exit if zero count.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
ARMEmitter::ForwardLabel AbsPos {};
|
||||
@@ -2174,11 +2174,11 @@ DEF_OP(MemCpy) {
|
||||
ARMEmitter::BackwardLabel AgainInternal256 {};
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP4, TMP2, TMP3);
|
||||
tbz(TMP4, 63, &AbsPos);
|
||||
(void)tbz(TMP4, 63, &AbsPos);
|
||||
neg(ARMEmitter::Size::i64Bit, TMP4, TMP4);
|
||||
Bind(&AbsPos);
|
||||
(void)Bind(&AbsPos);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
|
||||
tbnz(TMP4, 63, &AgainInternal);
|
||||
(void)tbnz(TMP4, 63, &AgainInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
@@ -2190,30 +2190,30 @@ DEF_OP(MemCpy) {
|
||||
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
|
||||
// single copy loop if size < 64.
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
|
||||
Bind(&AgainInternal256);
|
||||
(void)Bind(&AgainInternal256);
|
||||
MemCpy(32, 32 * Direction);
|
||||
MemCpy(32, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal256);
|
||||
(void)tbz(TMP1, 63, &AgainInternal256);
|
||||
|
||||
Bind(&AgainInternal256Exit);
|
||||
(void)Bind(&AgainInternal256Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
Bind(&AgainInternal128);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128);
|
||||
MemCpy(32, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal128);
|
||||
(void)tbz(TMP1, 63, &AgainInternal128);
|
||||
|
||||
Bind(&AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
@@ -2221,16 +2221,16 @@ DEF_OP(MemCpy) {
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&AgainInternal);
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemCpyTSO(OpSize, SizeDirection);
|
||||
} else {
|
||||
MemCpy(OpSize, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
|
||||
Bind(&DoneInternal);
|
||||
(void)Bind(&DoneInternal);
|
||||
|
||||
// Needs to use temporaries just in case of overwrite
|
||||
mov(TMP1, MemRegDest.X());
|
||||
@@ -2288,186 +2288,15 @@ DEF_OP(MemCpy) {
|
||||
for (int32_t Direction : {1, -1}) {
|
||||
EmitMemcpy(Direction);
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
(void)b(&Done);
|
||||
(void)Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
const auto Dst = GetReg(Node);
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: ldapurh(Dst, MemReg, Offset); break;
|
||||
case IR::OpSize::i32Bit: ldapur(Dst.W(), MemReg, Offset); break;
|
||||
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: ldaprh(Dst.W(), MemReg); break;
|
||||
case IR::OpSize::i32Bit: ldapr(Dst.W(), MemReg); break;
|
||||
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: ldarb(Dst, MemReg); break;
|
||||
case IR::OpSize::i16Bit: ldarh(Dst, MemReg); break;
|
||||
case IR::OpSize::i32Bit: ldar(Dst.W(), MemReg); break;
|
||||
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
ldarb(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
ldarh(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case IR::OpSize::i32Bit:
|
||||
ldar(TMP1.W(), MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case IR::OpSize::i64Bit:
|
||||
ldar(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
|
||||
break;
|
||||
case IR::OpSize::i128Bit:
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
|
||||
clrex();
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case IR::OpSize::i256Bit:
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
|
||||
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
|
||||
case IR::OpSize::i64Bit: stlur(Src.X(), MemReg, Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
|
||||
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
|
||||
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
|
||||
case IR::OpSize::i64Bit: stlr(Src.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
|
||||
stlrb(TMP1, MemReg);
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
|
||||
stlrh(TMP1, MemReg);
|
||||
break;
|
||||
case IR::OpSize::i32Bit:
|
||||
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1.W(), MemReg);
|
||||
break;
|
||||
case IR::OpSize::i64Bit:
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1, MemReg);
|
||||
break;
|
||||
case IR::OpSize::i128Bit: {
|
||||
// Move vector to GPRs
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP2, Src, 1);
|
||||
ARMEmitter::BackwardLabel B;
|
||||
Bind(&B);
|
||||
|
||||
// ldaxp must not have both the destination registers be the same
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i256Bit: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
|
||||
dmb(ARMEmitter::BarrierScope::SY);
|
||||
|
||||
@@ -1,79 +1,89 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
enum class RelocationTypes : uint8_t {
|
||||
enum class RelocationTypes : uint32_t {
|
||||
// 8 byte literal in memory for symbol
|
||||
// Aligned to struct RelocNamedSymbolLiteral
|
||||
RELOC_NAMED_SYMBOL_LITERAL,
|
||||
|
||||
// Fixed size named thunk move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// 4 instruction constant generation
|
||||
// Aligned to struct RelocNamedThunkMove
|
||||
RELOC_NAMED_THUNK_MOVE,
|
||||
|
||||
// 8 byte literal (relative to binary base address)
|
||||
RELOC_GUEST_RIP_LITERAL,
|
||||
|
||||
// Fixed size guest RIP move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocGuestRIPMove
|
||||
// 4 instruction constant generation
|
||||
// Aligned to struct RelocGuestRIP
|
||||
RELOC_GUEST_RIP_MOVE,
|
||||
};
|
||||
|
||||
struct RelocationTypeHeader final {
|
||||
struct FEX_PACKED RelocationHeader final {
|
||||
// Offset to the relocated host code data
|
||||
uint64_t Offset {};
|
||||
|
||||
RelocationTypes Type;
|
||||
};
|
||||
|
||||
struct RelocNamedSymbolLiteral final {
|
||||
enum class NamedSymbol : uint8_t {
|
||||
enum class NamedSymbol : uint32_t {
|
||||
///< Thread specific relocations
|
||||
// JIT Literal pointers
|
||||
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
|
||||
};
|
||||
|
||||
RelocationTypeHeader Header {};
|
||||
RelocationHeader Header {};
|
||||
|
||||
NamedSymbol Symbol;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
uint32_t Pad[8];
|
||||
};
|
||||
|
||||
struct RelocNamedThunkMove final {
|
||||
RelocationTypeHeader Header {};
|
||||
RelocationHeader Header {};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
uint32_t RegisterIndex;
|
||||
|
||||
// The thunk SHA256 hash
|
||||
IR::SHA256Sum Symbol;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
};
|
||||
|
||||
struct RelocGuestRIPMove final {
|
||||
RelocationTypeHeader Header {};
|
||||
struct RelocGuestRIP final {
|
||||
RelocationHeader Header {};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
// GPR index the constant is being moved to (for non-literal relocations)
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
char Pad[3];
|
||||
|
||||
// The unrelocated RIP that is being moved
|
||||
// The base RIP (to be moved by the register for non-literal relocations).
|
||||
// In a serialized code cache, this is relative to the binary base address.
|
||||
uint64_t GuestRIP;
|
||||
|
||||
uint32_t pad2[6] {};
|
||||
};
|
||||
|
||||
union Relocation {
|
||||
RelocationTypeHeader Header {};
|
||||
RelocationHeader Header {};
|
||||
|
||||
RelocNamedSymbolLiteral NamedSymbolLiteral;
|
||||
// This makes our union of relocations at least 48 bytes
|
||||
// It might be more efficient to not use a union
|
||||
RelocNamedThunkMove NamedThunkMove;
|
||||
|
||||
RelocGuestRIPMove GuestRIPMove;
|
||||
RelocGuestRIP GuestRIP;
|
||||
};
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl&, RelocNamedSymbolLiteral::NamedSymbol);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -15,7 +15,7 @@ $end_info$
|
||||
|
||||
namespace FEXCore {
|
||||
GuestToHostMap::GuestToHostMap()
|
||||
: BlockLinks_mbr {fextl::pmr::get_default_resource()} {
|
||||
: BlockLinks_mbr {"FEXMem_BlockLinks"} {
|
||||
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
|
||||
// Setup our PMR map.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
@@ -24,7 +24,7 @@ GuestToHostMap::GuestToHostMap()
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
: ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE + MAX_L1_SIZE;
|
||||
|
||||
// Block cache ends up looking like this
|
||||
// PageMemoryMap[VirtualMemoryRegion >> 12]
|
||||
@@ -39,6 +39,10 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// We need one pointer per page of virtual memory
|
||||
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize, false, false));
|
||||
LOGMAN_THROW_A_FMT(PagePointer != -1ULL, "Failed to allocate PagePointer");
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Lookup", reinterpret_cast<void*>(PagePointer),
|
||||
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE);
|
||||
CTX->SyscallHandler->MarkOvercommitRange(PagePointer, TotalCacheSize);
|
||||
|
||||
// Allocate our memory backing our pages
|
||||
@@ -46,14 +50,21 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// XXX: We can drop down to 16KB if we store 4byte offsets from the code base
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / 4096 * 8;
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8;
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = PageMemory + CODE_SIZE;
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Lookup_L1", reinterpret_cast<void*>(L1Pointer), MAX_L1_SIZE);
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
|
||||
if (DynamicL1Cache()) {
|
||||
// Start at minimum size when dynamic.
|
||||
L1PointerMask = MIN_L1_ENTRIES - 1;
|
||||
} else {
|
||||
// Start at maximum instead.
|
||||
L1PointerMask = MAX_L1_ENTRIES - 1;
|
||||
}
|
||||
}
|
||||
|
||||
LookupCache::~LookupCache() {
|
||||
@@ -64,31 +75,27 @@ LookupCache::~LookupCache() {
|
||||
// These will get freed when their memory allocators are deallocated.
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
void LookupCache::ClearL2Cache(const FEXCore::LookupCacheBaseLockToken& lk) {
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer),
|
||||
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE, false);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
void LookupCache::ClearThreadLocalCaches() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
void LookupCache::ClearThreadLocalCaches(const LookupCacheWriteLockToken&) {
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
CachedCodePages.clear();
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
void LookupCache::ClearCache(const LookupCacheWriteLockToken& lk) {
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
|
||||
ClearThreadLocalCaches(lk);
|
||||
Shared->ClearCache(lk);
|
||||
}
|
||||
|
||||
void GuestToHostMap::ClearCache(const LockToken&) {
|
||||
void GuestToHostMap::ClearCache(const LookupCacheWriteLockToken&) {
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
|
||||
@@ -2,30 +2,57 @@
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/SHMStats.h>
|
||||
#include "Utils/WritePriorityMutex.h"
|
||||
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/robin_set.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
struct LookupCacheBaseLockToken {
|
||||
protected:
|
||||
// Protected constructor - only derived classes can construct
|
||||
LookupCacheBaseLockToken() = default;
|
||||
};
|
||||
|
||||
struct LookupCacheWriteLockToken : public LookupCacheBaseLockToken {
|
||||
private:
|
||||
// Only constructible by GuestToHostMap
|
||||
friend struct GuestToHostMap;
|
||||
LookupCacheWriteLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
|
||||
: Lock {Mutex} {}
|
||||
std::lock_guard<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
|
||||
};
|
||||
|
||||
struct LookupCacheReadLockToken : public LookupCacheBaseLockToken {
|
||||
private:
|
||||
// Only constructible by GuestToHostMap
|
||||
friend struct GuestToHostMap;
|
||||
LookupCacheReadLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
|
||||
: Lock {Mutex} {}
|
||||
std::shared_lock<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
|
||||
};
|
||||
|
||||
struct GuestToHostMap {
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
struct LockToken {
|
||||
std::lock_guard<std::recursive_mutex> Lock;
|
||||
};
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex Lock {};
|
||||
|
||||
[[nodiscard]]
|
||||
LockToken AcquireLock() {
|
||||
return LockToken {std::lock_guard {WriteLock}};
|
||||
LookupCacheWriteLockToken AcquireWriteLock() {
|
||||
return LookupCacheWriteLockToken {Lock};
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
LookupCacheReadLockToken AcquireReadLock() {
|
||||
return LookupCacheReadLockToken {Lock};
|
||||
}
|
||||
|
||||
struct BlockLinkTag {
|
||||
@@ -49,53 +76,72 @@ struct GuestToHostMap {
|
||||
// walking each block member and destructing objects.
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
fextl::pmr::named_monotonic_page_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType* BlockLinks;
|
||||
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
struct BlockEntry {
|
||||
uint64_t HostCode;
|
||||
fextl::vector<uint64_t> CodePages;
|
||||
};
|
||||
|
||||
fextl::robin_map<uint64_t, BlockEntry> BlockList;
|
||||
|
||||
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
||||
|
||||
GuestToHostMap();
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode, const LockToken&) {
|
||||
const BlockEntry& AddBlockMapping(uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
|
||||
// This may replace an existing mapping
|
||||
// NOTE: Generally no previous entry should exist, however there is one exception:
|
||||
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
||||
// may already contain the block address. Since is comparatively rare, we'll just leak
|
||||
// one of the two blocks in this case.
|
||||
BlockList[Address] = (uintptr_t)HostCode;
|
||||
return BlockList.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, CodePages}).first->second;
|
||||
}
|
||||
|
||||
std::optional<uintptr_t> FindBlock(uint64_t Address, const LockToken&) {
|
||||
const BlockEntry* FindBlock(uint64_t Address, const LookupCacheReadLockToken&) {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
if (HostCode == BlockList.end()) {
|
||||
return std::nullopt;
|
||||
return nullptr;
|
||||
}
|
||||
return HostCode->second;
|
||||
return &HostCode->second;
|
||||
}
|
||||
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LockToken&) {
|
||||
bool Erase(uint64_t Address, const LookupCacheWriteLockToken&) {
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second(Frame, it->first.HostLink);
|
||||
it->second(it->first.HostLink);
|
||||
}
|
||||
|
||||
// Remove from BlockList
|
||||
return BlockList.erase(Address) != 0;
|
||||
}
|
||||
|
||||
void InvalidateRange(uint64_t Start, uint64_t Length) {
|
||||
auto lk = AcquireWriteLock();
|
||||
|
||||
auto lower = CodePages.lower_bound(Start >> 12);
|
||||
auto upper = CodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (const auto& Entry : it->second) {
|
||||
Erase(Entry, lk);
|
||||
}
|
||||
}
|
||||
CodePages.erase(lower, upper);
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LockToken&) {
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken&) {
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LockToken&) {
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LookupCacheWriteLockToken&) {
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
@@ -107,7 +153,7 @@ struct GuestToHostMap {
|
||||
return rv;
|
||||
}
|
||||
|
||||
void ClearCache(const LockToken&);
|
||||
void ClearCache(const LookupCacheWriteLockToken&);
|
||||
};
|
||||
|
||||
class LookupCache {
|
||||
@@ -122,122 +168,199 @@ public:
|
||||
|
||||
// Swaps out the underlying GuestToHostMap and clears all associated caches.
|
||||
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap) {
|
||||
ClearThreadLocalCaches();
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap, const LookupCacheWriteLockToken& lk) {
|
||||
ClearThreadLocalCaches(lk);
|
||||
Shared = &NewMap;
|
||||
}
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
uintptr_t FindBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) {
|
||||
// Try L1, no lock needed
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
// L2 and L3 need to be locked
|
||||
auto lk = Shared->AcquireLock();
|
||||
uintptr_t HostPtr {};
|
||||
{
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheReadLockTime : nullptr);
|
||||
auto lk = Shared->AcquireReadLock();
|
||||
LockTime.reset();
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
if (!DisableL2Cache()) {
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == Address) {
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
return L1Entry.HostCode;
|
||||
if (BlockPointers[PageOffset].GuestCode == Address) {
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
HostPtr = L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!HostPtr) {
|
||||
// Try L3
|
||||
auto Entry = Shared->FindBlock(Address, lk);
|
||||
if (Entry) {
|
||||
CacheBlockMapping(Address, *Entry, false, lk);
|
||||
HostPtr = Entry->HostCode;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Try L3
|
||||
auto HostCode = Shared->FindBlock(Address, lk);
|
||||
if (HostCode) {
|
||||
CacheBlockMapping(Address, HostCode.value());
|
||||
return HostCode.value();
|
||||
if (HostPtr && DynamicL1Cache()) {
|
||||
UpdateDynamicL1Stats(Thread);
|
||||
}
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedCacheMissCount, 1);
|
||||
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// If host pointer was found in L2 or L3, then add it to the counter.
|
||||
// Keeping track not L1 misses, but specifically L2/L3 hits.
|
||||
++L2L3CacheHits;
|
||||
|
||||
const auto CurrentTime = std::chrono::system_clock::now();
|
||||
const auto Period = CurrentTime - LastPeriod;
|
||||
if (Period >= SamplePeriod) {
|
||||
// If larger than the sample period then check if we need to increase L1 cache size.
|
||||
const double AveragePerSecond = static_cast<double>(L2L3CacheHits) /
|
||||
static_cast<double>(std::chrono::duration_cast<std::chrono::milliseconds>(Period).count()) * 1000.0;
|
||||
|
||||
if (AveragePerSecond >= DynamicL1CacheIncreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries < MAX_L1_ENTRIES) {
|
||||
CurrentL1Entries <<= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
}
|
||||
} else if (AveragePerSecond < DynamicL1CacheDecreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries > MIN_L1_ENTRIES) {
|
||||
CurrentL1Entries >>= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Madvise the entries that we are dropping. Gives the memory back to the OS.
|
||||
LookupCacheEntry* FirstZeroL1Entry = &reinterpret_cast<LookupCacheEntry*>(L1Pointer)[CurrentL1Entries];
|
||||
size_t ZeroMemorySize = (MAX_L1_ENTRIES - CurrentL1Entries) * sizeof(LookupCacheEntry);
|
||||
FEXCore::Allocator::VirtualDontNeed(FirstZeroL1Entry, ZeroMemorySize, false);
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
}
|
||||
}
|
||||
|
||||
// Update Last period to start again.
|
||||
LastPeriod = CurrentTime;
|
||||
L2L3CacheHits = 0;
|
||||
}
|
||||
}
|
||||
|
||||
GuestToHostMap* Shared = nullptr;
|
||||
|
||||
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
LockTime.reset();
|
||||
|
||||
return Shared->AddBlockExecutableRange(Addresses, Start, Length, lk);
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
LockTime.reset();
|
||||
|
||||
Shared->AddBlockMapping(Address, HostCode, lk);
|
||||
const auto& Entry = Shared->AddBlockMapping(Address, CodePages, HostCode, lk);
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
CacheBlockMapping(Address, Entry, true, lk);
|
||||
}
|
||||
|
||||
// NOTE: It's the caller's responsibility to call Erase() for all other
|
||||
// GuestToHostMaps that share the same LookupCache. Otherwise, the
|
||||
// L1/L2 caches will contain stale references to deallocated memory.
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
bool ErasedAny = Shared->Erase(Frame, Address, lk);
|
||||
|
||||
// Invalidates L1/L2 for a given guest block
|
||||
void InvalidateCache(uint64_t Address, const LookupCacheWriteLockToken& lk) {
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = 0;
|
||||
ErasedAny = true;
|
||||
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
|
||||
// This is a soft guarantee for cross thread invalidation, as atomics are not used
|
||||
// and it hasn't been thoroughly tested
|
||||
}
|
||||
|
||||
// Do full map
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
if (!DisableL2Cache()) {
|
||||
// Do full map
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// Page for this code didn't even exist, nothing to do
|
||||
return ErasedAny;
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// Page for this code didn't even exist, nothing to do
|
||||
return;
|
||||
}
|
||||
|
||||
// Page exists, just set the offset to zero
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
BlockPointers[PageOffset].GuestCode = 0;
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
}
|
||||
|
||||
// Page exists, just set the offset to zero
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
BlockPointers[PageOffset].GuestCode = 0;
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
// Invalidates all L1/L2 entries for all guest block that intersect the given range
|
||||
bool InvalidateCacheRange(uint64_t Start, uint64_t Length) {
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
|
||||
auto lower = CachedCodePages.lower_bound(Start >> 12);
|
||||
auto upper = CachedCodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (const auto& Entry : it->second) {
|
||||
InvalidateCache(Entry, lk);
|
||||
}
|
||||
}
|
||||
bool ret = upper != lower;
|
||||
CachedCodePages.erase(lower, upper);
|
||||
return ret;
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken& lk) {
|
||||
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
void ClearThreadLocalCaches();
|
||||
void ClearCache(const LookupCacheWriteLockToken&);
|
||||
void ClearL2Cache(const LookupCacheBaseLockToken&);
|
||||
void ClearThreadLocalCaches(const LookupCacheWriteLockToken&);
|
||||
|
||||
uintptr_t GetL1Pointer() const {
|
||||
return L1Pointer;
|
||||
}
|
||||
uintptr_t GetScaledL1PointerMask() const {
|
||||
return L1PointerMask << FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry));
|
||||
}
|
||||
uintptr_t GetPagePointer() const {
|
||||
return PagePointer;
|
||||
}
|
||||
@@ -245,9 +368,6 @@ public:
|
||||
return VirtualMemSize;
|
||||
}
|
||||
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
@@ -255,45 +375,52 @@ public:
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
auto AcquireLock() {
|
||||
return Shared->AcquireLock();
|
||||
auto AcquireWriteLock() {
|
||||
return Shared->AcquireWriteLock();
|
||||
}
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = HostCode;
|
||||
|
||||
// Do ful map
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
// Allocate one now if we can
|
||||
uintptr_t NewPageBacking = AllocateBackingForPage();
|
||||
if (!NewPageBacking) {
|
||||
// Couldn't allocate, clear L2 and retry
|
||||
ClearL2Cache();
|
||||
CacheBlockMapping(Address, HostCode);
|
||||
return;
|
||||
}
|
||||
Pointers[Address] = NewPageBacking;
|
||||
LocalPagePointer = NewPageBacking;
|
||||
void CacheBlockMapping(uint64_t Address, const GuestToHostMap::BlockEntry& Entry, bool L1Only, const LookupCacheBaseLockToken& lk) {
|
||||
for (const auto& CodePage : Entry.CodePages) {
|
||||
CachedCodePages[CodePage >> 12].insert(Address);
|
||||
}
|
||||
|
||||
// Add the new pointer to the page block
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = Entry.HostCode;
|
||||
|
||||
// This silently replaces existing mappings
|
||||
BlockPointers[PageOffset].GuestCode = FullAddress;
|
||||
BlockPointers[PageOffset].HostCode = HostCode;
|
||||
if (!DisableL2Cache() && !L1Only) {
|
||||
// Do ful map
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
// Allocate one now if we can
|
||||
uintptr_t NewPageBacking = AllocateBackingForPage();
|
||||
if (!NewPageBacking) {
|
||||
// Couldn't allocate, clear L2 and retry
|
||||
ClearL2Cache(lk);
|
||||
CacheBlockMapping(Address, Entry, false, lk);
|
||||
return;
|
||||
}
|
||||
Pointers[Address] = NewPageBacking;
|
||||
LocalPagePointer = NewPageBacking;
|
||||
}
|
||||
|
||||
// Add the new pointer to the page block
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
// This silently replaces existing mappings
|
||||
BlockPointers[PageOffset].GuestCode = FullAddress;
|
||||
BlockPointers[PageOffset].HostCode = Entry.HostCode;
|
||||
}
|
||||
}
|
||||
|
||||
uintptr_t AllocateBackingForPage() {
|
||||
@@ -310,19 +437,38 @@ private:
|
||||
return PageMemory + NewBase;
|
||||
}
|
||||
|
||||
// Maps from a page index to all blocks in the page that have at some point been fetched into L1/L2
|
||||
fextl::map<uint64_t, fextl::robin_set<uint64_t>> CachedCodePages;
|
||||
|
||||
uintptr_t PagePointer;
|
||||
uintptr_t PageMemory;
|
||||
uintptr_t L1Pointer;
|
||||
uintptr_t L1PointerMask;
|
||||
|
||||
size_t TotalCacheSize;
|
||||
|
||||
// Start with 8k entries in L1 to give 128KB of L1 cache to each thread.
|
||||
// Max out at 1 million entries to give each thread 16MB of L1 cache maximum.
|
||||
constexpr static size_t MIN_L1_ENTRIES = 8 * 1024; // Must be a power of 2
|
||||
constexpr static size_t MAX_L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t L1_SIZE = L1_ENTRIES * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t SIZE_PER_PAGE = FEXCore::Utils::FEX_PAGE_SIZE * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t MAX_L1_SIZE = MAX_L1_ENTRIES * sizeof(LookupCacheEntry);
|
||||
|
||||
size_t AllocateOffset {};
|
||||
|
||||
FEXCore::Context::ContextImpl* ctx;
|
||||
uint64_t VirtualMemSize {};
|
||||
|
||||
size_t CurrentL1Entries = MIN_L1_ENTRIES;
|
||||
uint64_t L2L3CacheHits {};
|
||||
std::chrono::time_point<std::chrono::system_clock> LastPeriod {};
|
||||
constexpr static std::chrono::seconds SamplePeriod {1};
|
||||
FEX_CONFIG_OPT(DynamicL1CacheIncreaseCountHeuristic, DYNAMICL1CACHEINCREASECOUNTHEURISTIC);
|
||||
FEX_CONFIG_OPT(DynamicL1CacheDecreaseCountHeuristic, DYNAMICL1CACHEDECREASECOUNTHEURISTIC);
|
||||
|
||||
FEX_CONFIG_OPT(DynamicL1Cache, DYNAMICL1CACHE);
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -28,7 +28,6 @@ $end_info$
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
@@ -51,8 +50,6 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
FEXCore::X86State::REG_RSI, FEXCore::X86State::REG_RDI, FEXCore::X86State::REG_RBP,
|
||||
};
|
||||
|
||||
SyscallFlags DefaultSyscallFlags = FEXCore::IR::SyscallFlags::DEFAULT;
|
||||
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
if (OSABI == FEXCore::HLE::SyscallOSABI::OS_LINUX64) {
|
||||
NumArguments = GPRIndexes_64.size();
|
||||
@@ -64,7 +61,6 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
// All registers will be spilled before the syscall and filled afterwards so no JIT-side argument handling is necessary.
|
||||
NumArguments = 0;
|
||||
GPRIndexes = nullptr;
|
||||
DefaultSyscallFlags = FEXCore::IR::SyscallFlags::NORETURNEDRESULT;
|
||||
} else {
|
||||
ERROR_AND_DIE_FMT("Unhandled OSABI syscall");
|
||||
}
|
||||
@@ -98,9 +94,10 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
}
|
||||
|
||||
FlushRegisterCache();
|
||||
auto SyscallOp = _Syscall(Arguments[0], Arguments[1], Arguments[2], Arguments[3], Arguments[4], Arguments[5], Arguments[6], DefaultSyscallFlags);
|
||||
auto SyscallOp = _Syscall(Arguments[0], Arguments[1], Arguments[2], Arguments[3], Arguments[4], Arguments[5], Arguments[6]);
|
||||
|
||||
if ((DefaultSyscallFlags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Generic ABI doesn't store result in RAX.
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
StoreGPRRegister(X86State::REG_RAX, SyscallOp);
|
||||
}
|
||||
|
||||
@@ -153,12 +150,6 @@ void OpDispatchBuilder::NOPOp(OpcodeArgs) {}
|
||||
void OpDispatchBuilder::RETOp(OpcodeArgs) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
// ABI Optimization: Flags don't survive calls or rets
|
||||
if (CTX->Config.ABILocalFlags) {
|
||||
_InvalidateFlags(~0UL); // all flags
|
||||
InvalidatePF_AF();
|
||||
}
|
||||
|
||||
Ref SP = _RMWHandle(LoadGPRRegister(X86State::REG_RSP));
|
||||
Ref NewRIP = Pop(GPRSize, SP);
|
||||
|
||||
@@ -522,25 +513,18 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) {
|
||||
|
||||
BlockSetRIP = true;
|
||||
|
||||
// ABI Optimization: Flags don't survive calls or rets
|
||||
if (CTX->Config.ABILocalFlags) {
|
||||
_InvalidateFlags(~0UL); // all flags
|
||||
InvalidatePF_AF();
|
||||
}
|
||||
|
||||
// Call instruction only uses up to 32-bit signed displacement
|
||||
int64_t TargetOffset = Op->Src[0].Literal();
|
||||
const int64_t TargetOffset = Op->Src[0].Literal();
|
||||
|
||||
auto ConstantPC = GetRelocatedPC(Op);
|
||||
const auto ConstantPC = GetRelocatedPC(Op);
|
||||
|
||||
// Push the return address.
|
||||
Push(GPRSize, ConstantPC);
|
||||
|
||||
const uint64_t NextRIP = Op->PC + Op->InstSize;
|
||||
uint64_t TargetRIP = NextRIP + TargetOffset;
|
||||
|
||||
if (NextRIP != TargetRIP) {
|
||||
if (TargetOffset != 0) {
|
||||
// Store the RIP
|
||||
const uint64_t NextRIP = Op->PC + Op->InstSize;
|
||||
|
||||
ExitRelocatedPC(Op, TargetOffset, BranchHint::Call, ConstantPC, [&]() {
|
||||
auto CallReturnJumpTarget = JumpTargets.find(NextRIP);
|
||||
if (CallReturnJumpTarget != JumpTargets.end() && CallReturnJumpTarget->second.IsEntryPoint) {
|
||||
@@ -2765,7 +2749,8 @@ void OpDispatchBuilder::NOTOp(OpcodeArgs) {
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Ref DestMem = MakeSegmentAddress(Op, Op->Dest);
|
||||
_AtomicXor(Size, MaskConst, DestMem);
|
||||
// Result unused
|
||||
_AtomicFetchXor(Size, MaskConst, DestMem);
|
||||
} else if (!Op->Dest.IsGPR()) {
|
||||
// GPR version plays fast and loose with sizes, be safe for memory tho.
|
||||
Ref Src = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
@@ -3214,7 +3199,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) {
|
||||
LogMan::Msg::EFmt("Can't handle adddress size");
|
||||
LogMan::Msg::EFmt("STOSOp: Can't handle address size override (OP: 0x{:04X}, Flags: 0x{:08X})", Op->OP, Op->Flags);
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
}
|
||||
@@ -3230,7 +3215,11 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
Ref Dest = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
// Store to memory where RDI points
|
||||
_StoreMemGPRAutoTSO(Size, Dest, Src, Size);
|
||||
if (CTX->IsMemcpyAtomicTSOEnabled()) {
|
||||
_StoreMemGPRAutoTSO(Size, Dest, Src, Size);
|
||||
} else {
|
||||
_StoreMem(RegClass::GPR, Size, Src, Dest, Invalid(), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
// Offset the pointer
|
||||
Ref TailDest = LoadGPRRegister(X86State::REG_RDI);
|
||||
@@ -3255,7 +3244,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) {
|
||||
LogMan::Msg::EFmt("Can't handle adddress size");
|
||||
LogMan::Msg::EFmt("MOVSOp: Can't handle address size override (OP: 0x{:04X}, Flags: 0x{:08X})", Op->OP, Op->Flags);
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
}
|
||||
@@ -3298,50 +3287,67 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
Ref RSI = MakeSegmentAddress(X86State::REG_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
Ref RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
auto Src = _LoadMemGPRAutoTSO(Size, RSI, Size);
|
||||
if (CTX->IsMemcpyAtomicTSOEnabled()) {
|
||||
auto Src = _LoadMemGPRAutoTSO(Size, RSI, Size);
|
||||
|
||||
// Store to memory where RDI points
|
||||
_StoreMemGPRAutoTSO(Size, RDI, Src, Size);
|
||||
// Store to memory where RDI points
|
||||
_StoreMemGPRAutoTSO(Size, RDI, Src, Size);
|
||||
} else {
|
||||
auto Src = _LoadMem(RegClass::GPR, Size, RSI, Invalid(), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
_StoreMem(RegClass::GPR, Size, Src, RDI, Invalid(), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
auto PtrDir = LoadDir(IR::OpSizeToSize(Size));
|
||||
RSI = Add(OpSize::i64Bit, RSI, PtrDir);
|
||||
RDI = Add(OpSize::i64Bit, RDI, PtrDir);
|
||||
RSI = OffsetByDir(RSI, IR::OpSizeToSize(Size));
|
||||
RDI = OffsetByDir(RDI, IR::OpSizeToSize(Size));
|
||||
|
||||
StoreGPRRegister(X86State::REG_RSI, RSI);
|
||||
StoreGPRRegister(X86State::REG_RDI, RDI);
|
||||
}
|
||||
}
|
||||
|
||||
IR::OpSize OpDispatchBuilder::GetStringOpSize(X86Tables::DecodedOp Op) const {
|
||||
LOGMAN_THROW_A_FMT(Is64BitMode || !(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE), "Invalid modifier on 32bit address");
|
||||
return !Is64BitMode || (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) ? OpSize::i32Bit : OpSize::i64Bit;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) {
|
||||
LogMan::Msg::EFmt("Can't handle adddress size");
|
||||
if (!Is64BitMode && (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE)) {
|
||||
LogMan::Msg::EFmt("CMPSOp: Address size override (0x67) not supported in 32-bit mode (OP: 0x{:04X}).", Op->OP);
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
}
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
OpSize AddrSize = GetStringOpSize(Op);
|
||||
|
||||
bool Repeat = Op->Flags & (FEXCore::X86Tables::DecodeFlags::FLAG_REPNE_PREFIX | FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX);
|
||||
if (!Repeat) {
|
||||
// Default DS prefix
|
||||
Ref Dest_RSI = MakeSegmentAddress(X86State::REG_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
// Only ES prefix
|
||||
Ref Dest_RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
Ref Src_RSI = LoadGPRRegister(X86State::REG_RSI, AddrSize);
|
||||
Ref Src_RDI = LoadGPRRegister(X86State::REG_RDI, AddrSize);
|
||||
|
||||
Ref Dest_RSI = AppendSegmentOffset(Src_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
Ref Dest_RDI = AppendSegmentOffset(Src_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
auto Src1 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size);
|
||||
auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size);
|
||||
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Src2, Src1);
|
||||
|
||||
auto PtrDir = LoadDir(IR::OpSizeToSize(Size));
|
||||
Dest_RDI = OffsetByDir(Src_RDI, IR::OpSizeToSize(Size));
|
||||
if (Is64BitMode && AddrSize == OpSize::i32Bit) {
|
||||
Dest_RDI = _Bfe(OpSize::i64Bit, 32, 0, Dest_RDI);
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI);
|
||||
} else {
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI, AddrSize);
|
||||
}
|
||||
|
||||
// Offset the pointer
|
||||
Dest_RDI = Add(OpSize::i64Bit, Dest_RDI, PtrDir);
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI);
|
||||
|
||||
// Offset second pointer
|
||||
Dest_RSI = Add(OpSize::i64Bit, Dest_RSI, PtrDir);
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
Dest_RSI = OffsetByDir(Src_RSI, IR::OpSizeToSize(Size));
|
||||
if (Is64BitMode && AddrSize == OpSize::i32Bit) {
|
||||
Dest_RSI = _Bfe(OpSize::i64Bit, 32, 0, Dest_RSI);
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
} else {
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI, AddrSize);
|
||||
}
|
||||
} else {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
@@ -3357,7 +3363,7 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(BeforeLoop);
|
||||
StartNewBlock();
|
||||
|
||||
ForeachDirection([this, Op, Size, REPE](int32_t PtrDir) {
|
||||
ForeachDirection([this, Op, Size, AddrSize, REPE](int32_t PtrDir) {
|
||||
IRPair<IROp_CondJump> InnerJump;
|
||||
auto JumpIntoLoop = Jump();
|
||||
|
||||
@@ -3369,10 +3375,11 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
|
||||
// Working loop
|
||||
{
|
||||
// Default DS prefix
|
||||
Ref Dest_RSI = MakeSegmentAddress(X86State::REG_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
// Only ES prefix
|
||||
Ref Dest_RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
Ref Src_RSI = LoadGPRRegister(X86State::REG_RSI, AddrSize);
|
||||
Ref Src_RDI = LoadGPRRegister(X86State::REG_RDI, AddrSize);
|
||||
|
||||
Ref Dest_RSI = AppendSegmentOffset(Src_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
Ref Dest_RDI = AppendSegmentOffset(Src_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
auto Src1 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size);
|
||||
auto Src2 = _LoadMemGPR(Size, Dest_RSI, Size);
|
||||
@@ -3389,13 +3396,21 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
// Store the counter since we don't have phis
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
Dest_RDI = Add(OpSize::i64Bit, Dest_RDI, PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI);
|
||||
Dest_RDI = Add(AddrSize, Src_RDI, PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
if (Is64BitMode && AddrSize == OpSize::i32Bit) {
|
||||
Dest_RDI = _Bfe(OpSize::i64Bit, 32, 0, Dest_RDI);
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI);
|
||||
} else {
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI, AddrSize);
|
||||
}
|
||||
|
||||
// Offset second pointer
|
||||
Dest_RSI = Add(OpSize::i64Bit, Dest_RSI, PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
Dest_RSI = Add(AddrSize, Src_RSI, PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
if (Is64BitMode && AddrSize == OpSize::i32Bit) {
|
||||
Dest_RSI = _Bfe(OpSize::i64Bit, 32, 0, Dest_RSI);
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
} else {
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI, AddrSize);
|
||||
}
|
||||
|
||||
// If TailCounter != 0, compare sources.
|
||||
// If TailCounter == 0, set ZF iff that would break.
|
||||
@@ -3434,7 +3449,7 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) {
|
||||
LogMan::Msg::EFmt("Can't handle adddress size");
|
||||
LogMan::Msg::EFmt("LODSOp: Can't handle address size override (OP: 0x{:04X}, Flags: 0x{:08X})", Op->OP, Op->Flags);
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
}
|
||||
@@ -3516,31 +3531,37 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) {
|
||||
LogMan::Msg::EFmt("Can't handle adddress size");
|
||||
if (!Is64BitMode && (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE)) {
|
||||
LogMan::Msg::EFmt("SCASOp: Address size override (0x67) not supported in 32-bit mode (OP: 0x{:04X}).", Op->OP);
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
}
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
OpSize AddrSize = GetStringOpSize(Op);
|
||||
const bool Repeat = (Op->Flags & (FEXCore::X86Tables::DecodeFlags::FLAG_REPNE_PREFIX | FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX)) != 0;
|
||||
|
||||
if (!Repeat) {
|
||||
Ref Dest_RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
Ref Src_RDI = LoadGPRRegister(X86State::REG_RDI, AddrSize);
|
||||
Ref Dest_RDI = AppendSegmentOffset(Src_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size);
|
||||
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Src1, Src2);
|
||||
|
||||
// Offset the pointer
|
||||
Ref TailDest_RDI = LoadGPRRegister(X86State::REG_RDI);
|
||||
StoreGPRRegister(X86State::REG_RDI, OffsetByDir(TailDest_RDI, IR::OpSizeToSize(Size)));
|
||||
Ref TailDest_RDI = OffsetByDir(Src_RDI, IR::OpSizeToSize(Size));
|
||||
if (Is64BitMode && AddrSize == OpSize::i32Bit) {
|
||||
TailDest_RDI = _Bfe(OpSize::i64Bit, 32, 0, TailDest_RDI);
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI);
|
||||
} else {
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI, AddrSize);
|
||||
}
|
||||
} else {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size](int32_t Dir) {
|
||||
ForeachDirection([this, Op, Size, AddrSize](int32_t Dir) {
|
||||
bool REPE = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX;
|
||||
|
||||
auto JumpStart = Jump();
|
||||
@@ -3564,7 +3585,8 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
|
||||
// Working loop
|
||||
{
|
||||
Ref Dest_RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
Ref Src_RDI = LoadGPRRegister(X86State::REG_RDI, AddrSize);
|
||||
Ref Dest_RDI = AppendSegmentOffset(Src_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size);
|
||||
@@ -3575,7 +3597,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
Ref TailCounter = LoadGPRRegister(X86State::REG_RCX);
|
||||
Ref TailDest_RDI = LoadGPRRegister(X86State::REG_RDI);
|
||||
Ref Src_RDI_Tail = LoadGPRRegister(X86State::REG_RDI, AddrSize);
|
||||
|
||||
// Decrement counter
|
||||
TailCounter = Sub(OpSize::i64Bit, TailCounter, 1);
|
||||
@@ -3583,9 +3605,13 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
// Store the counter since we don't have phis
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RDI = Add(OpSize::i64Bit, TailDest_RDI, Dir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI);
|
||||
Ref TailDest_RDI = Add(AddrSize, Src_RDI_Tail, Dir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
if (Is64BitMode && AddrSize == OpSize::i32Bit) {
|
||||
TailDest_RDI = _Bfe(OpSize::i64Bit, 32, 0, TailDest_RDI);
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI);
|
||||
} else {
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI, AddrSize);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
InternalCondJump = CondJumpNZCV(REPE ? CondClass::EQ : CondClass::NEQ);
|
||||
|
||||
@@ -1330,7 +1330,6 @@ protected:
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ReducedPrecisionMode, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(StrictReducedPrecisionMode, X87STRICTREDUCEDPRECISION);
|
||||
|
||||
struct JumpTargetInfo {
|
||||
Ref BlockEntry;
|
||||
@@ -1656,6 +1655,9 @@ private:
|
||||
return IR::SizeToOpSize(GetSrcSize(Op));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
IR::OpSize GetStringOpSize(X86Tables::DecodedOp Op) const;
|
||||
|
||||
// Set flag tracking to prepare for an operation that directly writes NZCV.
|
||||
void HandleNZCVWrite() {
|
||||
CachedNZCV = nullptr;
|
||||
@@ -2528,7 +2530,7 @@ private:
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use ldp if possible, otherwise fallback on two loads.
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit && Size <= OpSize::i128Bit) {
|
||||
const auto B = SelectPairAddressMode(A, Size);
|
||||
return LoadMemPair(Class, Size, B.Base, B.Offset);
|
||||
}
|
||||
@@ -2567,7 +2569,7 @@ private:
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use stp if possible, otherwise fallback on two stores.
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit && Size <= OpSize::i128Bit) {
|
||||
const auto B = SelectPairAddressMode(A, Size);
|
||||
_StoreMemPair(Class, Size, Value1, Value2, B.Base, B.Offset);
|
||||
} else {
|
||||
|
||||
@@ -1956,7 +1956,7 @@ void OpDispatchBuilder::AVX128_VFMAImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx,
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx) {
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const OpSize ElementSize = Op->Flags & X86Tables::DecodeFlags::FLAG_OPTION_AVX_W ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
auto Dest = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, false).Low;
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false).Low;
|
||||
@@ -1964,13 +1964,13 @@ void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Sr
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false).Low;
|
||||
} else {
|
||||
Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags);
|
||||
}
|
||||
|
||||
Ref Sources[3] = {Dest, Src1, Src2};
|
||||
|
||||
DeriveOp(Result_Low, IROp,
|
||||
_VFMLAScalarInsert(OpSize::i128Bit, SrcSize, Dest, Sources[Src1Idx - 1], Sources[Src2Idx - 1], Sources[AddendIdx - 1]));
|
||||
_VFMLAScalarInsert(OpSize::i128Bit, ElementSize, Dest, Sources[Src1Idx - 1], Sources[Src2Idx - 1], Sources[AddendIdx - 1]));
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result_Low));
|
||||
}
|
||||
|
||||
|
||||
@@ -145,7 +145,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
|
||||
#ifndef _WIN32
|
||||
// FEX reserved instructions
|
||||
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
{0x3E, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
#endif
|
||||
};
|
||||
|
||||
@@ -552,7 +552,7 @@ void OpDispatchBuilder::AVXInsertScalarRound(OpcodeArgs) {
|
||||
const uint64_t Mode = Op->Src[2].Literal();
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
|
||||
Ref Result = InsertScalarRoundImpl(Op, DstSize, ElementSize, Op->Dest, Op->Src[0], Mode, true);
|
||||
Ref Result = InsertScalarRoundImpl(Op, DstSize, ElementSize, Op->Src[0], Op->Src[1], Mode, true);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize);
|
||||
}
|
||||
|
||||
|
||||
@@ -17,7 +17,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/FPState.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
@@ -61,15 +60,13 @@ void OpDispatchBuilder::SetX87Top(Ref Value) {
|
||||
|
||||
// Float LoaD operation with memory operand
|
||||
void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref ConvertedData = Data;
|
||||
// Convert to 80bit float
|
||||
if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
ConvertedData = _F80CVTTo(Data, ReadWidth);
|
||||
ConvertedData = _F80CVTTo(Data, Width);
|
||||
}
|
||||
_PushStack(ConvertedData, Data, ReadWidth, true);
|
||||
_PushStack(ConvertedData, Data, Width);
|
||||
}
|
||||
|
||||
// Float LoaD operation with memory operand
|
||||
@@ -81,7 +78,7 @@ void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
_PushStack(ConvertedData, Data, OpSize::i128Bit, true);
|
||||
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
@@ -93,7 +90,7 @@ void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant K) {
|
||||
// Update TOP
|
||||
Ref Data = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, K);
|
||||
_PushStack(Data, Data, OpSize::i128Bit, true);
|
||||
_PushStack(Data, Data, OpSize::f80Bit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
@@ -124,15 +121,16 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
Ref ConvertedData = _VLoadTwoGPRs(shifted, upper);
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Width == OpSize::i32Bit || Width == OpSize::i64Bit || Width == OpSize::f80Bit, "Invalid store width for FST");
|
||||
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::f80Bit;
|
||||
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
|
||||
|
||||
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, false, false, Width);
|
||||
_StoreStackMem(SourceSize, Width, A.Base, A.Index, OpSize::iInvalid, A.IndexType, A.IndexScale, /*Float=*/true);
|
||||
_StoreStackMem(SourceSize, Width, A.Base, A.Index, OpSize::iInvalid, A.IndexType, A.IndexScale);
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
@@ -878,8 +876,8 @@ void OpDispatchBuilder::X87FXTRACT(OpcodeArgs) {
|
||||
_PopStackDestroy();
|
||||
auto Exp = _F80XTRACT_EXP(Top);
|
||||
auto Sig = _F80XTRACT_SIG(Top);
|
||||
_PushStack(Exp, Exp, OpSize::f80Bit, true);
|
||||
_PushStack(Sig, Sig, OpSize::f80Bit, true);
|
||||
_PushStack(Exp, Invalid(), OpSize::iInvalid);
|
||||
_PushStack(Sig, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -59,7 +59,6 @@ void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
// F64 ops
|
||||
// Float load op with memory operand
|
||||
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
Ref ConvertedData = Data;
|
||||
@@ -68,7 +67,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
} else if (Width == OpSize::f80Bit) {
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, Data);
|
||||
}
|
||||
_PushStack(ConvertedData, Data, ReadWidth, true);
|
||||
_PushStack(ConvertedData, Data, Width);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
@@ -76,7 +75,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
|
||||
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
|
||||
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
@@ -88,7 +87,7 @@ void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FLDF64_Const(OpcodeArgs, uint64_t Num) {
|
||||
auto Data = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(Num));
|
||||
_PushStack(Data, Data, OpSize::i64Bit, true);
|
||||
_PushStack(Data, Data, OpSize::i64Bit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
@@ -100,7 +99,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
|
||||
}
|
||||
auto ConvertedData = _Float_FromGPR_S(OpSize::i64Bit, ReadWidth == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, Data);
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
@@ -397,7 +396,7 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
|
||||
Ref Exp = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, ExpZV, ExpNZV);
|
||||
|
||||
_PopStackDestroy();
|
||||
_PushStack(Exp, Exp, OpSize::i64Bit, true);
|
||||
_PushStack(Sig, Sig, OpSize::i64Bit, true);
|
||||
_PushStack(Exp, Invalid(), OpSize::iInvalid);
|
||||
_PushStack(Sig, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,87 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: glue|x86-guest-code
|
||||
desc: Guest-side assembly helpers used by the backends
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore {
|
||||
constexpr size_t CODE_SIZE = 0x1000;
|
||||
|
||||
X86GeneratedCode::X86GeneratedCode() {
|
||||
#ifdef _WIN32
|
||||
// No need to allocate anything in this config.
|
||||
#else
|
||||
|
||||
// Allocate a page for our emulated guest
|
||||
CodePtr = AllocateGuestCodeSpace(CODE_SIZE);
|
||||
|
||||
constexpr std::array<uint8_t, 2> SignalReturnCode = {
|
||||
0x0F, 0x37, // CALLBACKRET FEX Instruction
|
||||
};
|
||||
|
||||
CallbackReturn = reinterpret_cast<uint64_t>(CodePtr);
|
||||
|
||||
memcpy(reinterpret_cast<void*>(CallbackReturn), SignalReturnCode.data(), SignalReturnCode.size());
|
||||
|
||||
mprotect(CodePtr, CODE_SIZE, PROT_READ);
|
||||
#endif
|
||||
}
|
||||
|
||||
X86GeneratedCode::~X86GeneratedCode() {
|
||||
#ifndef _WIN32
|
||||
FEXCore::Allocator::VirtualFree(CodePtr, CODE_SIZE);
|
||||
#endif
|
||||
}
|
||||
|
||||
void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
|
||||
#ifndef _WIN32
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
|
||||
if (Is64BitMode()) {
|
||||
// 64bit mode can have its sigret handler anywhere
|
||||
return FEXCore::Allocator::VirtualAlloc(Size);
|
||||
}
|
||||
|
||||
// First 64bit page
|
||||
constexpr uintptr_t LOCATION_MAX = 0x1'0000'0000;
|
||||
|
||||
// 32bit mode
|
||||
// We need to have the sigret handler in the lower 32bits of memory space
|
||||
// Scan top down and try to allocate a location
|
||||
for (size_t Location = 0xFFFF'E000; Location != 0x0; Location -= 0x1000) {
|
||||
void* Ptr = ::mmap(reinterpret_cast<void*>(Location), Size, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
if (Ptr != MAP_FAILED && reinterpret_cast<uintptr_t>(Ptr) >= LOCATION_MAX) {
|
||||
// Failed to map in the lower 32bits
|
||||
// Try again
|
||||
// Can happen in the case that host kernel ignores MAP_FIXED_NOREPLACE
|
||||
::munmap(Ptr, Size);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Ptr != MAP_FAILED) {
|
||||
return Ptr;
|
||||
}
|
||||
}
|
||||
|
||||
// Can't do anything about this
|
||||
// Here's hoping the application doesn't use signals
|
||||
return MAP_FAILED;
|
||||
#else
|
||||
return nullptr;
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -1,25 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: glue|x86-guest-code
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore {
|
||||
class X86GeneratedCode final {
|
||||
public:
|
||||
X86GeneratedCode();
|
||||
~X86GeneratedCode();
|
||||
|
||||
uint64_t CallbackReturn {};
|
||||
|
||||
private:
|
||||
void* CodePtr {};
|
||||
void* AllocateGuestCodeSpace(size_t Size);
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -100,10 +100,11 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
{0x34, 1, X86InstInfo{"SYSENTER", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x35, 1, X86InstInfo{"SYSEXIT", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x36, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x37, 1, X86InstInfo{"GETSEC", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x38, 1, X86InstInfo{"", TYPE_0F38_TABLE, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x39, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x3A, 1, X86InstInfo{"", TYPE_0F3A_TABLE, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x3B, 4, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x3B, 3, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
|
||||
{0x40, 1, X86InstInfo{"CMOVO", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
{0x41, 1, X86InstInfo{"CMOVNO", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
@@ -299,7 +300,7 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
// FEX reserved instructions
|
||||
// Unused x86 encoding instruction.
|
||||
|
||||
{0x37, 1, X86InstInfo{"CALLBACKRET", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
{0x3E, 1, X86InstInfo{"CALLBACKRET", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
|
||||
// This was originally used by VIA to jump to its alternative instruction set. Used for OP_THUNK
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
|
||||
@@ -60,24 +60,7 @@ struct NodeID final {
|
||||
Value = 0;
|
||||
}
|
||||
|
||||
[[nodiscard]] friend constexpr bool operator==(NodeID, NodeID) noexcept = default;
|
||||
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator<(NodeID lhs, NodeID rhs) noexcept {
|
||||
return lhs.Value < rhs.Value;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator>(NodeID lhs, NodeID rhs) noexcept {
|
||||
return operator<(rhs, lhs);
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator<=(NodeID lhs, NodeID rhs) noexcept {
|
||||
return !operator>(lhs, rhs);
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator>=(NodeID lhs, NodeID rhs) noexcept {
|
||||
return !operator<(lhs, rhs);
|
||||
}
|
||||
[[nodiscard]] constexpr auto operator<=>(const NodeID&) const noexcept = default;
|
||||
|
||||
friend std::ostream& operator<<(std::ostream& out, NodeID ID) {
|
||||
out << ID.Value;
|
||||
|
||||
@@ -138,7 +138,6 @@
|
||||
"FenceType": "FenceType",
|
||||
"RegisterClass": "RegClass",
|
||||
"CondClass": "CondClass",
|
||||
"SyscallFlags": "FEXCore::IR::SyscallFlags",
|
||||
"SHA256Sum": "SHA256Sum",
|
||||
"MemOffsetType": "MemOffsetType",
|
||||
"BreakDefinition": "BreakDefinition",
|
||||
@@ -314,25 +313,13 @@
|
||||
"CallbackReturn": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Syscall GPR:$SyscallID, GPR:$Arg0, GPR:$Arg1, GPR:$Arg2, GPR:$Arg3, GPR:$Arg4, GPR:$Arg5, SyscallFlags:$Flags": {
|
||||
"GPR = Syscall GPR:$SyscallID, GPR:$Arg0, GPR:$Arg1, GPR:$Arg2, GPR:$Arg3, GPR:$Arg4, GPR:$Arg5": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Dispatches a guest syscall through to the SyscallHandler class"
|
||||
],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"GPR = InlineSyscall GPR:$Arg0, GPR:$Arg1, GPR:$Arg2, GPR:$Arg3, GPR:$Arg4, GPR:$Arg5, i32:$HostSyscallNumber, SyscallFlags:$Flags": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Dispatches a guest syscall directly to the host syscall interface,",
|
||||
"bypassing the SyscallHandler class used by Syscall.",
|
||||
"This has significantly less overhead than Syscall, which needs to save JIT state first.",
|
||||
"Can only be used for syscalls that match across architecture,",
|
||||
"such as gettid (matches on x86/x86-64/Arm64)."
|
||||
],
|
||||
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"Thunk GPR:$ArgPtr, SHA256Sum:$ThunkNameHash": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
@@ -578,8 +565,7 @@
|
||||
"Desc": ["Does a x86 TSO compatible load from memory. Offset must be Invalid()."
|
||||
],
|
||||
"Inline": ["", "Memtso"],
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StoreMemTSO RegisterClass:$Class, OpSize:#Size, SSA:$Value, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
@@ -587,8 +573,7 @@
|
||||
],
|
||||
"Inline": ["Zero", "", "Memtso"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"FPR = VLoadVectorMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Mask, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
@@ -819,16 +804,6 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"AtomicXor OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer xor",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AtomicSwap OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer swap"
|
||||
@@ -2833,17 +2808,13 @@
|
||||
"X87": true,
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"PushStack FPR:$X80Src, SSA:$OriginalValue, OpSize:$LoadSize, i1:$Float": {
|
||||
"PushStack FPR:$X80Src, FPR:$OriginalValue, OpSize:$LoadSize": {
|
||||
"Desc": [
|
||||
"Pushes the provided X80Src source on to the x87 stack.",
|
||||
"Tracks OriginalValue as the original value of X80Src.",
|
||||
"Tracks OriginalValue as the original value of X80Src. OriginalValue can be Invalid() in which case no tracking is done.",
|
||||
"Opsize is 128bit for F80 values, 64-bit for low precision.",
|
||||
"LoadSize the original load size, i.e. of size of OriginalValue.",
|
||||
"Float: 80-bit, 64-bit, 32-bit",
|
||||
"Int: 64-bit, 32-bit, 16-bit"
|
||||
],
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($OriginalValue) == RegClass::FPR || WalkFindRegClass($OriginalValue) == RegClass::GPR"
|
||||
"Float: 80-bit, 64-bit, 32-bit"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
@@ -2855,13 +2826,12 @@
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
},
|
||||
"StoreStackMem OpSize:$SourceSize, OpSize:$StoreSize, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale, i1:$Float": {
|
||||
"StoreStackMem OpSize:$SourceSize, OpSize:$StoreSize, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": [
|
||||
"Takes the top value off the x87 stack and stores it to memory.",
|
||||
"SourceSize is 128bit for F80 values, 64-bit for low precision.",
|
||||
"StoreSize is the store size for conversion:",
|
||||
"Float: 80-bit, 64-bit, or 32-bit",
|
||||
"Int: 64-bit, 32-bit, 16-bit"
|
||||
"Float: 80-bit, 64-bit, or 32-bit"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
|
||||
@@ -149,20 +149,6 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, RoundMode Arg)
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, SyscallFlags Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case SyscallFlags::DEFAULT: return "Default";
|
||||
case SyscallFlags::OPTIMIZETHROUGH: return "Optimize Through";
|
||||
case SyscallFlags::NOSYNCSTATEONENTRY: return "No Sync State on Entry";
|
||||
case SyscallFlags::NORETURN: return "No Return";
|
||||
case SyscallFlags::NOSIDEEFFECTS: return "No Side Effects";
|
||||
case SyscallFlags::NORETURNEDRESULT: return "No Returned Result";
|
||||
}
|
||||
return "<Unknown Syscall Flags>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorConstant Arg) {
|
||||
*out << [Arg] {
|
||||
// clang-format off
|
||||
@@ -367,8 +353,7 @@ void Dump(fextl::stringstream* out, const IRListView* IR) {
|
||||
auto BlockIROp = BlockHeader->C<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
AddIndent();
|
||||
*out << "(%" << IR->GetID(BlockNode) << ") "
|
||||
<< "CodeBlock ";
|
||||
*out << "(%" << IR->GetID(BlockNode) << ") " << "CodeBlock ";
|
||||
|
||||
*out << "%" << BlockIROp->Begin.ID() << ", ";
|
||||
*out << "%" << BlockIROp->Last.ID() << std::endl;
|
||||
|
||||
@@ -39,7 +39,7 @@ public:
|
||||
}
|
||||
|
||||
protected:
|
||||
PassManager* Manager;
|
||||
PassManager* Manager {};
|
||||
};
|
||||
|
||||
class PassManager final {
|
||||
|
||||
@@ -63,9 +63,9 @@ public:
|
||||
private:
|
||||
RegisterClassData Classes[IR::NumClasses];
|
||||
|
||||
IREmitter* IREmit;
|
||||
IRListView* IR;
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
IREmitter* IREmit {};
|
||||
IRListView* IR {};
|
||||
const FEXCore::CPUIDEmu* CPUID {};
|
||||
|
||||
// Map of nodes to their preferred register, to coalesce load/store reg.
|
||||
fextl::vector<PhysicalRegister> PreferredReg;
|
||||
@@ -83,7 +83,7 @@ private:
|
||||
fextl::vector<bool> Seen;
|
||||
|
||||
// SourcesNextUses is read backwards, this tracks the index
|
||||
int64_t SourceIndex;
|
||||
int64_t SourceIndex {};
|
||||
|
||||
bool Rematerializable(IROp_Header* IROp) {
|
||||
return IROp->Op == OP_CONSTANT;
|
||||
@@ -110,7 +110,7 @@ private:
|
||||
// block, so we don't need to size the block up-front.
|
||||
fextl::vector<uint32_t> NextUses;
|
||||
|
||||
bool AnySpilled;
|
||||
bool AnySpilled {};
|
||||
|
||||
bool IsValidArg(OrderedNodeWrapper Arg) {
|
||||
if (Arg.IsInvalid()) {
|
||||
|
||||
@@ -19,7 +19,7 @@ public:
|
||||
virtual void AddRegisters(RegClass Class, uint32_t RegisterCount) = 0;
|
||||
|
||||
// Number of GPRs usable for pairs at start of GPR set. Must be even.
|
||||
uint32_t PairRegs;
|
||||
uint32_t PairRegs {};
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -6,7 +6,6 @@
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/Profiler.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/Core/HostFeatures.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
@@ -66,7 +65,7 @@ public:
|
||||
int8_t TopOffset = 0;
|
||||
|
||||
FixedSizeStack()
|
||||
: buffer(FixedSizeStack::size, {StackSlot::UNUSED, T()}) {}
|
||||
: buffer(FixedSizeStack::size, {StackSlot::UNUSED, T::Invalid}) {}
|
||||
|
||||
void push(const T& Value) {
|
||||
rotate();
|
||||
@@ -85,7 +84,7 @@ public:
|
||||
}
|
||||
|
||||
void pop() {
|
||||
buffer.front() = {StackSlot::INVALID, T()};
|
||||
buffer.front() = {StackSlot::INVALID, T::Invalid};
|
||||
rotate(false);
|
||||
}
|
||||
|
||||
@@ -103,7 +102,7 @@ public:
|
||||
|
||||
void clear() {
|
||||
for (auto& Elem : buffer) {
|
||||
Elem = {StackSlot::UNUSED, T()};
|
||||
Elem = {StackSlot::UNUSED, T::Invalid};
|
||||
}
|
||||
TopOffset = 0;
|
||||
}
|
||||
@@ -158,9 +157,7 @@ public:
|
||||
: Features(Features)
|
||||
, GPROpSize(GPROpSize) {
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(StrictReducedPrecision, X87STRICTREDUCEDPRECISION);
|
||||
ReducedPrecisionMode = ReducedPrecision;
|
||||
StrictReducedPrecisionMode = StrictReducedPrecision;
|
||||
}
|
||||
void Run(IREmitter* Emit) override;
|
||||
|
||||
@@ -168,20 +165,13 @@ private:
|
||||
const FEXCore::HostFeatures& Features;
|
||||
const OpSize GPROpSize;
|
||||
bool ReducedPrecisionMode;
|
||||
bool StrictReducedPrecisionMode;
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
|
||||
// Helpers
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
Ref SilenceNaN(Ref Value);
|
||||
|
||||
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode, Ref AddrNode, Ref Offset, OpSize Align, MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale) {
|
||||
IREmit->_StoreMemFPR(OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
|
||||
@@ -196,7 +186,24 @@ private:
|
||||
IREmit->_StoreMemGPR(OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale);
|
||||
}
|
||||
|
||||
void Store80BitToMem(const IROp_StoreStackMem* Op, Ref StackNode, Ref AddrNode, Ref Offset, OpSize Align, MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale) {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
|
||||
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
|
||||
} else {
|
||||
F80SplitStore_Helper(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
LOGMAN_THROW_A_FMT(!ReducedPrecisionMode, "Full precision mode expected.");
|
||||
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
@@ -208,25 +215,12 @@ private:
|
||||
case OpSize::i32Bit:
|
||||
case OpSize::i64Bit: {
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
|
||||
StackNode = SilenceNaN(StackNode);
|
||||
}
|
||||
IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
case OpSize::f80Bit: {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
|
||||
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
|
||||
} else { // 80bit requires split-store
|
||||
F80SplitStore_Helper(Op, StackNode);
|
||||
}
|
||||
Store80BitToMem(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
@@ -236,16 +230,14 @@ private:
|
||||
// Performs a store to memory from a value the stack passed in as StackNode.
|
||||
// This is the version dealing with the reduced precision case.
|
||||
void StoreStackMem_Reduced_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
LOGMAN_THROW_A_FMT(ReducedPrecisionMode, "Reduced precision mode expected.");
|
||||
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
if ((!ReducedPrecisionMode || StrictReducedPrecisionMode) && Op->StoreSize != OpSize::f80Bit) {
|
||||
StackNode = SilenceNaN(StackNode);
|
||||
}
|
||||
|
||||
switch (Op->StoreSize) {
|
||||
case OpSize::i32Bit: {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
@@ -256,10 +248,9 @@ private:
|
||||
break;
|
||||
}
|
||||
|
||||
// 80bit requires split-store
|
||||
case OpSize::f80Bit: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
|
||||
F80SplitStore_Helper(Op, StackNode);
|
||||
Store80BitToMem(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
@@ -301,23 +292,24 @@ private:
|
||||
void Reset();
|
||||
|
||||
struct StackMemberInfo {
|
||||
StackMemberInfo() {}
|
||||
StackMemberInfo() = delete;
|
||||
StackMemberInfo(Ref Data)
|
||||
: StackDataNode(Data) {}
|
||||
StackMemberInfo(Ref Data, Ref Source, OpSize Size, bool Float)
|
||||
StackMemberInfo(Ref Data, Ref Source, OpSize Size)
|
||||
: StackDataNode(Data)
|
||||
, Source({Size, Source})
|
||||
, InterpretAsFloat(Float) {}
|
||||
, Source({Size, Source}) {}
|
||||
Ref StackDataNode {}; // Reference to the data in the Stack.
|
||||
// This is the source data node in the stack format, possibly converted to 64/80 bits.
|
||||
struct StackMemberData final {
|
||||
OpSize Size;
|
||||
Ref Node;
|
||||
};
|
||||
|
||||
static const StackMemberInfo Invalid;
|
||||
|
||||
// Tuple is only valid if we have information about the Source of the Stack Data Node.
|
||||
// In it's valid then OpSize is the original source size and Ref is the original source node.
|
||||
std::optional<StackMemberData> Source {};
|
||||
bool InterpretAsFloat {false}; // True if this is a floating point value, false if integer
|
||||
};
|
||||
|
||||
// StackData, TopCache need to be always properly set to ensure
|
||||
@@ -370,6 +362,8 @@ private:
|
||||
IRListView* IR = nullptr;
|
||||
};
|
||||
|
||||
inline const X87StackOptimization::StackMemberInfo X87StackOptimization::StackMemberInfo::Invalid {nullptr};
|
||||
|
||||
inline void X87StackOptimization::InvalidateCaches() {
|
||||
InvalidateCachedRegs();
|
||||
ConstantPool.fill(nullptr);
|
||||
@@ -499,15 +493,6 @@ inline Ref X87StackOptimization::RotateRight8(uint32_t V, Ref Amount) {
|
||||
return IREmit->_Lshr(OpSize::i32Bit, GetConstant(V | (V << 8)), Amount);
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::SilenceNaN(Ref Value) {
|
||||
Ref GPRValue = IREmit->_VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, Value, 0);
|
||||
|
||||
IREmit->_FCmp(OpSize::i64Bit, Value, Value); // Comparison with itself should set VS if nan
|
||||
Ref QuietNaNGPR = IREmit->_Or(OpSize::i64Bit, GPRValue, IREmit->_Constant(0x0008000000000000ULL));
|
||||
Ref SilencedValue = IREmit->_VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, QuietNaNGPR);
|
||||
return IREmit->_NZCVSelectV(OpSize::i64Bit, CondClass::VS, SilencedValue, Value);
|
||||
}
|
||||
|
||||
inline std::optional<X87StackOptimization::StackMemberInfo> X87StackOptimization::MigrateToSlowPath_IfInvalid(uint8_t Offset) {
|
||||
const auto& [Valid, StackMember] = StackData.top(Offset);
|
||||
MigrateToSlowPathIf(Valid != StackSlot::VALID);
|
||||
@@ -748,6 +733,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// The optimization should run per-block
|
||||
Reset();
|
||||
|
||||
IREmit->SetCurrentCodeBlock(BlockNode);
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (!LoweredX87(IROp->Op)) {
|
||||
continue;
|
||||
@@ -947,8 +933,13 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
StoreStackValueAtOffset_Slow(SourceNode);
|
||||
} else {
|
||||
auto* SourceNode = CurrentIR.GetNode(Op->X80Src);
|
||||
auto* OriginalNode = CurrentIR.GetNode(Op->OriginalValue);
|
||||
StackData.push(StackMemberInfo {SourceNode, OriginalNode, Op->LoadSize, Op->Float});
|
||||
if (Op->OriginalValue.IsInvalid()) {
|
||||
// No original value to track - just push the converted data
|
||||
StackData.push(StackMemberInfo {SourceNode});
|
||||
} else {
|
||||
auto* OriginalNode = CurrentIR.GetNode(Op->OriginalValue);
|
||||
StackData.push(StackMemberInfo {SourceNode, OriginalNode, Op->LoadSize});
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -1013,9 +1004,16 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// str w2, [x1]
|
||||
// or similar. As long as the source size and dest size are one and the same.
|
||||
// This will avoid any conversions between source and stack element size and conversion back.
|
||||
if (!SlowPath && Value->Source && Value->Source->Size == Op->StoreSize && Value->InterpretAsFloat) {
|
||||
const auto ClassType = Value->InterpretAsFloat ? RegClass::FPR : RegClass::GPR;
|
||||
IREmit->_StoreMem(ClassType, Op->StoreSize, Value->Source->Node, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
OpSize StoreSize = Op->StoreSize;
|
||||
LOGMAN_THROW_A_FMT(Op->StoreSize == OpSize::i32Bit || Op->StoreSize == OpSize::i64Bit || Op->StoreSize == OpSize::f80Bit,
|
||||
"Invalid store size in x87 store stack mem");
|
||||
if (!SlowPath && Value->Source && Value->Source->Size == StoreSize) {
|
||||
Ref SourceValue = Value->Source->Node;
|
||||
if (Op->StoreSize == OpSize::f80Bit) {
|
||||
Store80BitToMem(Op, SourceValue, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
} else {
|
||||
IREmit->_StoreMemFPR(StoreSize, SourceValue, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1055,11 +1053,26 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
case OP_F80STACKXCHANGE: {
|
||||
const auto* Op = IROp->C<IROp_F80StackXchange>();
|
||||
auto Offset = Op->SrcStack;
|
||||
Ref ValueTop = LoadStackValue();
|
||||
Ref ValueOffset = LoadStackValue(Offset);
|
||||
|
||||
StoreStackValue(ValueOffset);
|
||||
StoreStackValue(ValueTop, Offset);
|
||||
if (Offset == 0) {
|
||||
// No-op
|
||||
break;
|
||||
}
|
||||
|
||||
const auto [ValidTop, StackMemberTop] = StackData.top(0);
|
||||
const auto [ValidOffset, StackMemberOffset] = StackData.top(Offset);
|
||||
|
||||
if (ValidTop != StackSlot::VALID || ValidOffset != StackSlot::VALID) {
|
||||
// Slow path: do actual memory operations
|
||||
Ref ValueTop = LoadStackValue();
|
||||
Ref ValueOffset = LoadStackValue(Offset);
|
||||
StoreStackValue(ValueOffset);
|
||||
StoreStackValue(ValueTop, Offset);
|
||||
} else {
|
||||
// Fast path: swap complete StackMemberInfo preserving Source metadata
|
||||
StackData.setTop(StackMemberOffset, 0);
|
||||
StackData.setTop(StackMemberTop, Offset);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
@@ -53,10 +53,15 @@ void* FEX_mmap(void* addr, size_t length, int prot, int flags, int fd, off_t off
|
||||
}
|
||||
|
||||
if (flags & MAP_ANONYMOUS) {
|
||||
prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Result, length, "FEXMem");
|
||||
VirtualName("FEXMem", Result, length);
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
void VirtualName(const char* Name, void* Ptr, size_t Size) {
|
||||
prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, Name);
|
||||
}
|
||||
|
||||
int FEX_munmap(void* addr, size_t length) {
|
||||
int Result = Alloc64->Munmap(addr, length);
|
||||
|
||||
@@ -88,7 +93,7 @@ void* DisableSBRKAllocations() {
|
||||
// calls won't allocate any memory through that.
|
||||
void* AlignedBRK = reinterpret_cast<void*>(FEXCore::AlignUp(reinterpret_cast<uintptr_t>(StartingSBRK), FEXCore::Utils::FEX_PAGE_SIZE));
|
||||
void* AfterBRK =
|
||||
mmap(AlignedBRK, FEXCore::Utils::FEX_PAGE_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED_NOREPLACE | MAP_NORESERVE, -1, 0);
|
||||
::mmap(AlignedBRK, FEXCore::Utils::FEX_PAGE_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED_NOREPLACE | MAP_NORESERVE, -1, 0);
|
||||
if (AfterBRK == INVALID_PTR) {
|
||||
// Couldn't allocate the page after the aligned brk? This should never happen.
|
||||
// FEXCore::LogMan isn't configured yet so we just need to print the message.
|
||||
@@ -135,10 +140,7 @@ void ClearHooks() {
|
||||
FEXCore::Allocator::mmap = ::mmap;
|
||||
FEXCore::Allocator::munmap = ::munmap;
|
||||
|
||||
// XXX: This is currently a leak.
|
||||
// We can't work around this yet until static initializers that allocate memory are completely removed from our codebase
|
||||
// Luckily we only remove this on process shutdown, so the kernel will do the cleanup for us
|
||||
Alloc64.release();
|
||||
Alloc::OSAllocator::ReleaseAllocatorWorkaround(std::move(Alloc64));
|
||||
}
|
||||
#pragma GCC diagnostic pop
|
||||
|
||||
@@ -289,7 +291,7 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
--StackRegionIt;
|
||||
|
||||
auto Alloc =
|
||||
mmap(StackRegionIt->Ptr, StackRegionIt->Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
::mmap(StackRegionIt->Ptr, StackRegionIt->Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({},{:x}) failed", fmt::ptr(StackRegionIt->Ptr), StackRegionIt->Size);
|
||||
LogMan::Throw::AFmt(Alloc == StackRegionIt->Ptr, "mmap returned {} instead of {}", Alloc, fmt::ptr(StackRegionIt->Ptr));
|
||||
@@ -300,7 +302,7 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
|
||||
// Block remaining memory gaps
|
||||
for (auto RegionIt = Regions.begin(); RegionIt != Regions.end(); ++RegionIt) {
|
||||
auto Alloc = mmap(RegionIt->Ptr, RegionIt->Size, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
|
||||
auto Alloc = ::mmap(RegionIt->Ptr, RegionIt->Size, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({},{:x}) failed", fmt::ptr(RegionIt->Ptr), RegionIt->Size);
|
||||
LogMan::Throw::AFmt(Alloc == RegionIt->Ptr, "mmap returned {} instead of {}", Alloc, fmt::ptr(RegionIt->Ptr));
|
||||
|
||||
@@ -98,7 +98,7 @@ private:
|
||||
// Align UsedPages so it pads to the next page.
|
||||
// Necessary to take advantage of madvise zero page pooling.
|
||||
using FlexBitElementType = uint64_t;
|
||||
alignas(4096) FEXCore::FlexBitSet<FlexBitElementType> UsedPages;
|
||||
alignas(FEXCore::Utils::FEX_PAGE_SIZE) FEXCore::FlexBitSet<FlexBitElementType> UsedPages;
|
||||
|
||||
// This returns the size of the LiveVMARegion in addition to the flex set that tracks the used data
|
||||
// The LiveVMARegion lives at the start of the VMA region which means on initialization we need to set that
|
||||
@@ -140,7 +140,7 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(LiveVMARegion) == 4096, "Needs to be the size of a page");
|
||||
static_assert(sizeof(LiveVMARegion) == FEXCore::Utils::FEX_PAGE_SIZE, "Needs to be the size of a page");
|
||||
|
||||
static_assert(std::is_trivially_copyable<LiveVMARegion>::value, "Needs to be trivially copyable");
|
||||
static_assert(offsetof(LiveVMARegion, UsedPages) == sizeof(LiveVMARegion), "FlexBitSet needs to be at the end");
|
||||
@@ -168,6 +168,7 @@ private:
|
||||
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno,
|
||||
strerror(errno));
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData);
|
||||
LiveVMARegion* LiveRange = new (reinterpret_cast<void*>(ReservedRegion->Base)) LiveVMARegion();
|
||||
|
||||
// Copy over the reserved data
|
||||
@@ -206,7 +207,7 @@ OSAllocator_64Bit::LiveVMARegion* OSAllocator_64Bit::FindLiveRegionForAddress(ui
|
||||
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
|
||||
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
|
||||
|
||||
if (Addr >= RegionBegin && Addr < RegionEnd) {
|
||||
if (Addr >= RegionBegin && AddrEnd < RegionEnd) {
|
||||
LiveRegion = *it;
|
||||
// Leave our loop
|
||||
break;
|
||||
@@ -404,14 +405,18 @@ again:
|
||||
// Mark the pages as used
|
||||
uintptr_t RegionBegin = LiveRegion->SlabInfo->Base;
|
||||
uintptr_t MappedBegin = (AllocatedOffset - RegionBegin) >> FEXCore::Utils::FEX_PAGE_SHIFT;
|
||||
size_t PagesSet {};
|
||||
|
||||
for (size_t i = 0; i < NumberOfPages; ++i) {
|
||||
LiveRegion->UsedPages.Set(MappedBegin + i);
|
||||
PagesSet += LiveRegion->UsedPages.TestAndSet(MappedBegin + i) == false;
|
||||
}
|
||||
|
||||
// Change our last allocation region
|
||||
LiveRegion->LastPageAllocation = MappedBegin + NumberOfPages;
|
||||
LiveRegion->FreeSpace -= length;
|
||||
LiveRegion->FreeSpace -= PagesSet * FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
LOGMAN_THROW_A_FMT(LiveRegion->FreeSpace <= LiveRegion->SlabInfo->RegionSize,
|
||||
"Corrupt LiveRegion free space! 0x{:x} > 0x{:x}. After allocating 0x{:x} (0x{:x} overlapped)", LiveRegion->FreeSpace,
|
||||
LiveRegion->SlabInfo->RegionSize, length, PagesSet);
|
||||
}
|
||||
|
||||
if (!AllocatedOffset) {
|
||||
@@ -473,7 +478,7 @@ int OSAllocator_64Bit::Munmap(void* addr, size_t length) {
|
||||
::mmap(addr, length, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
|
||||
}
|
||||
|
||||
(*it)->FreeSpace += FreedPages * 4096;
|
||||
(*it)->FreeSpace += FreedPages * FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
|
||||
// Set the last allocated page to the minimum of last page allocation or this slab
|
||||
// This will let us more quickly fill holes
|
||||
@@ -505,6 +510,8 @@ void OSAllocator_64Bit::AllocateMemoryRegions(fextl::vector<FEXCore::Allocator::
|
||||
// This enables the kernel to use transparent large pages in the allocator which can reduce memory pressure
|
||||
::madvise(it.Ptr, ObjectAllocSize, MADV_HUGEPAGE);
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(it.Ptr), ObjectAllocSize);
|
||||
|
||||
ObjectAlloc = new (it.Ptr) Alloc::ForwardOnlyIntrusiveArenaAllocator(it.Ptr, ObjectAllocSize);
|
||||
ReservedRegions = ObjectAlloc->new_construct(ReservedRegions, ObjectAlloc);
|
||||
LiveRegions = ObjectAlloc->new_construct(LiveRegions, ObjectAlloc);
|
||||
@@ -602,6 +609,8 @@ fextl::unique_ptr<T> make_alloc_unique(FEXCore::Allocator::MemoryRegion& Base, A
|
||||
ERROR_AND_DIE_FMT("Couldn't allocate memory region");
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(ptr), MinPage);
|
||||
|
||||
// Remove the page from the base region.
|
||||
// Could be zero after this.
|
||||
Base.Size -= MinPage;
|
||||
|
||||
@@ -27,6 +27,11 @@ struct FlexBitSet final {
|
||||
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
|
||||
return Value;
|
||||
}
|
||||
bool TestAndSet(size_t Element) {
|
||||
bool Value = Get(Element);
|
||||
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
|
||||
return Value;
|
||||
}
|
||||
void Set(size_t Element) {
|
||||
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
@@ -70,12 +75,17 @@ struct FlexBitSet final {
|
||||
template<bool WantUnset>
|
||||
BitsetScanResults BackwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t MinimumElement) {
|
||||
bool FoundHole {};
|
||||
for (size_t CurrentPage = BeginningElement; CurrentPage >= (MinimumElement + ElementCount);) {
|
||||
|
||||
// Final element to iterate to.
|
||||
const size_t FinalElement = MinimumElement + ElementCount - 1;
|
||||
|
||||
for (size_t CurrentPage = BeginningElement; CurrentPage >= FinalElement;) {
|
||||
size_t Remaining = ElementCount;
|
||||
LOGMAN_THROW_A_FMT(Remaining <= CurrentPage, "Scanning less than available range");
|
||||
LOGMAN_THROW_A_FMT(CurrentPage <= BeginningElement && CurrentPage >= FinalElement, "BackwardScanForRange: Scanning less than "
|
||||
"available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentPage - Remaining) == WantUnset) {
|
||||
if (this->Get(CurrentPage - Remaining + 1) == WantUnset) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
@@ -92,7 +102,7 @@ struct FlexBitSet final {
|
||||
CurrentPage -= Remaining;
|
||||
} else {
|
||||
// We have a slab range
|
||||
return BitsetScanResults {CurrentPage - ElementCount, FoundHole};
|
||||
return BitsetScanResults {CurrentPage - ElementCount + 1, FoundHole};
|
||||
}
|
||||
}
|
||||
|
||||
@@ -108,11 +118,15 @@ struct FlexBitSet final {
|
||||
BitsetScanResults ForwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t ElementsInSet) {
|
||||
bool FoundHole {};
|
||||
|
||||
for (size_t CurrentElement = BeginningElement; CurrentElement < (ElementsInSet - ElementCount);) {
|
||||
// Final element to iterate to.
|
||||
const size_t FinalElement = ElementsInSet - ElementCount + 1;
|
||||
|
||||
for (size_t CurrentElement = BeginningElement; CurrentElement <= FinalElement;) {
|
||||
// If we have enough free space, check if we have enough free pages that are contiguous
|
||||
size_t Remaining = ElementCount;
|
||||
|
||||
LOGMAN_THROW_A_FMT((CurrentElement + Remaining - 1) < ElementsInSet, "Scanning less than available range");
|
||||
LOGMAN_THROW_A_FMT(CurrentElement >= BeginningElement && CurrentElement <= FinalElement, "ForwardScanForRange: Scanning less than "
|
||||
"available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentElement + Remaining - 1) == WantUnset) {
|
||||
|
||||
@@ -53,4 +53,12 @@ public:
|
||||
namespace Alloc::OSAllocator {
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocator();
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocatorWithRegions(fextl::vector<FEXCore::Allocator::MemoryRegion>& Regions);
|
||||
static inline void ReleaseAllocatorWorkaround(fextl::unique_ptr<Alloc::HostAllocator> Allocator) {
|
||||
// XXX: This is currently a leak.
|
||||
// We can't work around this yet until static initializers that allocate memory are completely removed from our codebase
|
||||
// The allocator is also intrusively allocated, so the unique_ptr tries to double free the HostAllocator object.
|
||||
// Luckily we only remove this on process shutdown, so the kernel will do the cleanup for us
|
||||
Allocator.release();
|
||||
}
|
||||
|
||||
} // namespace Alloc::OSAllocator
|
||||
@@ -144,33 +144,33 @@ static __uint128_t LoadAcquire128(uint64_t Addr) {
|
||||
}
|
||||
|
||||
static uint64_t LoadAcquire64(uint64_t Addr) {
|
||||
std::atomic<uint64_t>* Atom = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
return Atom->load(std::memory_order_acquire);
|
||||
auto Atom = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
return Atom.load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static bool StoreCAS64(uint64_t& Expected, uint64_t Val, uint64_t Addr) {
|
||||
std::atomic<uint64_t>* Atom = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
auto Atom = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
return Atom.compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
static uint32_t LoadAcquire32(uint64_t Addr) {
|
||||
std::atomic<uint32_t>* Atom = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
return Atom->load(std::memory_order_acquire);
|
||||
auto Atom = std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(Addr));
|
||||
return Atom.load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static bool StoreCAS32(uint32_t& Expected, uint32_t Val, uint64_t Addr) {
|
||||
std::atomic<uint32_t>* Atom = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
auto Atom = std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(Addr));
|
||||
return Atom.compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
static uint8_t LoadAcquire8(uint64_t Addr) {
|
||||
std::atomic<uint8_t>* Atom = reinterpret_cast<std::atomic<uint8_t>*>(Addr);
|
||||
return Atom->load(std::memory_order_acquire);
|
||||
auto Atom = std::atomic_ref<uint8_t>(*reinterpret_cast<uint8_t*>(Addr));
|
||||
return Atom.load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static bool StoreCAS8(uint8_t& Expected, uint8_t Val, uint64_t Addr) {
|
||||
std::atomic<uint8_t>* Atom = reinterpret_cast<std::atomic<uint8_t>*>(Addr);
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
auto Atom = std::atomic_ref<uint8_t>(*reinterpret_cast<uint8_t*>(Addr));
|
||||
return Atom.compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
static uint16_t DoLoad16(uint64_t Addr) {
|
||||
@@ -211,8 +211,8 @@ static uint16_t DoLoad16(uint64_t Addr) {
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
|
||||
std::atomic<uint64_t>* Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
uint64_t TmpResult = Atomic->load();
|
||||
auto Atomic = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
uint64_t TmpResult = Atomic.load();
|
||||
|
||||
// Zexts the result
|
||||
uint16_t Result = TmpResult >> (Alignment * 8);
|
||||
@@ -224,8 +224,8 @@ static uint16_t DoLoad16(uint64_t Addr) {
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
|
||||
std::atomic<uint32_t>* Atomic = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
uint32_t TmpResult = Atomic->load();
|
||||
auto Atomic = std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(Addr));
|
||||
uint32_t TmpResult = Atomic.load();
|
||||
|
||||
// Zexts the result
|
||||
uint16_t Result = TmpResult >> (Alignment * 8);
|
||||
@@ -272,8 +272,8 @@ static uint32_t DoLoad32(uint64_t Addr) {
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
|
||||
std::atomic<uint64_t>* Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
uint64_t TmpResult = Atomic->load();
|
||||
auto Atomic = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
uint64_t TmpResult = Atomic.load();
|
||||
|
||||
return TmpResult >> (Alignment * 8);
|
||||
}
|
||||
@@ -465,7 +465,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t>* Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
auto Atomic128 = std::atomic_ref<__uint128_t>(*reinterpret_cast<__uint128_t*>(Addr));
|
||||
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
@@ -480,7 +480,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
TmpExpected = Atomic128.load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
@@ -491,7 +491,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic128.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
@@ -617,36 +617,6 @@ static uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uin
|
||||
}
|
||||
}
|
||||
|
||||
static bool HandleAtomicVectorStore(uint32_t Instr, uintptr_t ProgramCounter) {
|
||||
uint32_t* PC = (uint32_t*)ProgramCounter;
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
|
||||
if (Size == 1) {
|
||||
// 64-bit pair happens on paranoid vector stores
|
||||
// [0] ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
// [1] stlxp(TMP3, TMP1, TMP2, MemSrc); // <- Can also hit SIGBUS
|
||||
// [2] cbnz(TMP3, &B); // < Overwritten with DMB
|
||||
if (DataReg == 31) {
|
||||
uint32_t NextInstr = PC[1];
|
||||
uint32_t AddrReg = (NextInstr >> 5) & 0x1F;
|
||||
DataReg = NextInstr & 0x1F;
|
||||
uint32_t DataReg2 = (NextInstr >> 10) & 0x1F;
|
||||
uint32_t STP = (0b10 << 30) | (0b101001000000000 << 15) | (DataReg2 << 10) | (AddrReg << 5) | DataReg;
|
||||
|
||||
PC[0] = DMB;
|
||||
PC[1] = STP;
|
||||
PC[2] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ClearICache(&PC[0], 12);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
using CASExpectedFn = T (*)(T Src, T Expected);
|
||||
template<typename T>
|
||||
@@ -740,7 +710,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t>* Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
auto Atomic128 = std::atomic_ref<__uint128_t>(*reinterpret_cast<__uint128_t*>(Addr));
|
||||
|
||||
__uint128_t Mask = 0xFFFF;
|
||||
Mask <<= Alignment * 8;
|
||||
@@ -749,7 +719,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
__uint128_t TmpDesired {};
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
TmpExpected = Atomic128.load();
|
||||
|
||||
__uint128_t Desired = DesiredFunction(TmpExpected >> (Alignment * 8), DesiredSrc);
|
||||
Desired <<= Alignment * 8;
|
||||
@@ -766,7 +736,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic128.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return Expected >> (Alignment * 8);
|
||||
@@ -810,9 +780,9 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
uint64_t TmpExpected {};
|
||||
uint64_t TmpDesired {};
|
||||
|
||||
std::atomic<uint64_t>* Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
auto Atomic = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
while (1) {
|
||||
TmpExpected = Atomic->load();
|
||||
TmpExpected = Atomic.load();
|
||||
|
||||
uint64_t Desired = DesiredFunction(TmpExpected >> (Alignment * 8), DesiredSrc);
|
||||
Desired <<= Alignment * 8;
|
||||
@@ -829,7 +799,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return Expected >> (Alignment * 8);
|
||||
@@ -873,9 +843,9 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
uint32_t TmpExpected {};
|
||||
uint32_t TmpDesired {};
|
||||
|
||||
std::atomic<uint32_t>* Atomic = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
auto Atomic = std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(Addr));
|
||||
while (1) {
|
||||
TmpExpected = Atomic->load();
|
||||
TmpExpected = Atomic.load();
|
||||
|
||||
|
||||
uint32_t Desired = DesiredFunction(TmpExpected >> (Alignment * 8), DesiredSrc);
|
||||
@@ -893,7 +863,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return Expected >> (Alignment * 8);
|
||||
@@ -1040,7 +1010,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t>* Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
auto Atomic128 = std::atomic_ref<__uint128_t>(*reinterpret_cast<__uint128_t*>(Addr));
|
||||
|
||||
__uint128_t Mask = ~0U;
|
||||
Mask <<= Alignment * 8;
|
||||
@@ -1049,7 +1019,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
__uint128_t TmpDesired {};
|
||||
|
||||
while (1) {
|
||||
__uint128_t TmpActual = Atomic128->load();
|
||||
__uint128_t TmpActual = Atomic128.load();
|
||||
|
||||
__uint128_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc);
|
||||
__uint128_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc);
|
||||
@@ -1064,7 +1034,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired << (Alignment * 8);
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic128.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Stored successfully
|
||||
return Expected;
|
||||
@@ -1108,9 +1078,9 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
uint64_t TmpExpected {};
|
||||
uint64_t TmpDesired {};
|
||||
|
||||
std::atomic<uint64_t>* Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
auto Atomic = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
while (1) {
|
||||
uint64_t TmpActual = Atomic->load();
|
||||
uint64_t TmpActual = Atomic.load();
|
||||
|
||||
uint64_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc);
|
||||
uint64_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc);
|
||||
@@ -1125,7 +1095,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired << (Alignment * 8);
|
||||
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Stored successfully
|
||||
return Expected;
|
||||
@@ -1270,7 +1240,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
std::atomic<__uint128_t>* Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
auto Atomic128 = std::atomic_ref<__uint128_t>(*reinterpret_cast<__uint128_t*>(Addr));
|
||||
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
@@ -1279,7 +1249,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
__uint128_t TmpDesired {};
|
||||
|
||||
while (1) {
|
||||
__uint128_t TmpActual = Atomic128->load();
|
||||
__uint128_t TmpActual = Atomic128.load();
|
||||
|
||||
__uint128_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc);
|
||||
__uint128_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc);
|
||||
@@ -1294,7 +1264,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired << (Alignment * 8);
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic128.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Stored successfully
|
||||
return Expected;
|
||||
@@ -1326,9 +1296,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
}
|
||||
}
|
||||
|
||||
static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg, uint32_t* StrictSplitLockMutex) {
|
||||
uint64_t Addr = GPRs[AddressReg];
|
||||
|
||||
static std::optional<uint64_t> DoCAS(uint32_t Size, uint64_t Desired, uint64_t Expected, uint64_t Addr, uint32_t* StrictSplitLockMutex) {
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
// It isn't even guaranteed to work on x86
|
||||
// Intel will do a "split lock" which locks the full bus
|
||||
@@ -1341,7 +1309,7 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
// Only need to handle 16, 32, 64
|
||||
if (Size == 2) {
|
||||
auto Res = DoCAS16<false>(
|
||||
GPRs[DesiredReg], GPRs[ExpectedReg], Addr,
|
||||
Desired, Expected, Addr,
|
||||
[](uint16_t, uint16_t Expected) -> uint16_t {
|
||||
// Expected is just Expected
|
||||
return Expected;
|
||||
@@ -1351,16 +1319,10 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
return Desired;
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
return Res;
|
||||
} else if (Size == 4) {
|
||||
auto Res = DoCAS32<false>(
|
||||
GPRs[DesiredReg], GPRs[ExpectedReg], Addr,
|
||||
Desired, Expected, Addr,
|
||||
[](uint32_t, uint32_t Expected) -> uint32_t {
|
||||
// Expected is just Expected
|
||||
return Expected;
|
||||
@@ -1370,16 +1332,10 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
return Desired;
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
return Res;
|
||||
} else if (Size == 8) {
|
||||
auto Res = DoCAS64<false>(
|
||||
GPRs[DesiredReg], GPRs[ExpectedReg], Addr,
|
||||
Desired, Expected, Addr,
|
||||
[](uint64_t, uint64_t Expected) -> uint64_t {
|
||||
// Expected is just Expected
|
||||
return Expected;
|
||||
@@ -1389,16 +1345,24 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
return Desired;
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
return Res;
|
||||
}
|
||||
|
||||
return false;
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg, uint32_t* StrictSplitLockMutex) {
|
||||
std::optional<uint64_t> Res = DoCAS(Size, GPRs[DesiredReg], GPRs[ExpectedReg], GPRs[AddressReg], StrictSplitLockMutex);
|
||||
if (!Res.has_value()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = *Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool HandleCASAL(uint64_t* GPRs, uint32_t Instr, uint32_t* StrictSplitLockMutex) {
|
||||
@@ -1560,38 +1524,43 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs, uint32_t* StrictSp
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool HandleAtomicLoad(uint32_t Instr, uint64_t* GPRs, int64_t Offset) {
|
||||
static bool HandleAtomicLoad(uint32_t Instr, uint64_t* GPRs, int64_t Offset, Core::UnalignedExclusiveStore* Store = nullptr) {
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t ResultReg = Instr & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = GPRs[AddressReg] + Offset;
|
||||
uint64_t Res;
|
||||
|
||||
if (Size == 2) {
|
||||
auto Res = DoLoad16(Addr);
|
||||
Res = DoLoad16(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
} else if (Size == 4) {
|
||||
auto Res = DoLoad32(Addr);
|
||||
Res = DoLoad32(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
} else if (Size == 8) {
|
||||
auto Res = DoLoad64(Addr);
|
||||
Res = DoLoad64(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
|
||||
return false;
|
||||
if (Store) {
|
||||
Store->Addr = Addr;
|
||||
Store->Store = Res;
|
||||
Store->Size = Size;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset, uint32_t* StrictSplitLockMutex) {
|
||||
@@ -1952,8 +1921,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
std::optional<int32_t>
|
||||
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType, uintptr_t ProgramCounter, uint64_t* GPRs) {
|
||||
std::optional<int32_t> HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType,
|
||||
uintptr_t ProgramCounter, uint64_t* GPRs, bool IsJIT) {
|
||||
#ifdef _M_ARM_64
|
||||
constexpr bool is_arm64 = true;
|
||||
#else
|
||||
@@ -1977,8 +1946,7 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
uint32_t* StrictSplitLockMutex {CTX->Config.StrictInProcessSplitLocks ? &CTX->StrictSplitLockMutex : nullptr};
|
||||
|
||||
// ParanoidTSO path doesn't modify any code.
|
||||
if (HandleType == UnalignedHandlerType::Paranoid) [[unlikely]] {
|
||||
if (!IsJIT) [[unlikely]] {
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
@@ -2016,7 +1984,29 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::LDAXR_MASK) == ArchHelpers::Arm64::LDAXR_INST) { // LDAXR*
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0, &Thread->ExclusiveStore)) {
|
||||
return 4;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::STLXR_MASK) == ArchHelpers::Arm64::STLXR_INST) { // STLXR*
|
||||
uint32_t StatusReg = Instr << 11 >> 27;
|
||||
// // Emulate exclusive store by validating the address and value against the last unaligned LDAXR*.
|
||||
if (GPRs[AddrReg] != Thread->ExclusiveStore.Addr || Size > Thread->ExclusiveStore.Size) {
|
||||
if (StatusReg != 31) {
|
||||
GPRs[StatusReg] = 1;
|
||||
}
|
||||
return 4;
|
||||
}
|
||||
if (std::optional<uint64_t> Prev =
|
||||
DoCAS(Size, DataReg == 31 ? 0 : GPRs[DataReg], Thread->ExclusiveStore.Store, GPRs[AddrReg], StrictSplitLockMutex)) {
|
||||
if (StatusReg != 31) {
|
||||
GPRs[StatusReg] = !!memcmp(&Thread->ExclusiveStore.Store, &*Prev, Size);
|
||||
}
|
||||
Thread->ExclusiveStore.Size = 0;
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
@@ -2068,6 +2058,9 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
return BytesToSkip;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2131,19 +2124,6 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
ClearICache(&PC[-1], 8);
|
||||
// Back up one instruction and have another go
|
||||
return -4;
|
||||
} else if ((Instr & ArchHelpers::Arm64::LDAXP_MASK) == ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
/// This is handling the case of paranoid ARMv8.0-a atomic stores.
|
||||
/// This backpatches the ldaxp+stlxp+cbnz if the previous `HandleCASPAL_ARMv8` didn't handle the case.
|
||||
if (ArchHelpers::Arm64::HandleAtomicVectorStore(Instr, ProgramCounter)) {
|
||||
return 0;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::STLXP_MASK) == ArchHelpers::Arm64::STLXP_INST) { // STLXP
|
||||
// Should not trigger - middle of an LDAXP/STAXP pair.
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
// Check if another thread backpatched this instruction before this thread got here
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::UncheckedLongJump {
|
||||
#if defined(_M_ARM_64)
|
||||
[[nodiscard]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
|
||||
__asm volatile(R"(
|
||||
// x0 contains the jumpbuffer
|
||||
stp x19, x20, [x0, #( 0 * 8)];
|
||||
stp x21, x22, [x0, #( 2 * 8)];
|
||||
stp x23, x24, [x0, #( 4 * 8)];
|
||||
stp x25, x26, [x0, #( 6 * 8)];
|
||||
stp x27, x28, [x0, #( 8 * 8)];
|
||||
stp x29, x30, [x0, #(10 * 8)];
|
||||
|
||||
// FPRs
|
||||
stp d8, d9, [x0, #(12 * 8)];
|
||||
stp d10, d11, [x0, #(14 * 8)];
|
||||
stp d12, d13, [x0, #(16 * 8)];
|
||||
stp d14, d15, [x0, #(18 * 8)];
|
||||
|
||||
// Move SP in to a temporary to store.
|
||||
mov x1, sp;
|
||||
str x1, [x0, #(20 * 8)];
|
||||
|
||||
// Return zero to signify this is the SetJump.
|
||||
mov x0, #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
[[noreturn]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(const JumpBuf& Buffer, uint64_t Value) {
|
||||
__asm volatile(R"(
|
||||
// x0 contains the jumpbuffer
|
||||
ldp x19, x20, [x0, #( 0 * 8)];
|
||||
ldp x21, x22, [x0, #( 2 * 8)];
|
||||
ldp x23, x24, [x0, #( 4 * 8)];
|
||||
ldp x25, x26, [x0, #( 6 * 8)];
|
||||
ldp x27, x28, [x0, #( 8 * 8)];
|
||||
ldp x29, x30, [x0, #(10 * 8)];
|
||||
|
||||
// FPRs
|
||||
ldp d8, d9, [x0, #(12 * 8)];
|
||||
ldp d10, d11, [x0, #(14 * 8)];
|
||||
ldp d12, d13, [x0, #(16 * 8)];
|
||||
ldp d14, d15, [x0, #(18 * 8)];
|
||||
|
||||
// Load SP in to temporary then move
|
||||
ldr x0, [x0, #(20 * 8)];
|
||||
mov sp, x0;
|
||||
|
||||
// Move value in to result register
|
||||
mov x0, x1;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void ManuallyLoadJumpBuf(const JumpBuf& Buffer, uint64_t Value, uint64_t* GPRs, __uint128_t* FPRs, uint64_t* PC) {
|
||||
// First 12 values are registers [x19,x30].
|
||||
memcpy(&GPRs[19], &Buffer.Registers[0], sizeof(uint64_t) * 12);
|
||||
|
||||
// Next 8 values are [D8,D15]
|
||||
// Retain upper 64-bits of the register, only modifying lower 64-bits.
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
memcpy(&FPRs[8 + i], &Buffer.Registers[12 + i], sizeof(uint64_t));
|
||||
}
|
||||
|
||||
// Last value is stack pointer
|
||||
memcpy(&GPRs[31], &Buffer.Registers[20], sizeof(uint64_t));
|
||||
|
||||
// Load the expected value in to X0
|
||||
GPRs[0] = Value;
|
||||
|
||||
// Load the PC with the current LR.
|
||||
*PC = GPRs[30];
|
||||
}
|
||||
|
||||
#else
|
||||
[[nodiscard]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
|
||||
__asm volatile(R"(
|
||||
.intel_syntax noprefix;
|
||||
// rdi contains the jumpbuffer
|
||||
mov [rdi + (0 * 8)], rbx;
|
||||
mov [rdi + (1 * 8)], rsp;
|
||||
mov [rdi + (2 * 8)], rbp;
|
||||
mov [rdi + (3 * 8)], r12;
|
||||
mov [rdi + (4 * 8)], r13;
|
||||
mov [rdi + (5 * 8)], r14;
|
||||
mov [rdi + (6 * 8)], r15;
|
||||
|
||||
// Return address is on the stack, load it and store
|
||||
mov rsi, [rsp];
|
||||
mov [rdi + (7 * 8)], rsi;
|
||||
|
||||
// Return zero to signify this is the SetJump.
|
||||
mov rax, 0;
|
||||
ret;
|
||||
|
||||
.att_syntax prefix;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
[[noreturn]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(const JumpBuf& Buffer, uint64_t Value) {
|
||||
__asm volatile(R"(
|
||||
.intel_syntax noprefix;
|
||||
// rdi contains the jumpbuffer
|
||||
mov rbx, [rdi + (0 * 8)];
|
||||
mov rsp, [rdi + (1 * 8)];
|
||||
mov rbp, [rdi + (2 * 8)];
|
||||
mov r12, [rdi + (3 * 8)];
|
||||
mov r13, [rdi + (4 * 8)];
|
||||
mov r14, [rdi + (5 * 8)];
|
||||
mov r15, [rdi + (6 * 8)];
|
||||
|
||||
// Move value in to result register
|
||||
mov rax, rsi;
|
||||
|
||||
// Pop the dead return address off the stack
|
||||
pop rsi;
|
||||
|
||||
// Load the original return address from the jumpbuffer
|
||||
mov rsi, [rdi + (7 * 8)];
|
||||
|
||||
// Return using a jump
|
||||
jmp rsi;
|
||||
|
||||
.att_syntax prefix;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void ManuallyLoadJumpBuf(JumpBuf& Buffer, uint64_t Value, uint64_t* GPRs, __uint128_t* FPRs, uint64_t* PC) {
|
||||
LOGMAN_MSG_A_FMT("This is unimplemented on x86-64");
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::UncheckedLongJump
|
||||
@@ -9,6 +9,7 @@
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
@@ -70,6 +71,10 @@ static std::array<const char*, 2> TraceFSDirectories {
|
||||
};
|
||||
|
||||
void Init() {
|
||||
FEX_CONFIG_OPT(EnableGpuvisProfiling, ENABLEGPUVISPROFILING);
|
||||
if (!EnableGpuvisProfiling()) {
|
||||
return;
|
||||
}
|
||||
for (auto Path : TraceFSDirectories) {
|
||||
#ifdef _WIN32
|
||||
constexpr auto flags = O_WRONLY;
|
||||
|
||||
@@ -1,9 +1,14 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <mutex>
|
||||
#include <type_traits>
|
||||
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
namespace FEXCore::Utils::SpinWaitLock {
|
||||
/**
|
||||
* @brief This provides routines to implement implement an "efficient spin-loop" using ARM's WFE and exclusive monitor interfaces.
|
||||
@@ -123,35 +128,26 @@ static inline uint64_t WFELoadAtomic(uint64_t* Futex) {
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<typename T, typename TT = T>
|
||||
static inline void Wait(T* Futex, TT ExpectedValue) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
T Result = AtomicFutex->load();
|
||||
template<typename Pred, typename T>
|
||||
static inline void WaitPred(T* Futex, T ComparisonValue) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
return;
|
||||
}
|
||||
|
||||
do {
|
||||
while (!Pred {}(Result, ComparisonValue)) {
|
||||
Result = LoadExclusive(Futex);
|
||||
if (Result == ExpectedValue) {
|
||||
if (Pred {}(Result, ComparisonValue)) {
|
||||
return;
|
||||
}
|
||||
Result = WFELoadAtomic(Futex);
|
||||
} while (Result != ExpectedValue);
|
||||
}
|
||||
|
||||
template void Wait<uint8_t>(uint8_t*, uint8_t);
|
||||
template void Wait<uint16_t>(uint16_t*, uint16_t);
|
||||
template void Wait<uint32_t>(uint32_t*, uint32_t);
|
||||
template void Wait<uint64_t>(uint64_t*, uint64_t);
|
||||
Result = WFELoadAtomic(Futex);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename T, typename TT>
|
||||
static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanoseconds& Timeout) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
|
||||
T Result = AtomicFutex->load();
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
@@ -184,27 +180,43 @@ template bool Wait<uint16_t>(uint16_t*, uint16_t, const std::chrono::nanoseconds
|
||||
template bool Wait<uint32_t>(uint32_t*, uint32_t, const std::chrono::nanoseconds&);
|
||||
template bool Wait<uint64_t>(uint64_t*, uint64_t, const std::chrono::nanoseconds&);
|
||||
|
||||
#else
|
||||
template<typename T, typename TT>
|
||||
static inline void Wait(T* Futex, TT ExpectedValue) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
T Result = AtomicFutex->load();
|
||||
template<typename T>
|
||||
static inline T OneShotWFEBitComparison(T* Futex, T Mask, T Comp) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
return;
|
||||
if ((Result & Mask) == Comp) {
|
||||
return Result;
|
||||
}
|
||||
|
||||
do {
|
||||
Result = AtomicFutex->load();
|
||||
} while (Result != ExpectedValue);
|
||||
Result = LoadExclusive(Futex);
|
||||
if ((Result & Mask) == Comp) {
|
||||
return Result;
|
||||
}
|
||||
|
||||
// Waits for write and returns result.
|
||||
Result = WFELoadAtomic(Futex);
|
||||
return Result;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
template<typename Pred, typename T>
|
||||
static inline void WaitPred(T* Futex, T ComparisonValue) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
while (!Pred {}(Result, ComparisonValue)) {
|
||||
Result = AtomicFutex.load();
|
||||
}
|
||||
}
|
||||
|
||||
template<typename T, typename TT>
|
||||
static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanoseconds& Timeout) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
|
||||
T Result = AtomicFutex->load();
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
@@ -214,7 +226,7 @@ static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanosecon
|
||||
const auto Begin = std::chrono::high_resolution_clock::now();
|
||||
|
||||
do {
|
||||
Result = AtomicFutex->load();
|
||||
Result = AtomicFutex.load();
|
||||
|
||||
const auto CurrentCycleCounter = std::chrono::high_resolution_clock::now();
|
||||
if ((CurrentCycleCounter - Begin) >= Timeout) {
|
||||
@@ -228,14 +240,24 @@ static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanosecon
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename T, typename TT = T>
|
||||
static inline void Wait(T* Futex, TT ExpectedValue) {
|
||||
WaitPred<std::equal_to<>, T>(Futex, ExpectedValue);
|
||||
}
|
||||
|
||||
template void Wait<uint8_t>(uint8_t*, uint8_t);
|
||||
template void Wait<uint16_t>(uint16_t*, uint16_t);
|
||||
template void Wait<uint32_t>(uint32_t*, uint32_t);
|
||||
template void Wait<uint64_t>(uint64_t*, uint64_t);
|
||||
|
||||
template<typename T>
|
||||
static inline void lock(T* Futex) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Expected {};
|
||||
T Desired {1};
|
||||
|
||||
// Try to CAS immediately.
|
||||
if (AtomicFutex->compare_exchange_strong(Expected, Desired)) {
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired)) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -243,17 +265,17 @@ static inline void lock(T* Futex) {
|
||||
// Wait until the futex is unlocked.
|
||||
Wait(Futex, 0);
|
||||
Expected = 0;
|
||||
} while (!AtomicFutex->compare_exchange_strong(Expected, Desired));
|
||||
} while (!AtomicFutex.compare_exchange_strong(Expected, Desired));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static inline bool try_lock(T* Futex) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Expected {};
|
||||
T Desired {1};
|
||||
|
||||
// Try to CAS immediately.
|
||||
if (AtomicFutex->compare_exchange_strong(Expected, Desired)) {
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -262,8 +284,8 @@ static inline bool try_lock(T* Futex) {
|
||||
|
||||
template<typename T>
|
||||
static inline void unlock(T* Futex) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
AtomicFutex->store(0);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
AtomicFutex.store(0);
|
||||
}
|
||||
|
||||
#undef SPINLOOP_8BIT
|
||||
|
||||
@@ -0,0 +1,381 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
|
||||
#if !defined(_WIN32)
|
||||
#include <linux/futex.h> /* Definition of FUTEX_* constants */
|
||||
#include <sys/syscall.h> /* Definition of SYS_* constants */
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <synchapi.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
namespace FEXCore::Utils::WritePriorityMutex {
|
||||
|
||||
// A custom mutex that prioritizes exclusive locks.
|
||||
// In highly contested scenarios, this can help minimize overall contention time.
|
||||
//
|
||||
// Features:
|
||||
// - Up to 32767 pending exclusive locks ("writers")
|
||||
// - Up to 32767 pending shared_locks ("readers")
|
||||
// - Low-overhead waiting via WFE with a fallback to futex on timeout
|
||||
// - Direct writer->reader hand-off and vice-versa to further reduce overhead
|
||||
//
|
||||
// Trade-offs:
|
||||
// - No guaranteed order of wake-ups besides prioritizing writers
|
||||
// - No support for recursive locking
|
||||
// - We can't use FUTEX_LOCK_PI to enable priority inheritance
|
||||
class Mutex final {
|
||||
public:
|
||||
Mutex() = default;
|
||||
|
||||
// Move-only type
|
||||
Mutex(const Mutex&) = delete;
|
||||
Mutex& operator=(const Mutex&) = delete;
|
||||
Mutex(Mutex&& rhs) = delete;
|
||||
Mutex& operator=(Mutex&&) = delete;
|
||||
|
||||
void lock() {
|
||||
// Try a non-blocking lock first.
|
||||
if (try_lock()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Try a quick WFE write-lock.
|
||||
if (Attempt_WFE_WriteLock()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Still couldn't get it. Start waiting.
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected {};
|
||||
uint32_t Desired {};
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
do {
|
||||
// Increment the number of write waiters.
|
||||
Desired = Expected + WRITE_WAITER_INCREMENT;
|
||||
|
||||
LOGMAN_THROW_A_FMT((Desired & WRITE_WAITER_COUNT_MASK) != 0, "Overflow in write-waiters!");
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
#else
|
||||
Expected = AtomicFutex.fetch_add(WRITE_WAITER_INCREMENT);
|
||||
Desired = Expected + WRITE_WAITER_INCREMENT;
|
||||
#endif
|
||||
|
||||
// Thread added to waiter list.
|
||||
Expected = Desired;
|
||||
|
||||
while (true) {
|
||||
bool Sleep = false;
|
||||
|
||||
do {
|
||||
if ((Expected & WRITE_OWNED_BIT) == 0 && (Expected & READ_OWNER_COUNT_MASK) == 0) {
|
||||
// If not write-owned, and no read-owners, try to acquire.
|
||||
LOGMAN_THROW_A_FMT((Expected & WRITE_WAITER_COUNT_MASK) != 0, "Underflow in write-waiters!");
|
||||
|
||||
// Add write-owned bit.
|
||||
Desired = Expected | WRITE_OWNED_BIT;
|
||||
|
||||
// Remove ourselves from the wait list.
|
||||
Desired -= WRITE_WAITER_INCREMENT;
|
||||
|
||||
Sleep = false;
|
||||
} else {
|
||||
// Already write-owned or read-locked. Go to sleep.
|
||||
Desired = Expected;
|
||||
Sleep = true;
|
||||
break;
|
||||
}
|
||||
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
|
||||
if (!Sleep) {
|
||||
// Acquired early.
|
||||
LOGMAN_THROW_A_FMT((Desired & WRITE_OWNED_BIT) == WRITE_OWNED_BIT, "Somehow acquired a write-lock without it being set!");
|
||||
return;
|
||||
}
|
||||
FutexWaitForWriteAvailable(Desired);
|
||||
|
||||
Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
void lock_shared() {
|
||||
// Try an uncontended lock first.
|
||||
if (try_lock_shared()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Try a quick WFE read-lock.
|
||||
if (Attempt_WFE_ReadLock()) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
uint32_t Desired {};
|
||||
|
||||
while (true) {
|
||||
bool Sleep = false;
|
||||
do {
|
||||
if ((Expected & WRITE_OWNED_BIT) == 0 && (Expected & WRITE_WAITER_COUNT_MASK) == 0) {
|
||||
// If no write-owner and no write-waiting, try and acquire.
|
||||
|
||||
Desired = Expected + READ_OWNER_INCREMENT;
|
||||
LOGMAN_THROW_A_FMT((Desired & READ_OWNER_COUNT_MASK) != 0, "Overflow in read-owners!");
|
||||
Sleep = false;
|
||||
} else {
|
||||
// Waiting for lock to become available. Add to waiters.
|
||||
Desired = Expected | READ_WAITER_BIT;
|
||||
Sleep = true;
|
||||
}
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
|
||||
if (!Sleep) {
|
||||
// Acquired early.
|
||||
LOGMAN_THROW_A_FMT((Desired & WRITE_OWNED_BIT) != WRITE_OWNED_BIT, "Somehow read-locked and got a write lock!");
|
||||
return;
|
||||
}
|
||||
|
||||
FutexWaitForReadAvailable(Desired);
|
||||
|
||||
Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
void unlock() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
uint32_t Desired {};
|
||||
do {
|
||||
LOGMAN_THROW_A_FMT((Expected & WRITE_OWNED_BIT) == WRITE_OWNED_BIT, "Trying to write-unlock something not write-locked!");
|
||||
// Remove the exclusive lock bit.
|
||||
Desired = Expected & ~WRITE_OWNED_BIT;
|
||||
|
||||
// If no more writers, then make sure to clear the read-waiters bit as well.
|
||||
if ((Desired & WRITE_WAITER_COUNT_MASK) == 0) {
|
||||
Desired &= ~READ_WAITER_BIT;
|
||||
}
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
|
||||
// If success, then `Expected` has old value. Containing `READ_WAITER_BIT` which was just masked off, and also `WRITE_WAITER_COUNT_MASK`.
|
||||
if ((Expected & WRITE_WAITER_COUNT_MASK)) {
|
||||
// Handle write-write handoff.
|
||||
FutexWakeWriter();
|
||||
} else if ((Expected & READ_WAITER_BIT)) {
|
||||
// Handle write-reader handoff.
|
||||
FutexWakeReaders();
|
||||
}
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Desired {};
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
do {
|
||||
LOGMAN_THROW_A_FMT((Expected & WRITE_OWNED_BIT) != WRITE_OWNED_BIT, "Trying to read-unlock something write-locked!");
|
||||
LOGMAN_THROW_A_FMT((Expected & READ_OWNER_COUNT_MASK) != 0, "Trying to read-unlock something not read-locked!");
|
||||
|
||||
// Decrement the shared counter.
|
||||
Desired = Expected - READ_OWNER_INCREMENT;
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
#else
|
||||
Desired = AtomicFutex.fetch_sub(READ_OWNER_INCREMENT) - READ_OWNER_INCREMENT;
|
||||
#endif
|
||||
|
||||
// Handle read->write handoff if there are any waiting writers, and no readers left.
|
||||
if ((Desired & WRITE_WAITER_COUNT_MASK) && (Desired & READ_OWNER_COUNT_MASK) == 0) {
|
||||
FutexWakeWriter();
|
||||
}
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected = 0;
|
||||
|
||||
// Try and grab the owned bit.
|
||||
uint32_t Desired = WRITE_OWNED_BIT;
|
||||
|
||||
// try to CAS immediately.
|
||||
return AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire);
|
||||
}
|
||||
|
||||
// Can race with other threads trying to lock shared!
|
||||
bool try_lock_shared() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
|
||||
// Exclusively owned or has a list of waiting owners. Can't pass.
|
||||
if ((Expected & WRITE_OWNED_BIT) || (Expected & WRITE_WAITER_COUNT_MASK)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Try to add reader.
|
||||
uint32_t Desired = Expected + READ_OWNER_INCREMENT;
|
||||
LOGMAN_THROW_A_FMT((Desired & READ_OWNER_COUNT_MASK) != 0, "Overflow in read-owners!");
|
||||
|
||||
// Uncontended mutex check
|
||||
return AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire);
|
||||
}
|
||||
|
||||
#if !defined(_WIN32)
|
||||
// Initialize the internal mutex object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Futex = 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
private:
|
||||
|
||||
#if !defined(_WIN32)
|
||||
void FutexWaitForWriteAvailable(uint32_t Expected) {
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAIT_BITSET, Expected, nullptr, nullptr, FUTEX_BITSET_WAIT_WRITERS);
|
||||
}
|
||||
|
||||
// Read-lock waiting for writers to drain out.
|
||||
void FutexWaitForReadAvailable(uint32_t Expected) {
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAIT_BITSET, Expected, nullptr, nullptr, FUTEX_BITSET_WAIT_READERS);
|
||||
}
|
||||
|
||||
// Read-Lock or Write-lock unlocked, wake one writer.
|
||||
// - Read->Write handoff.
|
||||
// - Write->Write handoff.
|
||||
void FutexWakeWriter() {
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAKE_BITSET, 1, nullptr, nullptr, FUTEX_BITSET_WAIT_WRITERS);
|
||||
}
|
||||
|
||||
// Write-lock unlocked, wake read-locks waiting.
|
||||
void FutexWakeReaders() {
|
||||
// Wake all readers.
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAKE_BITSET, INT_MAX, nullptr, nullptr, FUTEX_BITSET_WAIT_READERS);
|
||||
}
|
||||
#else
|
||||
// Writers wait for the full 32-bit futex.
|
||||
void FutexWaitForWriteAvailable(uint32_t Expected) {
|
||||
WaitOnAddress(&Futex, &Expected, sizeof(Futex), INFINITE);
|
||||
}
|
||||
|
||||
// Readers wait for Futex bits [31:16] to be zero.
|
||||
void FutexWaitForReadAvailable(uint32_t Expected) {
|
||||
auto ReadWaiterAddress = reinterpret_cast<uint8_t*>(&Futex) + 2;
|
||||
uint16_t smol_Expected = Expected >> 16;
|
||||
WaitOnAddress(ReadWaiterAddress, &smol_Expected, sizeof(smol_Expected), INFINITE);
|
||||
}
|
||||
|
||||
void FutexWakeWriter() {
|
||||
WakeByAddressSingle(&Futex);
|
||||
}
|
||||
|
||||
void FutexWakeReaders() {
|
||||
auto ReadWaiterAddress = reinterpret_cast<uint8_t*>(&Futex) + 2;
|
||||
WakeByAddressAll(ReadWaiterAddress);
|
||||
}
|
||||
#endif
|
||||
|
||||
// Reuse the SpinWaitLock WFE implementations for read/write lock acquiring with WFE.
|
||||
// Can't reuse the spin-lock directly as some bit-representations are different.
|
||||
// WFE-write-lock is less likely to occur the more read-lock threads are participating. Can still occur so good to try.
|
||||
// WFE-read-lock is actually quite likely to succeed.
|
||||
// Return: true if the lock was acquired.
|
||||
bool Attempt_WFE_WriteLock() {
|
||||
#ifdef _M_ARM_64
|
||||
const auto Begin = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
auto Now = Begin;
|
||||
const auto Duration = FEXCore::Utils::SpinWaitLock::CycleCounterFrequency / CYCLECOUNT_DIVISOR;
|
||||
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
|
||||
while ((Now - Begin) < Duration) {
|
||||
if (Expected == 0) {
|
||||
// Try and grab the owned bit.
|
||||
uint32_t Desired = WRITE_OWNED_BIT;
|
||||
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// One-shot attempt to wait for mask to be zero.
|
||||
Expected = FEXCore::Utils::SpinWaitLock::OneShotWFEBitComparison(&Futex, ~0U, 0U);
|
||||
Now = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
}
|
||||
#endif
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
// Return: true if the lock was acquired.
|
||||
bool Attempt_WFE_ReadLock() {
|
||||
#ifdef _M_ARM_64
|
||||
// Spin on a WFE for a short-amount of time, waiting for write-owned and writer-count to be zero.
|
||||
// - Attempt to acquire read-lock at that point.
|
||||
// - Don't add read-waiters bit on failure, return false.
|
||||
const auto Begin = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
auto Now = Begin;
|
||||
const auto Duration = FEXCore::Utils::SpinWaitLock::CycleCounterFrequency / CYCLECOUNT_DIVISOR;
|
||||
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
uint32_t Desired {};
|
||||
|
||||
while ((Now - Begin) < Duration) {
|
||||
if ((Expected & WRITE_OWNED_BIT) == 0 && (Expected & WRITE_WAITER_COUNT_MASK) == 0) {
|
||||
// If no write-owner and no write-waiting, try and acquire.
|
||||
|
||||
Desired = Expected + READ_OWNER_INCREMENT;
|
||||
LOGMAN_THROW_A_FMT((Desired & READ_OWNER_COUNT_MASK) != 0, "Overflow in read-owners!");
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// One-shot attempt to wait for mask to be zero.
|
||||
Expected = FEXCore::Utils::SpinWaitLock::OneShotWFEBitComparison(&Futex, WRITE_OWNED_BIT | WRITE_WAITER_COUNT_MASK, 0U);
|
||||
Now = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
}
|
||||
#endif
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
constexpr static uint32_t WRITE_OWNED_BIT = 1U << 31;
|
||||
constexpr static uint32_t READ_WAITER_BIT = 1U << 15;
|
||||
constexpr static uint32_t WRITE_WAITER_OFFSET = 16;
|
||||
constexpr static uint32_t WRITE_WAITER_INCREMENT = 1U << WRITE_WAITER_OFFSET;
|
||||
constexpr static uint32_t READ_OWNER_INCREMENT = 1;
|
||||
|
||||
// Count masks
|
||||
constexpr static uint32_t WRITE_WAITER_COUNT_MASK = 0x7FFFU << WRITE_WAITER_OFFSET;
|
||||
constexpr static uint32_t READ_OWNER_COUNT_MASK = 0x7FFFU;
|
||||
|
||||
// Independent futex bit-set masks.
|
||||
// Wait for readers to drain.
|
||||
constexpr static uint32_t FUTEX_BITSET_WAIT_READERS = 1U << 0;
|
||||
// Wait for writers to drain.
|
||||
constexpr static uint32_t FUTEX_BITSET_WAIT_WRITERS = 1U << 1;
|
||||
|
||||
// Only spin on WFE for 0.01ms (10k ns).
|
||||
constexpr static uint64_t CYCLECOUNT_DIVISOR = 1'000'000'000ULL / 10'000U;
|
||||
|
||||
// Layout:
|
||||
// Bits[31]: Write-lock bit.
|
||||
// Bits[30:16]: Write-waiter count.
|
||||
// Bits[15]: Read-waiter bit.
|
||||
// Bits[14:0]: Read-owner count.
|
||||
uint32_t Futex {};
|
||||
};
|
||||
} // namespace FEXCore::Utils::WritePriorityMutex
|
||||
@@ -103,28 +103,25 @@ static inline std::optional<fextl::string> EnumParser(const ArrayPairType& EnumP
|
||||
return fextl::fmt::format("{}", EnumMask);
|
||||
}
|
||||
|
||||
namespace DefaultValues {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) extern const P(type) P(enum);
|
||||
#define OPT_STR(group, enum, json, default) extern const std::string_view P(enum);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
using StringArrayType = fextl::list<fextl::string>;
|
||||
|
||||
namespace Type {
|
||||
using StringArrayType = fextl::list<fextl::string>;
|
||||
#define OPT_BASE(type, group, enum, json, default) using P(enum) = P(type);
|
||||
#define OPT_STR(group, enum, json, default) using P(enum) = fextl::string;
|
||||
#define OPT_STRARRAY(group, enum, json, default) using P(enum) = StringArrayType;
|
||||
namespace detail {
|
||||
template<ConfigOption Option>
|
||||
struct ConfigOptionInfo;
|
||||
#define DEFINE_METAINFO(type, enum, default) \
|
||||
template<> \
|
||||
struct ConfigOptionInfo<ConfigOption::CONFIG_##enum> { \
|
||||
using Type = type; \
|
||||
static auto Default() { \
|
||||
extern default; \
|
||||
return enum; \
|
||||
} \
|
||||
};
|
||||
#define OPT_BASE(type, group, enum, json, default) DEFINE_METAINFO(type, enum, const type enum)
|
||||
#define OPT_STR(group, enum, json, default) DEFINE_METAINFO(fextl::string, enum, const std::string_view enum)
|
||||
#define OPT_STRARRAY(group, enum, json, default) DEFINE_METAINFO(StringArrayType, enum, const std::string_view enum)
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
} // namespace Type
|
||||
#define FEX_CONFIG_OPT(name, enum) \
|
||||
FEXCore::Config::Value<FEXCore::Config::DefaultValues::Type::enum> name { \
|
||||
FEXCore::Config::CONFIG_##enum, \
|
||||
FEXCore::Config::DefaultValues::enum \
|
||||
}
|
||||
|
||||
#undef P
|
||||
} // namespace DefaultValues
|
||||
} // namespace detail
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void SetDataDirectory(std::string_view Path, bool Global);
|
||||
FEX_DEFAULT_VISIBILITY void SetConfigDirectory(const std::string_view Path, bool Global);
|
||||
@@ -135,8 +132,7 @@ FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigDirectory(bool Global);
|
||||
FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigFileLocation(bool Global = false);
|
||||
FEX_DEFAULT_VISIBILITY fextl::string GetApplicationConfig(const std::string_view Program, bool Global);
|
||||
|
||||
using LayerValue =
|
||||
std::variant< fextl::string, DefaultValues::Type::StringArrayType, uint8_t, int8_t, uint16_t, int16_t, uint32_t, int32_t, uint64_t, int64_t, bool >;
|
||||
using LayerValue = std::variant< fextl::string, StringArrayType, uint8_t, int8_t, uint16_t, int16_t, uint32_t, int32_t, uint64_t, int64_t, bool >;
|
||||
|
||||
using LayerOptions = fextl::unordered_map<ConfigOption, LayerValue>;
|
||||
|
||||
@@ -151,16 +147,16 @@ public:
|
||||
return OptionMap.find(Option) != OptionMap.end();
|
||||
}
|
||||
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
std::optional<StringArrayType*> All(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
return &std::get<DefaultValues::Type::StringArrayType>(Value);
|
||||
return &std::get<StringArrayType>(Value);
|
||||
}
|
||||
|
||||
std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
@@ -201,12 +197,12 @@ public:
|
||||
auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
// If the option didn't exist as a StringArrayType yet, emplace it.
|
||||
it = OptionMap.emplace(Option, DefaultValues::Type::StringArrayType {}).first;
|
||||
it = OptionMap.emplace(Option, StringArrayType {}).first;
|
||||
}
|
||||
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
std::get<DefaultValues::Type::StringArrayType>(Value).emplace_back(Data);
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
std::get<StringArrayType>(Value).emplace_back(Data);
|
||||
}
|
||||
|
||||
void Erase(ConfigOption Option) {
|
||||
@@ -236,7 +232,9 @@ FEX_DEFAULT_VISIBILITY fextl::string FindContainerPrefix();
|
||||
FEX_DEFAULT_VISIBILITY void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool Exists(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<StringArrayType*> All(ConfigOption Option);
|
||||
template<typename T>
|
||||
FEX_DEFAULT_VISIBILITY std::optional<T> GetConv(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<fextl::string*> Get(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY void Set(ConfigOption Option, std::string_view Data);
|
||||
FEX_DEFAULT_VISIBILITY void Erase(ConfigOption Option);
|
||||
@@ -271,18 +269,18 @@ public:
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
Value(T Value) requires (!std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
Value(T Value) requires (!std::is_same_v<T, StringArrayType>)
|
||||
{
|
||||
ValueData = std::move(Value);
|
||||
}
|
||||
|
||||
// Array value types.
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) requires (std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) requires (std::is_same_v<T, StringArrayType>)
|
||||
{
|
||||
GetListIfExists(Option, &ValueData);
|
||||
}
|
||||
|
||||
DefaultValues::Type::StringArrayType& All() requires (std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
StringArrayType& All() requires (std::is_same_v<T, StringArrayType>)
|
||||
{
|
||||
return ValueData;
|
||||
}
|
||||
@@ -293,6 +291,38 @@ private:
|
||||
static T GetIfExists(FEXCore::Config::ConfigOption Option, T Default);
|
||||
static T GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default);
|
||||
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List);
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
|
||||
};
|
||||
|
||||
/**
|
||||
* Wrapper around Value that automatically picks the default for the given ConfigOption
|
||||
*/
|
||||
template<ConfigOption Option>
|
||||
struct FEX_DEFAULT_VISIBILITY Getter : public Value<typename detail::ConfigOptionInfo<Option>::Type> {
|
||||
using OptionInfo = detail::ConfigOptionInfo<Option>;
|
||||
Getter()
|
||||
: Value<typename OptionInfo::Type> {Option, OptionInfo::Default()} {}
|
||||
};
|
||||
|
||||
/**
|
||||
* Helper for reading a config value with caching.
|
||||
*
|
||||
* Typically this is used to declare class members so that the value is read
|
||||
* on construction of the parent.
|
||||
*/
|
||||
#define FEX_CONFIG_OPT(name, enum) FEXCore::Config::Getter<FEXCore::Config::ConfigOption::CONFIG_##enum> name {}
|
||||
|
||||
#define OPT_BASE(type, group, enum, json, default) \
|
||||
/** \
|
||||
* Helper for reading a config value. \
|
||||
* \
|
||||
* In contrast to FEX_CONFIG_OPT, this can be used in arbitrary expressions, \
|
||||
* at the expense of not caching the value. Use Getter instead if the value \
|
||||
* is read frequently. \
|
||||
*/ \
|
||||
inline auto Get_##enum() { \
|
||||
return Getter<FEXCore::Config::ConfigOption::CONFIG_##enum> {}; \
|
||||
}
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
|
||||
} // namespace FEXCore::Config
|
||||
@@ -1,10 +1,20 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <shared_mutex>
|
||||
#include <span>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
@@ -20,8 +30,14 @@ namespace HLE {
|
||||
struct ExecutableFileInfo {
|
||||
~ExecutableFileInfo();
|
||||
|
||||
#if __clang_major__ < 16
|
||||
// Workaround for broken aggregate-initialization with std::piecewise_construct
|
||||
ExecutableFileInfo(fextl::unique_ptr<HLE::SourcecodeMap>, uint64_t, fextl::string);
|
||||
ExecutableFileInfo() = default;
|
||||
#endif
|
||||
|
||||
fextl::unique_ptr<HLE::SourcecodeMap> SourcecodeMap;
|
||||
fextl::string FileId;
|
||||
uint64_t FileId = 0;
|
||||
fextl::string Filename;
|
||||
};
|
||||
|
||||
@@ -33,10 +49,129 @@ struct ExecutableFileSectionInfo {
|
||||
uintptr_t FileStartVA;
|
||||
};
|
||||
|
||||
using CodeMapFileId = uint64_t;
|
||||
|
||||
/**
|
||||
* Code maps capture information required for offline code cache generation
|
||||
* and are written to disk during execution of FEX.
|
||||
*
|
||||
* Almost all CodeMap data will be an Entry that indicates blocks to be
|
||||
* compiled for cache generation. The reserved value `LoadExternalLibrary`
|
||||
* indicates that an instance of ExternalLibraryInfo follows (the entry data
|
||||
* itself should be skipped in that case).
|
||||
*/
|
||||
struct CodeMap {
|
||||
// Describes the location of an entry block compiled during execution
|
||||
struct FEX_PACKED Entry {
|
||||
CodeMapFileId FileId;
|
||||
uint32_t BlockOffset;
|
||||
};
|
||||
|
||||
// Describes an external library referenced during execution
|
||||
struct ExternalLibraryInfo {
|
||||
CodeMapFileId ExternalFileId;
|
||||
|
||||
// null-terminated file path; EITHER relative to the main executable OR an absolute path OR starting with a magic identifier:
|
||||
// - WINE/: Path to Wine/Proton installation
|
||||
// - WINEPREFIX/: Path to Wine/Proton prefix
|
||||
// - SLR/: Path to Steam Linux Runtime
|
||||
// At runtime, FEX will always dump absolute paths
|
||||
char Path[];
|
||||
// Followed by padding to a 4 byte boundary
|
||||
};
|
||||
|
||||
// Followed by ExternalLibraryInfo
|
||||
static constexpr Entry LoadExternalLibrary = {0xffff'ffff'ffff'ffff, 0xffff'ffff};
|
||||
|
||||
struct FEX_PACKED SetExecutableFileId {
|
||||
Entry Marker = {0xffff'ffff'ffff'ffff, 0xffff'fffe};
|
||||
CodeMapFileId ExecutableFileId;
|
||||
};
|
||||
|
||||
struct ParsedContents {
|
||||
fextl::string Filename;
|
||||
fextl::set<uint64_t> Blocks;
|
||||
bool IsExecutable = false;
|
||||
};
|
||||
|
||||
// Follows scheme fileid[-nomb]
|
||||
// The nomb ("no multiblock") suffix signifies that the code map is for use without multiblock, only.
|
||||
static fextl::string GetBaseFilename(const ExecutableFileInfo& MainExecutable, bool AddNombSuffix);
|
||||
|
||||
static fextl::map<CodeMapFileId, ParsedContents> ParseCodeMap(std::ifstream& File);
|
||||
};
|
||||
|
||||
struct CodeMapOpener {
|
||||
virtual ~CodeMapOpener() = default;
|
||||
virtual int OpenCodeMapFile() = 0;
|
||||
};
|
||||
|
||||
class CodeMapWriter {
|
||||
public:
|
||||
CodeMapWriter(CodeMapOpener&, bool OpenEagerly = false);
|
||||
~CodeMapWriter();
|
||||
|
||||
// Checks if writing is enabled. Calls to this functions may also be interpreted as signals that writes are about to happen
|
||||
bool IsWriteEnabled(const ExecutableFileSectionInfo&);
|
||||
|
||||
void ResetAfterFork() {
|
||||
if (CodeMapFD.value_or(-1) != -1) {
|
||||
close(CodeMapFD.value());
|
||||
CodeMapFD.reset();
|
||||
}
|
||||
BufferOffset = 0;
|
||||
KnownFileIds.clear();
|
||||
}
|
||||
|
||||
bool IsBackingFD(int FD) const {
|
||||
if (FD == CodeMapFD) {
|
||||
LogMan::Msg::DFmt("Hiding directory entry for code map FD");
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void AppendBlock(const FEXCore::ExecutableFileSectionInfo&, uint64_t Entry);
|
||||
void AppendLibraryLoad(const FEXCore::ExecutableFileInfo&);
|
||||
void AppendSetMainExecutable(const FEXCore::ExecutableFileInfo&);
|
||||
|
||||
// Thread-safely commit any pending data to disk
|
||||
void Flush(size_t Offset);
|
||||
|
||||
private:
|
||||
// Queues data into an internal ring buffer.
|
||||
// Call Flush() to commit the data to disk.
|
||||
void AppendData(std::span<const std::byte> Data);
|
||||
|
||||
// Commit given data range to disk
|
||||
void Flush(size_t Offset, std::unique_lock<std::shared_mutex>&);
|
||||
|
||||
std::shared_mutex Mutex;
|
||||
fextl::vector<std::byte> Buffer;
|
||||
std::atomic<size_t> BufferOffset {0};
|
||||
|
||||
fextl::set<CodeMapFileId> KnownFileIds;
|
||||
|
||||
// std::nullopt: We haven't requested a CodeMapFD yet
|
||||
// value is -1: We requested a CodeMapFD but FEXServer told us not to write any data
|
||||
// other values: Code map writing is active
|
||||
std::optional<int> CodeMapFD;
|
||||
|
||||
CodeMapOpener& FileOpener;
|
||||
};
|
||||
|
||||
class AbstractCodeCache {
|
||||
public:
|
||||
virtual ~AbstractCodeCache() = default;
|
||||
|
||||
/**
|
||||
* Computes a unique identifier for the referenced binary file to be used for
|
||||
* generating the code map.
|
||||
* This identifier is independent of FEX build/runtime configuration and
|
||||
* stable across FEX updates.
|
||||
*/
|
||||
virtual uint64_t ComputeCodeMapId(std::string_view Filename, int FD) = 0;
|
||||
|
||||
/**
|
||||
* Loads a code cache from mapped memory and appends it to the current Core state.
|
||||
* TODO: Optionally recompiles all contained code blocks at runtime for validation.
|
||||
|
||||
@@ -42,9 +42,6 @@ enum OperatingMode {
|
||||
|
||||
using CodeRangeInvalidationFn = std::function<void(uint64_t start, uint64_t Length)>;
|
||||
|
||||
// Nested vector of guest block entrypoints
|
||||
using InvalidatedEntryAccumulator = fextl::vector<fextl::vector<uint64_t>>;
|
||||
|
||||
using CustomIREntrypointHandler = std::function<void(uintptr_t Entrypoint, IR::IREmitter*)>;
|
||||
|
||||
using ExitHandler = std::function<void(Core::InternalThreadState* Thread)>;
|
||||
@@ -139,10 +136,13 @@ public:
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) = 0;
|
||||
|
||||
virtual AbstractCodeCache& GetCodeCache() = 0;
|
||||
virtual void SetCodeMapWriter(fextl::unique_ptr<CodeMapWriter>) = 0;
|
||||
virtual void FlushAndCloseCodeMap() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(
|
||||
FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void
|
||||
InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void
|
||||
|
||||
@@ -104,6 +104,10 @@ struct CPUState {
|
||||
uint64_t avx_high[16][2];
|
||||
|
||||
uint64_t gregs[16] {};
|
||||
uint64_t L1Pointer {};
|
||||
uint64_t L1Mask {};
|
||||
uint64_t callret_sp {};
|
||||
uint64_t _pad1 {};
|
||||
XMMRegs xmm {};
|
||||
|
||||
// Raw segment register indexes
|
||||
@@ -116,8 +120,6 @@ struct CPUState {
|
||||
uint64_t gs_cached {};
|
||||
uint64_t fs_cached {};
|
||||
uint8_t flags[48] {};
|
||||
uint64_t callret_sp {};
|
||||
uint64_t _pad1 {};
|
||||
uint64_t mm[8][2] {};
|
||||
|
||||
// 32bit x86 state
|
||||
@@ -247,6 +249,8 @@ static_assert(offsetof(CPUState, xmm) % 32 == 0, "xmm needs to be 256-bit aligne
|
||||
static_assert(offsetof(CPUState, mm) % 16 == 0, "mm needs to be 128-bit aligned!");
|
||||
static_assert(offsetof(CPUState, gregs[15]) <= 504, "gregs maximum offset must be <= 504 for ldp/stp to work");
|
||||
static_assert(offsetof(CPUState, DeferredSignalRefCount) % 8 == 0, "Needs to be 8-byte aligned");
|
||||
static_assert(offsetof(CPUState, L1Pointer) <= 504, "This needs to be <= 504 for ldp");
|
||||
static_assert(offsetof(CPUState, L1Mask) == (offsetof(CPUState, L1Pointer) + 8), "These two variables are paired");
|
||||
|
||||
struct InternalThreadState;
|
||||
|
||||
@@ -349,13 +353,11 @@ struct JITPointers {
|
||||
uint64_t ExitFunctionLinker {};
|
||||
uint64_t ThreadStopHandlerSpillSRA {};
|
||||
uint64_t ThreadPauseHandlerSpillSRA {};
|
||||
uint64_t UnimplementedInstructionHandler {};
|
||||
uint64_t GuestSignal_SIGILL {};
|
||||
uint64_t GuestSignal_SIGTRAP {};
|
||||
uint64_t GuestSignal_SIGSEGV {};
|
||||
uint64_t SignalReturnHandler {};
|
||||
uint64_t SignalReturnHandlerRT {};
|
||||
uint64_t L1Pointer {};
|
||||
uint64_t L2Pointer {};
|
||||
/** @} */
|
||||
|
||||
@@ -368,8 +370,6 @@ struct JITPointers {
|
||||
// Process specific
|
||||
uint64_t LUDIV {};
|
||||
uint64_t LDIV {};
|
||||
uint64_t LUREM {};
|
||||
uint64_t LREM {};
|
||||
|
||||
// Thread Specific
|
||||
|
||||
@@ -378,8 +378,6 @@ struct JITPointers {
|
||||
* @{ */
|
||||
uint64_t LUDIVHandler {};
|
||||
uint64_t LDIVHandler {};
|
||||
uint64_t LUREMHandler {};
|
||||
uint64_t LREMHandler {};
|
||||
/** @} */
|
||||
} AArch64;
|
||||
|
||||
|
||||
@@ -67,6 +67,10 @@ public:
|
||||
return Config;
|
||||
}
|
||||
|
||||
virtual uintptr_t GetThunkCallbackRET() const {
|
||||
return 0;
|
||||
}
|
||||
|
||||
protected:
|
||||
SignalDelegatorConfig Config;
|
||||
};
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
@@ -80,7 +81,14 @@ private:
|
||||
static_assert(!std::is_move_constructible_v<NonMovableUniquePtr<int>>);
|
||||
static_assert(!std::is_move_assignable_v<NonMovableUniquePtr<int>>);
|
||||
|
||||
struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
// Store used for unaligned LDAXR*/STLXR* emulation.
|
||||
struct UnalignedExclusiveStore {
|
||||
uint64_t Addr;
|
||||
uint64_t Store;
|
||||
uint8_t Size;
|
||||
};
|
||||
|
||||
struct alignas(FEXCore::Utils::FEX_PAGE_SIZE) InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
FEXCore::Core::CpuStateFrame* const CurrentFrame = &BaseFrameState;
|
||||
|
||||
FEXCore::Context::Context* const CTX;
|
||||
@@ -101,6 +109,8 @@ struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
// This pointer is owned by the frontend.
|
||||
FEXCore::SHMStats::ThreadStats* ThreadStats {};
|
||||
|
||||
UnalignedExclusiveStore ExclusiveStore;
|
||||
|
||||
///< Data pointer for exclusive use by the frontend
|
||||
void* FrontendPtr;
|
||||
|
||||
@@ -109,6 +119,10 @@ struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
// The low address of the call-ret stack allocation (not including guard pages)
|
||||
void* CallRetStackBase {};
|
||||
|
||||
uintptr_t JITGuardPage {};
|
||||
uint64_t JITGuardOverflowArgument {};
|
||||
FEXCore::UncheckedLongJump::JumpBuf RestartJump;
|
||||
|
||||
// BaseFrameState should always be at the end, directly before the interrupt fault page
|
||||
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState {};
|
||||
|
||||
@@ -116,8 +130,9 @@ struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
alignas(FEXCore::Utils::FEX_PAGE_SIZE) uint8_t InterruptFaultPage[FEXCore::Utils::FEX_PAGE_SIZE];
|
||||
};
|
||||
static_assert(std::is_standard_layout_v<FEXCore::Core::InternalThreadState>);
|
||||
static_assert(
|
||||
(offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState)) < 4096,
|
||||
"Fault page is outside of immediate range from CPU state");
|
||||
static_assert((offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState)) <
|
||||
FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
"Fault page is outside of immediate range from CPU state");
|
||||
static_assert(sizeof(FEXCore::Core::InternalThreadState) == (FEXCore::Utils::FEX_PAGE_SIZE * 2));
|
||||
|
||||
} // namespace FEXCore::Core
|
||||
@@ -54,10 +54,6 @@ public:
|
||||
virtual ~SyscallHandler() = default;
|
||||
|
||||
virtual uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) = 0;
|
||||
virtual SyscallABI GetSyscallABI(uint64_t Syscall) = 0;
|
||||
virtual FEXCore::IR::SyscallFlags GetSyscallFlags(uint64_t Syscall) const {
|
||||
return FEXCore::IR::SyscallFlags::DEFAULT;
|
||||
}
|
||||
|
||||
SyscallOSABI GetOSABI() const {
|
||||
return OSABI;
|
||||
|
||||
@@ -3,31 +3,12 @@
|
||||
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
|
||||
#include <compare>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
enum class SyscallFlags : uint8_t {
|
||||
DEFAULT = 0,
|
||||
// Syscalldoesn't care about CPUState being serialized up to the syscall instruction.
|
||||
// Means dead code elimination can optimize through a syscall operation.
|
||||
OPTIMIZETHROUGH = 1 << 0,
|
||||
// Syscall only reads the passed in arguments. Doesn't read CPUState.
|
||||
NOSYNCSTATEONENTRY = 1 << 1,
|
||||
// Syscall doesn't return. Code generation after syscall return can be removed.
|
||||
NORETURN = 1 << 2,
|
||||
// Syscall doesn't have any side-effects, so if the result isn't used then it can be removed.
|
||||
NOSIDEEFFECTS = 1 << 3,
|
||||
// Syscall doesn't return a result.
|
||||
// Means the resulting register shouldn't be written (Usually RAX).
|
||||
// Usually used with !NOSYNCSTATEONENTRY, so the syscall can modify CPU state entirely.
|
||||
// Then on return FEXCore picks up the new state.
|
||||
NORETURNEDRESULT = 1 << 4,
|
||||
};
|
||||
|
||||
FEX_DEF_NUM_OPS(SyscallFlags)
|
||||
|
||||
// This enum of named vector constants are linked to an array in CPUBackend.cpp.
|
||||
// This is used with the IROp `LoadNamedVectorConstant` to load a vector constant
|
||||
// that would otherwise be costly to materialize.
|
||||
@@ -96,15 +77,7 @@ enum IndexNamedVectorConstant : uint8_t {
|
||||
|
||||
struct SHA256Sum final {
|
||||
uint8_t data[32];
|
||||
[[nodiscard]]
|
||||
bool operator<(const SHA256Sum& rhs) const {
|
||||
return memcmp(data, rhs.data, sizeof(data)) < 0;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool operator==(const SHA256Sum& rhs) const {
|
||||
return memcmp(data, rhs.data, sizeof(data)) == 0;
|
||||
}
|
||||
[[nodiscard]] auto operator<=>(const SHA256Sum&) const noexcept = default;
|
||||
};
|
||||
|
||||
typedef void ThunkedFunction(void* ArgsRv);
|
||||
|
||||
@@ -82,12 +82,15 @@ inline bool VirtualProtect(void* Ptr, size_t Size, ProtectOptions options) {
|
||||
return ::VirtualProtect(Ptr, Size, prot, nullptr) == 0;
|
||||
}
|
||||
|
||||
inline void VirtualName(const char*, void*, size_t) {}
|
||||
|
||||
#else
|
||||
using MMAP_Hook = void* (*)(void*, size_t, int, int, int, off_t);
|
||||
using MUNMAP_Hook = int (*)(void*, size_t);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY extern MMAP_Hook mmap;
|
||||
FEX_DEFAULT_VISIBILITY extern MUNMAP_Hook munmap;
|
||||
FEX_DEFAULT_VISIBILITY extern void VirtualName(const char* Name, void* Ptr, size_t Size);
|
||||
|
||||
// All commit parameters are ignored here, they are unnecessary as Linux supports overcommit
|
||||
|
||||
|
||||
@@ -12,8 +12,6 @@ struct InternalThreadState;
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
enum class UnalignedHandlerType {
|
||||
///< Don't backpatch code, instead handle inside SIGBUS handler.
|
||||
Paranoid,
|
||||
///< Backpatch unaligned access to half-barrier based atomic.
|
||||
HalfBarrier,
|
||||
///< Backpatch unaligned access to non-atomic.
|
||||
@@ -26,7 +24,7 @@ enum class UnalignedHandlerType {
|
||||
* This is an OS agnostic handler where the frontend must provide FEXCore with the information necessary to know if this is safe.
|
||||
* This does not check if the PC is within a JIT code buffer, the frontend must provide that safety with `CPUBackend::IsAddressInCodeBuffer`.
|
||||
*
|
||||
* @param ParanoidTSO If the unaligned fault needs to handled directly or can be backpatched.
|
||||
* @param HandleType Type of TSO handling to use.
|
||||
* @param ProgramCounter The location in memory for the instruction that did the access
|
||||
* @param GPRs The array of GPRs from the signal context. This will be modified and the host context needs to be updated on signal return.
|
||||
*
|
||||
@@ -34,6 +32,6 @@ enum class UnalignedHandlerType {
|
||||
* by. FEXCore will return a positive or negative offset depending on internal handling.
|
||||
*/
|
||||
[[nodiscard]]
|
||||
FEX_DEFAULT_VISIBILITY std::optional<int32_t>
|
||||
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType, uintptr_t ProgramCounter, uint64_t* GPRs);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<int32_t> HandleUnalignedAccess(
|
||||
FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType, uintptr_t ProgramCounter, uint64_t* GPRs, bool IsJIT = true);
|
||||
} // namespace FEXCore::ArchHelpers::Arm64
|
||||
@@ -0,0 +1,39 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
// Reimplementation of longjmp without glibc fortification checks.
|
||||
// This is useful when false positives need to be avoided or when using
|
||||
// a libc implementation that does not implement std::longjmp.
|
||||
namespace FEXCore::UncheckedLongJump {
|
||||
// JumpBuf definition needs to be public because the frontend needs to understand it.
|
||||
#if defined(_M_ARM_64)
|
||||
struct JumpBuf {
|
||||
// All the registers that are required by AAPCS64 to save.
|
||||
// GPRs
|
||||
// X19, X20, X21, X22,
|
||||
// X23, X24, X25, X26,
|
||||
// X27, X28, X29, X30,
|
||||
//
|
||||
// Lower 64-bits:
|
||||
// V8, V9, V10, V11,
|
||||
// V12, V13, V14, V15,
|
||||
//
|
||||
// SP,
|
||||
uint64_t Registers[21];
|
||||
};
|
||||
#else
|
||||
struct JumpBuf {
|
||||
// Registers to preserve
|
||||
// RBX, RSP, RBP, R12, R13, R14, R15,
|
||||
// <return address>
|
||||
uint64_t Registers[8];
|
||||
};
|
||||
#endif
|
||||
|
||||
[[nodiscard]] FEX_DEFAULT_VISIBILITY uint64_t SetJump(JumpBuf& Buffer);
|
||||
[[noreturn]] FEX_DEFAULT_VISIBILITY void LongJump(const JumpBuf& Buffer, uint64_t Value);
|
||||
FEX_DEFAULT_VISIBILITY void ManuallyLoadJumpBuf(const JumpBuf& Buffer, uint64_t Value, uint64_t* GPRs, __uint128_t* FPRs, uint64_t* PC);
|
||||
} // namespace FEXCore::UncheckedLongJump
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <atomic>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
#ifdef _M_X86_64
|
||||
@@ -39,10 +40,12 @@ enum class AppType : uint8_t {
|
||||
WIN_WOW64,
|
||||
};
|
||||
|
||||
// Only append new members to the end of {ThreadStatsHeader, ThreadStats} to allow old tools time to support new information.
|
||||
// FEX isn't guaranteeing /not/ breaking compatibility with versions, but trying to not cause too much churn.
|
||||
struct ThreadStatsHeader {
|
||||
uint8_t Version;
|
||||
AppType app_type;
|
||||
uint8_t _pad[2];
|
||||
uint16_t ThreadStatsSize;
|
||||
char fex_version[48];
|
||||
std::atomic<uint32_t> Head;
|
||||
std::atomic<uint32_t> Size;
|
||||
@@ -61,8 +64,17 @@ struct ThreadStats {
|
||||
uint64_t AccumulatedSIGBUSCount;
|
||||
uint64_t AccumulatedSMCCount;
|
||||
uint64_t AccumulatedFloatFallbackCount;
|
||||
|
||||
uint64_t AccumulatedCacheMissCount;
|
||||
uint64_t AccumulatedCacheReadLockTime;
|
||||
uint64_t AccumulatedCacheWriteLockTime;
|
||||
|
||||
uint64_t AccumulatedJITCount;
|
||||
};
|
||||
|
||||
// Ensure 16-byte alignment to take advantage of ARM single-copy atomicity.
|
||||
static_assert(sizeof(ThreadStats) % 16 == 0, "Needs to be 16-byte aligned!");
|
||||
|
||||
template<typename T, size_t FlatOffset = 0>
|
||||
class AccumulationBlock final {
|
||||
public:
|
||||
|
||||
@@ -28,4 +28,19 @@ inline fextl::string Trim(fextl::string String, std::string_view TrimTokens = "
|
||||
return RightTrim(LeftTrim(std::move(String), TrimTokens), TrimTokens);
|
||||
}
|
||||
|
||||
inline fextl::string& ReplaceAllInPlace(fextl::string& Str, std::string_view Token, std::string_view New) {
|
||||
const auto OriginalTokenSize = Token.size();
|
||||
const auto NewTokenSize = New.size();
|
||||
|
||||
size_t TokenPos {};
|
||||
auto TokenIter = Str.find(Token, TokenPos);
|
||||
while (TokenIter != Str.npos) {
|
||||
Str.replace(TokenIter, OriginalTokenSize, New);
|
||||
TokenPos += NewTokenSize;
|
||||
TokenIter = Str.find(Token, TokenPos);
|
||||
}
|
||||
|
||||
return Str;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::StringUtils
|
||||
@@ -3,6 +3,8 @@
|
||||
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
|
||||
#include <atomic>
|
||||
@@ -37,12 +39,6 @@ namespace FEXCore::Utils {
|
||||
*/
|
||||
class IntrusivePooledAllocator {
|
||||
public:
|
||||
template<typename T>
|
||||
struct AllocationInfo {
|
||||
T Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
struct MemoryBuffer;
|
||||
/**
|
||||
* @brief Container for tracking the buffers
|
||||
@@ -380,6 +376,8 @@ private:
|
||||
class PooledAllocatorVirtual final : public IntrusivePooledAllocator {
|
||||
public:
|
||||
PooledAllocatorVirtual() = default;
|
||||
PooledAllocatorVirtual(const char* Name)
|
||||
: Name {Name} {}
|
||||
|
||||
virtual ~PooledAllocatorVirtual() {
|
||||
FreeAllBuffers();
|
||||
@@ -387,12 +385,54 @@ public:
|
||||
|
||||
private:
|
||||
void* Alloc(size_t Size) override {
|
||||
return FEXCore::Allocator::VirtualAlloc(Size);
|
||||
auto Result = FEXCore::Allocator::VirtualAlloc(Size);
|
||||
if (Name) {
|
||||
FEXCore::Allocator::VirtualName(Name, Result, Size);
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
void Free(void* Ptr, size_t Size) override {
|
||||
FEXCore::Allocator::VirtualFree(Ptr, Size);
|
||||
}
|
||||
|
||||
const char* Name {};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Thread pool allocator that allocates and frees objects that uses mmap, with a guard page.
|
||||
*
|
||||
* The last page of the size provided has the guard.
|
||||
*/
|
||||
class PooledAllocatorVirtualWithGuard final : public IntrusivePooledAllocator {
|
||||
public:
|
||||
PooledAllocatorVirtualWithGuard() = default;
|
||||
PooledAllocatorVirtualWithGuard(const char* Name)
|
||||
: Name {Name} {}
|
||||
|
||||
virtual ~PooledAllocatorVirtualWithGuard() {
|
||||
FreeAllBuffers();
|
||||
}
|
||||
|
||||
private:
|
||||
void* Alloc(size_t Size) override {
|
||||
auto Ptr = FEXCore::Allocator::VirtualAlloc(Size);
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::None)) {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
if (Name) {
|
||||
FEXCore::Allocator::VirtualName(Name, Ptr, Size);
|
||||
}
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
void Free(void* Ptr, size_t Size) override {
|
||||
FEXCore::Allocator::VirtualFree(Ptr, Size);
|
||||
}
|
||||
|
||||
const char* Name {};
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -452,6 +492,11 @@ public:
|
||||
UnclaimBuffer();
|
||||
}
|
||||
|
||||
struct AllocationInfo {
|
||||
Type Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Return the owned buffer or allocate another one from the `Allocator`
|
||||
*
|
||||
@@ -460,9 +505,9 @@ public:
|
||||
*
|
||||
* @param NewSize Optional new size for managed data
|
||||
*
|
||||
* @return object of type `Type` allocated within the selected buffer
|
||||
* @return A usable pointer of type `Type` and the size of the backing store.
|
||||
*/
|
||||
Type ReownOrClaimBuffer(std::optional<size_t> NewSize = std::nullopt) {
|
||||
AllocationInfo ReownOrClaimBufferWithSize(std::optional<size_t> NewSize = std::nullopt) {
|
||||
// Check if we can cheaply re-own a previous buffer
|
||||
std::optional Buffer =
|
||||
IntrusivePooledAllocator::IsClientBufferOwned(ClientOwnedFlag) ? Info : ThreadAllocator.TryToReownBuffer(Info, Size, &ClientOwnedFlag);
|
||||
@@ -485,7 +530,14 @@ public:
|
||||
// Leaving this here for future excavation that will definitely occur here
|
||||
// memset((*Info)->Ptr, 0, Size);
|
||||
|
||||
return reinterpret_cast<Type>((*Info)->Ptr);
|
||||
return {
|
||||
.Ptr = reinterpret_cast<Type>((*Info)->Ptr),
|
||||
.Size = (*Info)->Size,
|
||||
};
|
||||
}
|
||||
|
||||
Type ReownOrClaimBuffer(std::optional<size_t> NewSize = std::nullopt) {
|
||||
return ReownOrClaimBufferWithSize(NewSize).Ptr;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -2,7 +2,9 @@
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/fextl/allocator.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
|
||||
#include <memory_resource>
|
||||
#include <fmt/format.h>
|
||||
@@ -26,6 +28,90 @@ namespace pmr {
|
||||
|
||||
FEX_DEFAULT_VISIBILITY std::pmr::memory_resource* get_default_resource();
|
||||
|
||||
/**
|
||||
* @brief A `std::pmr::monotonic_buffer_resource` compatible class.
|
||||
*
|
||||
* Allocates internal buffers on page boundaries and names them for buffer tracking.
|
||||
*/
|
||||
class named_monotonic_page_buffer_resource final : public std::pmr::memory_resource {
|
||||
public:
|
||||
explicit named_monotonic_page_buffer_resource(const char* Name)
|
||||
: Name {Name} {}
|
||||
|
||||
void release() noexcept {
|
||||
for (auto& Iter : Buffers) {
|
||||
FEXCore::Allocator::VirtualFree(Iter.Buffer, Iter.BufferSize);
|
||||
}
|
||||
Buffers.clear();
|
||||
|
||||
CurrentBufferRemaining = 0;
|
||||
CurrentAllocationSize = FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
}
|
||||
|
||||
protected:
|
||||
void* do_allocate(std::size_t bytes, std::size_t alignment) override {
|
||||
LOGMAN_THROW_A_FMT(bytes != 0, "Nope");
|
||||
LOGMAN_THROW_A_FMT(alignment <= FEXCore::Utils::FEX_PAGE_SIZE, "Nope");
|
||||
|
||||
// Wow, an actual use case of std::align in the wild.
|
||||
void* NewPointer = std::align(alignment, bytes, CurrentBuffer, CurrentBufferRemaining);
|
||||
if (!NewPointer) [[unlikely]] {
|
||||
AllocateNewBuffer(bytes, alignment);
|
||||
NewPointer = CurrentBuffer;
|
||||
}
|
||||
|
||||
CurrentBuffer = static_cast<char*>(CurrentBuffer) + bytes;
|
||||
CurrentBufferRemaining -= bytes;
|
||||
|
||||
return NewPointer;
|
||||
}
|
||||
|
||||
void do_deallocate(void*, std::size_t, std::size_t) override {
|
||||
// Explicit no-op.
|
||||
}
|
||||
|
||||
bool do_is_equal(const std::pmr::memory_resource& other) const noexcept override {
|
||||
return this == &other;
|
||||
}
|
||||
|
||||
private:
|
||||
const char* Name;
|
||||
|
||||
// Allocate a new buffer that can at least fit the passed in bytes with alignment.
|
||||
void AllocateNewBuffer(std::size_t bytes, std::size_t) {
|
||||
bytes = FEXCore::AlignUp(bytes, CurrentAllocationSize);
|
||||
void* Ptr = FEXCore::Allocator::VirtualAlloc(bytes);
|
||||
if (Name) {
|
||||
FEXCore::Allocator::VirtualName(Name, Ptr, bytes);
|
||||
}
|
||||
|
||||
Buffers.emplace_back(BufferData {
|
||||
.Buffer = Ptr,
|
||||
.BufferSize = bytes,
|
||||
});
|
||||
|
||||
CurrentBuffer = Ptr;
|
||||
CurrentBufferRemaining = bytes;
|
||||
|
||||
// Multiply the allocation size by 1.5 for the next allocation
|
||||
// Avoid double math because of ugly conversions.
|
||||
CurrentAllocationSize = FEXCore::AlignUp(CurrentAllocationSize + (CurrentAllocationSize >> 1), FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
|
||||
// Current buffer management.
|
||||
void* CurrentBuffer {};
|
||||
size_t CurrentBufferRemaining {};
|
||||
|
||||
struct BufferData final {
|
||||
void* Buffer;
|
||||
size_t BufferSize;
|
||||
};
|
||||
|
||||
fextl::list<BufferData> Buffers {};
|
||||
|
||||
size_t CurrentAllocationSize = FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is similar to the std::pmr::monotonic_buffer_resource.
|
||||
*
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/allocator.h>
|
||||
|
||||
#include <tsl/robin_set.h>
|
||||
|
||||
namespace fextl {
|
||||
template<class Key, class Hash = std::hash<Key>, class KeyEqual = std::equal_to<Key>, class Allocator = fextl::FEXAlloc<Key>>
|
||||
using robin_set = tsl::robin_set<Key, Hash, KeyEqual, Allocator>;
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <sys/mman.h>
|
||||
|
||||
template<typename T>
|
||||
bool HasSyscallError(T Result) {
|
||||
constexpr uint64_t MAX_ERRNO = 0xFFFF'FFFF'FFFF'0001ULL;
|
||||
return reinterpret_cast<uint64_t>(Result) >= MAX_ERRNO;
|
||||
}
|
||||
|
||||
TEST_CASE("Allocator - Fixed replacement") {
|
||||
const auto RegionSize = 128 * 1024 * 1024;
|
||||
fextl::vector<FEXCore::Allocator::MemoryRegion> MemoryRegions {};
|
||||
for (size_t i = 0; i < 2; ++i) {
|
||||
auto Ptr = mmap(nullptr, RegionSize, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
MemoryRegions.emplace_back(FEXCore::Allocator::MemoryRegion {
|
||||
.Ptr = Ptr,
|
||||
.Size = RegionSize,
|
||||
});
|
||||
}
|
||||
|
||||
auto Allocator = Alloc::OSAllocator::Create64BitAllocatorWithRegions(MemoryRegions);
|
||||
auto Base = Allocator->Mmap(nullptr, 4096, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
REQUIRE(!HasSyscallError(Base));
|
||||
|
||||
// Allocate perfectly overlapping pages. Allocate as many pages as the region.
|
||||
// FEX had a bug where the allocator could run out of memory with MAP_FIXED.
|
||||
for (size_t i = 0; i < (RegionSize / 4096); ++i) {
|
||||
auto NewBase = Allocator->Mmap(Base, 4096, PROT_NONE, MAP_FIXED | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
REQUIRE(Base == NewBase);
|
||||
}
|
||||
|
||||
Alloc::OSAllocator::ReleaseAllocatorWorkaround(std::move(Allocator));
|
||||
}
|
||||
|
||||
TEST_CASE("Allocator - Non-Fit") {
|
||||
const auto RegionSize = 128 * 1024 * 1024;
|
||||
fextl::vector<FEXCore::Allocator::MemoryRegion> MemoryRegions {};
|
||||
for (size_t i = 0; i < 2; ++i) {
|
||||
auto Ptr = mmap(nullptr, RegionSize, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
MemoryRegions.emplace_back(FEXCore::Allocator::MemoryRegion {
|
||||
.Ptr = Ptr,
|
||||
.Size = RegionSize,
|
||||
});
|
||||
}
|
||||
|
||||
auto Allocator = Alloc::OSAllocator::Create64BitAllocatorWithRegions(MemoryRegions);
|
||||
auto Base = Allocator->Mmap(nullptr, RegionSize / 4, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
REQUIRE(!HasSyscallError(Base));
|
||||
|
||||
// Try to allocate within the whole VMA size minus a small amount.
|
||||
// FEX had a bug where if the allocation fit within a VMA region, it would try and allocate past the end without checking.
|
||||
// Only occurred when `MAP_FIXED` was used.
|
||||
auto NewBase = Allocator->Mmap(Base, RegionSize - (4096 * 64), PROT_NONE, MAP_FIXED | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
// Must either fit in the VMA region, or fail.
|
||||
// - If it matches previous allocation, then it fit in the VMA region.
|
||||
// - This can happen if FEX's allocator gains support for VMA merging.
|
||||
// - If it errors, then it doesn't fit in the VMA region.
|
||||
REQUIRE((NewBase == Base || HasSyscallError(NewBase)));
|
||||
|
||||
Alloc::OSAllocator::ReleaseAllocatorWorkaround(std::move(Allocator));
|
||||
}
|
||||
@@ -3,6 +3,7 @@
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
|
||||
#include "Utils/Allocator/FlexBitSet.h"
|
||||
#include <sys/mman.h>
|
||||
|
||||
TEST_CASE("FlexBitSet - Sizing") {
|
||||
// Ensure that FlexBitSet sizing is correct.
|
||||
@@ -40,3 +41,25 @@ TEST_CASE("FlexBitSet - Sizing") {
|
||||
CHECK(FEXCore::FlexBitSet<uint32_t>::SizeInBits(sizeof(uint32_t) * 8) == sizeof(uint32_t) * 8);
|
||||
CHECK(FEXCore::FlexBitSet<uint64_t>::SizeInBits(sizeof(uint64_t) * 8) == sizeof(uint64_t) * 8);
|
||||
}
|
||||
|
||||
TEST_CASE("FlexBitSet - Limit") {
|
||||
// Ensure that the FlexBitSet doesn't read past the limits, and returns correct indexes.
|
||||
const auto Size = 4096 * 3;
|
||||
auto Ptr = mmap(nullptr, Size, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
auto PtrMiddle = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(Ptr) + 4096);
|
||||
REQUIRE(mprotect(PtrMiddle, 4096, PROT_READ | PROT_WRITE) != -1);
|
||||
|
||||
using ElementType = uint8_t;
|
||||
const size_t NumElements = 4096 * 8;
|
||||
auto FlexBit = reinterpret_cast<FEXCore::FlexBitSet<ElementType>*>(PtrMiddle);
|
||||
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
auto Result = FlexBit->ForwardScanForRange<true>(i, 1, NumElements);
|
||||
CHECK(Result.FoundElement == i);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
auto Result = FlexBit->BackwardScanForRange<true>(i, 1, 0);
|
||||
CHECK(Result.FoundElement == i);
|
||||
}
|
||||
}
|
||||
@@ -9,17 +9,17 @@ using namespace ARMEmitter;
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
adr(Reg::r30, &Label);
|
||||
(void)adr(Reg::r30, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x10fffffe);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
adr(Reg::r30, &Label);
|
||||
Bind(&Label);
|
||||
(void)adr(Reg::r30, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x1000003e);
|
||||
@@ -27,17 +27,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
adr(Reg::r30, &Label);
|
||||
(void)adr(Reg::r30, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x10fffffe);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
adr(Reg::r30, &Label);
|
||||
Bind(&Label);
|
||||
(void)adr(Reg::r30, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x1000003e);
|
||||
@@ -45,42 +45,42 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
adrp(Reg::r30, &Label);
|
||||
(void)adrp(Reg::r30, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x9000001e);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
adrp(Reg::r30, &Label);
|
||||
(void)adrp(Reg::r30, &Label);
|
||||
// Move label a page away
|
||||
for (size_t i = 0; i < 1023; ++i) {
|
||||
nop();
|
||||
}
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb000001e);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
adrp(Reg::r30, &Label);
|
||||
(void)adrp(Reg::r30, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x9000001e);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
adrp(Reg::r30, &Label);
|
||||
(void)adrp(Reg::r30, &Label);
|
||||
// Move label a page away
|
||||
for (size_t i = 0; i < 1023; ++i) {
|
||||
nop();
|
||||
}
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb000001e);
|
||||
}
|
||||
@@ -88,47 +88,49 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
|
||||
{
|
||||
// Will generate adr.
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
CHECK(DisassembleEncoding(1) == 0x10fffffe);
|
||||
}
|
||||
{
|
||||
// Will generate nop + adr.
|
||||
// Will generate nop + nop + adr.
|
||||
ForwardLabel Label;
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
Bind(&Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(1) == 0x1000003e);
|
||||
CHECK(DisassembleEncoding(0) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(2) == 0x1000003e);
|
||||
}
|
||||
{
|
||||
// Will generate adr.
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x10fffffe);
|
||||
}
|
||||
|
||||
{
|
||||
// Will generate nop + adr.
|
||||
// Will generate nop + nop + adr.
|
||||
BiDirectionalLabel Label;
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
Bind(&Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(1) == 0x1000003e);
|
||||
CHECK(DisassembleEncoding(1) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(2) == 0x1000003e);
|
||||
}
|
||||
|
||||
{
|
||||
// Will generate adrp.
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
// Move adrp 1MB away.
|
||||
@@ -136,51 +138,53 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
|
||||
nop();
|
||||
}
|
||||
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
nop();
|
||||
CHECK(DisassembleEncoding(262145) == 0x90fff81e);
|
||||
CHECK(DisassembleEncoding(262146) == 0xd503201f);
|
||||
}
|
||||
|
||||
{
|
||||
// Will generate nop + adrp.
|
||||
// Will generate nop + nop + adrp.
|
||||
ForwardLabel Label;
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
|
||||
// Move label 1MB away, plus a page, and then aligned to a page.
|
||||
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 2); ++i) {
|
||||
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 3); ++i) {
|
||||
nop();
|
||||
}
|
||||
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(1) == 0x9000081e);
|
||||
CHECK(DisassembleEncoding(1) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(2) == 0x9000081e);
|
||||
}
|
||||
|
||||
{
|
||||
// Will generate adrp + add.
|
||||
// Will generate nop + adrp + add.
|
||||
ForwardLabel Label;
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
|
||||
// Move label 1MB away, plus a page, plus one instruction.
|
||||
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 1); ++i) {
|
||||
nop();
|
||||
}
|
||||
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb000081e);
|
||||
CHECK(DisassembleEncoding(1) == 0x910013de);
|
||||
CHECK(DisassembleEncoding(0) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(1) == 0xb000081e);
|
||||
CHECK(DisassembleEncoding(2) == 0x910013de);
|
||||
}
|
||||
|
||||
|
||||
{
|
||||
// Will generate adrp.
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
// Move adrp 1MB away.
|
||||
@@ -188,44 +192,46 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
|
||||
nop();
|
||||
}
|
||||
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
nop();
|
||||
CHECK(DisassembleEncoding(262145) == 0x90fff81e);
|
||||
CHECK(DisassembleEncoding(262146) == 0xd503201f);
|
||||
}
|
||||
|
||||
{
|
||||
// Will generate nop + adrp.
|
||||
// Will generate nop + nop + adrp.
|
||||
BiDirectionalLabel Label;
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
|
||||
// Move label 1MB away, plus a page, and then aligned to a page.
|
||||
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 2); ++i) {
|
||||
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 3); ++i) {
|
||||
nop();
|
||||
}
|
||||
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(1) == 0x9000081e);
|
||||
CHECK(DisassembleEncoding(1) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(2) == 0x9000081e);
|
||||
}
|
||||
|
||||
{
|
||||
// Will generate adrp + add.
|
||||
// Will generate nop + adrp + add.
|
||||
BiDirectionalLabel Label;
|
||||
LongAddressGen(Reg::r30, &Label);
|
||||
(void)LongAddressGen(Reg::r30, &Label);
|
||||
|
||||
// Move label 1MB away, plus a page, plus one instruction.
|
||||
for (size_t i = 0; i < ((1 * 1024 * 1024 + 4096) / 4 - 1); ++i) {
|
||||
nop();
|
||||
}
|
||||
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb000081e);
|
||||
CHECK(DisassembleEncoding(1) == 0x910013de);
|
||||
CHECK(DisassembleEncoding(0) == 0xd503201f);
|
||||
CHECK(DisassembleEncoding(1) == 0xb000081e);
|
||||
CHECK(DisassembleEncoding(2) == 0x910013de);
|
||||
}
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Add/subtract immediate") {
|
||||
|
||||
@@ -9,17 +9,17 @@ using namespace ARMEmitter;
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Conditional branch immediate") {
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
b(Condition::CC_PL, &Label);
|
||||
(void)b(Condition::CC_PL, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x54ffffe5);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
b(Condition::CC_PL, &Label);
|
||||
Bind(&Label);
|
||||
(void)b(Condition::CC_PL, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x54000025);
|
||||
@@ -27,17 +27,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Conditional branch immediat
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
b(Condition::CC_PL, &Label);
|
||||
(void)b(Condition::CC_PL, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x54ffffe5);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
b(Condition::CC_PL, &Label);
|
||||
Bind(&Label);
|
||||
(void)b(Condition::CC_PL, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x54000025);
|
||||
@@ -46,17 +46,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Conditional branch immediat
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Branch consistent conditional") {
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
bc(Condition::CC_PL, &Label);
|
||||
(void)bc(Condition::CC_PL, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x54fffff5);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
bc(Condition::CC_PL, &Label);
|
||||
Bind(&Label);
|
||||
(void)bc(Condition::CC_PL, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x54000035);
|
||||
@@ -64,17 +64,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Branch consistent condition
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
bc(Condition::CC_PL, &Label);
|
||||
(void)bc(Condition::CC_PL, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x54fffff5);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
bc(Condition::CC_PL, &Label);
|
||||
Bind(&Label);
|
||||
(void)bc(Condition::CC_PL, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x54000035);
|
||||
@@ -89,17 +89,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch regist
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immediate") {
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
b(&Label);
|
||||
(void)b(&Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x17ffffff);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
b(&Label);
|
||||
Bind(&Label);
|
||||
(void)b(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x14000001);
|
||||
@@ -107,17 +107,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immedi
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
b(&Label);
|
||||
(void)b(&Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x17ffffff);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
b(&Label);
|
||||
Bind(&Label);
|
||||
(void)b(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x14000001);
|
||||
@@ -125,17 +125,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immedi
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
bl(&Label);
|
||||
(void)bl(&Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x97ffffff);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
bl(&Label);
|
||||
Bind(&Label);
|
||||
(void)bl(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x94000001);
|
||||
@@ -143,17 +143,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immedi
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
bl(&Label);
|
||||
(void)bl(&Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x97ffffff);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
bl(&Label);
|
||||
Bind(&Label);
|
||||
(void)bl(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x94000001);
|
||||
@@ -162,17 +162,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Unconditional branch immedi
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
cbz(Size::i32Bit, Reg::r29, &Label);
|
||||
(void)cbz(Size::i32Bit, Reg::r29, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x34fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
cbz(Size::i32Bit, Reg::r29, &Label);
|
||||
Bind(&Label);
|
||||
(void)cbz(Size::i32Bit, Reg::r29, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x3400003d);
|
||||
@@ -180,17 +180,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
cbz(Size::i32Bit, Reg::r29, &Label);
|
||||
(void)cbz(Size::i32Bit, Reg::r29, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x34fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
cbz(Size::i32Bit, Reg::r29, &Label);
|
||||
Bind(&Label);
|
||||
(void)cbz(Size::i32Bit, Reg::r29, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x3400003d);
|
||||
@@ -198,17 +198,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
cbz(Size::i64Bit, Reg::r29, &Label);
|
||||
(void)cbz(Size::i64Bit, Reg::r29, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0xb4fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
cbz(Size::i64Bit, Reg::r29, &Label);
|
||||
Bind(&Label);
|
||||
(void)cbz(Size::i64Bit, Reg::r29, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb400003d);
|
||||
@@ -216,17 +216,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
cbz(Size::i64Bit, Reg::r29, &Label);
|
||||
(void)cbz(Size::i64Bit, Reg::r29, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0xb4fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
cbz(Size::i64Bit, Reg::r29, &Label);
|
||||
Bind(&Label);
|
||||
(void)cbz(Size::i64Bit, Reg::r29, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb400003d);
|
||||
@@ -234,17 +234,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
cbnz(Size::i32Bit, Reg::r29, &Label);
|
||||
(void)cbnz(Size::i32Bit, Reg::r29, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x35fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
cbnz(Size::i32Bit, Reg::r29, &Label);
|
||||
Bind(&Label);
|
||||
(void)cbnz(Size::i32Bit, Reg::r29, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x3500003d);
|
||||
@@ -252,17 +252,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
cbnz(Size::i32Bit, Reg::r29, &Label);
|
||||
(void)cbnz(Size::i32Bit, Reg::r29, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x35fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
cbnz(Size::i32Bit, Reg::r29, &Label);
|
||||
Bind(&Label);
|
||||
(void)cbnz(Size::i32Bit, Reg::r29, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x3500003d);
|
||||
@@ -270,17 +270,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
cbnz(Size::i64Bit, Reg::r29, &Label);
|
||||
(void)cbnz(Size::i64Bit, Reg::r29, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0xb5fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
cbnz(Size::i64Bit, Reg::r29, &Label);
|
||||
Bind(&Label);
|
||||
(void)cbnz(Size::i64Bit, Reg::r29, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb500003d);
|
||||
@@ -288,17 +288,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
cbnz(Size::i64Bit, Reg::r29, &Label);
|
||||
(void)cbnz(Size::i64Bit, Reg::r29, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0xb5fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
cbnz(Size::i64Bit, Reg::r29, &Label);
|
||||
Bind(&Label);
|
||||
(void)cbnz(Size::i64Bit, Reg::r29, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb500003d);
|
||||
@@ -307,17 +307,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Compare and branch") {
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate") {
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
tbz(Reg::r29, 0, &Label);
|
||||
(void)tbz(Reg::r29, 0, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x3607fffd);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
tbz(Reg::r29, 0, &Label);
|
||||
Bind(&Label);
|
||||
(void)tbz(Reg::r29, 0, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x3600003d);
|
||||
@@ -325,17 +325,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
tbz(Reg::r29, 0, &Label);
|
||||
(void)tbz(Reg::r29, 0, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x3607fffd);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
tbz(Reg::r29, 0, &Label);
|
||||
Bind(&Label);
|
||||
(void)tbz(Reg::r29, 0, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x3600003d);
|
||||
@@ -343,17 +343,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
tbz(Reg::r29, 63, &Label);
|
||||
(void)tbz(Reg::r29, 63, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0xb6fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
tbz(Reg::r29, 63, &Label);
|
||||
Bind(&Label);
|
||||
(void)tbz(Reg::r29, 63, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb6f8003d);
|
||||
@@ -361,17 +361,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
tbz(Reg::r29, 63, &Label);
|
||||
(void)tbz(Reg::r29, 63, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0xb6fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
tbz(Reg::r29, 63, &Label);
|
||||
Bind(&Label);
|
||||
(void)tbz(Reg::r29, 63, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb6f8003d);
|
||||
@@ -379,17 +379,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
tbnz(Reg::r29, 0, &Label);
|
||||
(void)tbnz(Reg::r29, 0, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x3707fffd);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
tbnz(Reg::r29, 0, &Label);
|
||||
Bind(&Label);
|
||||
(void)tbnz(Reg::r29, 0, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x3700003d);
|
||||
@@ -397,17 +397,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
tbnz(Reg::r29, 0, &Label);
|
||||
(void)tbnz(Reg::r29, 0, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0x3707fffd);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
tbnz(Reg::r29, 0, &Label);
|
||||
Bind(&Label);
|
||||
(void)tbnz(Reg::r29, 0, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x3700003d);
|
||||
@@ -415,17 +415,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
tbnz(Reg::r29, 63, &Label);
|
||||
(void)tbnz(Reg::r29, 63, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0xb7fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
ForwardLabel Label;
|
||||
tbnz(Reg::r29, 63, &Label);
|
||||
Bind(&Label);
|
||||
(void)tbnz(Reg::r29, 63, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb7f8003d);
|
||||
@@ -433,17 +433,17 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Test and branch immediate")
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
tbnz(Reg::r29, 63, &Label);
|
||||
(void)tbnz(Reg::r29, 63, &Label);
|
||||
|
||||
CHECK(DisassembleEncoding(1) == 0xb7fffffd);
|
||||
}
|
||||
|
||||
{
|
||||
BiDirectionalLabel Label;
|
||||
tbnz(Reg::r29, 63, &Label);
|
||||
Bind(&Label);
|
||||
(void)tbnz(Reg::r29, 63, &Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xb7f8003d);
|
||||
|
||||
@@ -1323,7 +1323,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: LDAPR/STLR unscaled imme
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal") {
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
ldr(WReg::w30, &Label);
|
||||
|
||||
@@ -1332,7 +1332,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
ldr(SReg::s30, &Label);
|
||||
|
||||
@@ -1341,7 +1341,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
ldr(XReg::x30, &Label);
|
||||
|
||||
@@ -1350,7 +1350,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
ldr(DReg::d30, &Label);
|
||||
|
||||
@@ -1359,7 +1359,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
ldrsw(XReg::x30, &Label);
|
||||
|
||||
@@ -1368,7 +1368,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
ldr(QReg::q30, &Label);
|
||||
|
||||
@@ -1377,7 +1377,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
|
||||
{
|
||||
BackwardLabel Label;
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
prfm(Prefetch::PLDL1KEEP, &Label);
|
||||
|
||||
@@ -1387,7 +1387,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
{
|
||||
ForwardLabel Label;
|
||||
ldr(WReg::w30, &Label);
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x1800003e);
|
||||
@@ -1396,7 +1396,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
{
|
||||
ForwardLabel Label;
|
||||
ldr(SReg::s30, &Label);
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x1c00003e);
|
||||
@@ -1405,7 +1405,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
{
|
||||
ForwardLabel Label;
|
||||
ldr(XReg::x30, &Label);
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x5800003e);
|
||||
@@ -1414,7 +1414,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
{
|
||||
ForwardLabel Label;
|
||||
ldr(DReg::d30, &Label);
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x5c00003e);
|
||||
@@ -1423,7 +1423,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
{
|
||||
ForwardLabel Label;
|
||||
ldrsw(XReg::x30, &Label);
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x9800003e);
|
||||
@@ -1432,7 +1432,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
{
|
||||
ForwardLabel Label;
|
||||
ldr(QReg::q30, &Label);
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0x9c00003e);
|
||||
@@ -1441,7 +1441,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Load register literal")
|
||||
{
|
||||
ForwardLabel Label;
|
||||
prfm(Prefetch::PLDL1KEEP, &Label);
|
||||
Bind(&Label);
|
||||
(void)Bind(&Label);
|
||||
dc32(0);
|
||||
|
||||
CHECK(DisassembleEncoding(0) == 0xd8000020);
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
#!/usr/bin/python3
|
||||
import xxhash
|
||||
import hashlib
|
||||
import sys
|
||||
import os
|
||||
import shutil
|
||||
@@ -188,5 +187,5 @@ def main():
|
||||
return 0
|
||||
|
||||
if __name__ == "__main__":
|
||||
# execute only if run as a script
|
||||
# execute only if run as a script
|
||||
sys.exit(main())
|
||||
@@ -1,8 +1,5 @@
|
||||
#!/usr/bin/python3
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import platform
|
||||
|
||||
def ListContainsRequired(Features, RequiredFeatures):
|
||||
|
||||
@@ -4,7 +4,7 @@ from clang.cindex import CursorKind
|
||||
from clang.cindex import TypeKind
|
||||
from clang.cindex import TranslationUnit
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
import subprocess
|
||||
import logging
|
||||
logger = logging.getLogger()
|
||||
@@ -548,5 +548,5 @@ def main():
|
||||
PrintFunctionDecls()
|
||||
|
||||
if __name__ == "__main__":
|
||||
# execute only if run as a script
|
||||
# execute only if run as a script
|
||||
sys.exit(main())
|
||||
@@ -9,19 +9,19 @@ for fileid in ~/.fex-emu/aotir/*.path; do
|
||||
else
|
||||
args="$args --no-abilocalflags"
|
||||
fi
|
||||
|
||||
|
||||
if [ "${fileid: -7 : 1}" == "T" ]; then
|
||||
args="$args --tsoenabled"
|
||||
else
|
||||
args="$args --no-tsoenabled"
|
||||
fi
|
||||
|
||||
|
||||
if [ "${fileid: -8 : 1}" == "S" ]; then
|
||||
args="$args --smc=full"
|
||||
else
|
||||
args="$args --smc=mman"
|
||||
fi
|
||||
|
||||
|
||||
if [ -f "${fileid%.path}.aotir" ]; then
|
||||
echo "`basename $fileid` has already been generated"
|
||||
else
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/python3
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
import math
|
||||
import sys
|
||||
import logging
|
||||
@@ -282,5 +282,5 @@ def main():
|
||||
ExportCommonSyscallDefines()
|
||||
|
||||
if __name__ == "__main__":
|
||||
# execute only if run as a script
|
||||
# execute only if run as a script
|
||||
sys.exit(main())
|
||||
+27
-26
@@ -2,7 +2,6 @@
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import re
|
||||
|
||||
_Arch = None
|
||||
@@ -81,11 +80,7 @@ def IsSupportedDistro():
|
||||
|
||||
# We only support Ubuntu
|
||||
if Distro[0] == "ubuntu":
|
||||
# We only support what is available in ppa:fex-emu/fex
|
||||
return Distro[1] == "22.04" or \
|
||||
Distro[1] == "24.04" or \
|
||||
Distro[1] == "24.10" or \
|
||||
Distro[1] == "25.04"
|
||||
return Distro[1] in {"22.04", "24.04", "24.10", "25.04", "25.10"}
|
||||
|
||||
return False
|
||||
|
||||
@@ -207,10 +202,26 @@ def UpdatePPA():
|
||||
|
||||
return DidUpdate
|
||||
|
||||
def CheckAndInstallPackageUpdates():
|
||||
PackagesToInstall = GetPackagesToInstall()
|
||||
def InstallPackages(PackagesToInstall):
|
||||
DidInstall = False
|
||||
try:
|
||||
CmdResult = subprocess.call(["sudo", "apt-get", "-y", "install"] + PackagesToInstall)
|
||||
DidInstall = CmdResult == 0
|
||||
except KeyboardInterrupt:
|
||||
print ("Keyboard interrupt")
|
||||
DidInstall = False
|
||||
pass
|
||||
|
||||
if DidInstall:
|
||||
print("Packages updated")
|
||||
else:
|
||||
print("Packages failed to update")
|
||||
|
||||
return DidInstall
|
||||
|
||||
def CheckAndInstallPackageUpdates(PackagesToInstall, InstallIfNotFound=False):
|
||||
for Package in PackagesToInstall[:]:
|
||||
UpgradableStatus = subprocess.check_output(["apt", "list", "--upgradable", Package]).decode("utf-8")
|
||||
UpgradableStatus = subprocess.check_output(["apt", "list", "--upgradable", Package], stderr=None).decode("utf-8")
|
||||
Found = False
|
||||
for Line in UpgradableStatus.split("\n"):
|
||||
# If the package exists to be upgraded then it will appear in this list
|
||||
@@ -225,28 +236,14 @@ def CheckAndInstallPackageUpdates():
|
||||
if Package in Line and "upgradable" in Line:
|
||||
Found = True
|
||||
|
||||
if Found == False:
|
||||
if InstallIfNotFound == False and Found == False:
|
||||
PackagesToInstall.remove(Package)
|
||||
|
||||
if len(PackagesToInstall) > 0:
|
||||
print ("Found updates for packages: {}".format(PackagesToInstall))
|
||||
print ("This bit may ask for your password")
|
||||
|
||||
DidInstall = False
|
||||
try:
|
||||
CmdResult = subprocess.call(["sudo", "apt-get", "-y", "install"] + PackagesToInstall)
|
||||
DidInstall = CmdResult == 0
|
||||
except KeyboardInterrupt:
|
||||
print ("Keyboard interrupt")
|
||||
DidInstall = False
|
||||
pass
|
||||
|
||||
if DidInstall:
|
||||
print("Packages updated")
|
||||
else:
|
||||
print("Packages failed to update")
|
||||
|
||||
return DidInstall
|
||||
return InstallPackages(PackagesToInstall)
|
||||
|
||||
return True
|
||||
|
||||
@@ -358,10 +355,14 @@ def main():
|
||||
if not UpdatePPA():
|
||||
print ("apt sources failed to update. Not continuing")
|
||||
ExitWithStatus(-1)
|
||||
if not CheckAndInstallPackageUpdates():
|
||||
if not CheckAndInstallPackageUpdates(GetPackagesToInstall()):
|
||||
print ("apt packages failed to update. Not continuing")
|
||||
ExitWithStatus(-1)
|
||||
else:
|
||||
if not CheckAndInstallPackageUpdates(["software-properties-common"], True):
|
||||
print ("software-properties-common package failed to update. Not continuing")
|
||||
ExitWithStatus(-1)
|
||||
|
||||
if not InstallPPA():
|
||||
print ("PPA failed to install. Not continuing")
|
||||
ExitWithStatus(-1)
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/python3
|
||||
import base64
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
from enum import Flag
|
||||
import json
|
||||
import struct
|
||||
@@ -254,5 +254,5 @@ def main():
|
||||
return 0
|
||||
|
||||
if __name__ == "__main__":
|
||||
# execute only if run as a script
|
||||
# execute only if run as a script
|
||||
sys.exit(main())
|
||||
@@ -4,7 +4,7 @@ from clang.cindex import CursorKind
|
||||
from clang.cindex import TypeKind
|
||||
from clang.cindex import TranslationUnit
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
import subprocess
|
||||
import logging
|
||||
logger = logging.getLogger()
|
||||
@@ -774,5 +774,5 @@ def main():
|
||||
return Result
|
||||
|
||||
if __name__ == "__main__":
|
||||
# execute only if run as a script
|
||||
# execute only if run as a script
|
||||
sys.exit(main())
|
||||
@@ -1,13 +1,9 @@
|
||||
#!/usr/bin/python3
|
||||
from enum import Flag
|
||||
import json
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
import glob
|
||||
from threading import Thread
|
||||
import subprocess
|
||||
import time
|
||||
import multiprocessing
|
||||
from shutil import which
|
||||
|
||||
|
||||
@@ -76,6 +76,6 @@ def main():
|
||||
return 0
|
||||
|
||||
if __name__ == "__main__":
|
||||
# execute only if run as a script
|
||||
# execute only if run as a script
|
||||
sys.exit(main())
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
#!/usr/bin/python3
|
||||
import re
|
||||
import sys
|
||||
import subprocess
|
||||
try:
|
||||
from packaging.version import Version as version_check
|
||||
except:
|
||||
@@ -81,6 +80,8 @@ BigCoreIDs = {
|
||||
[ ["apple-a13", "0.0"], # If we aren't on 12.0+
|
||||
["apple-a14", "12.0"], # Only exists in 12.0+
|
||||
],
|
||||
# QEmu HVF 10.2+
|
||||
tuple([0x61, 0]): "apple-a13", # Can't determine variant, choose lowest.
|
||||
}
|
||||
|
||||
LittleCoreIDs = {
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
#!/bin/env python3
|
||||
import sys
|
||||
|
||||
import fileinput
|
||||
import re
|
||||
|
||||
# Handles the following formats:
|
||||
|
||||
@@ -4,8 +4,8 @@ import os
|
||||
import sys
|
||||
import subprocess
|
||||
|
||||
# Check if FEX indicates support for AVX
|
||||
def DoesFEXSupportAVX(mode):
|
||||
# Check if FEX indicates support for AVX
|
||||
fex_interpreter_path = os.path.dirname(sys.argv[7]) + "/FEX"
|
||||
|
||||
args = list()
|
||||
@@ -22,8 +22,8 @@ def DoesFEXSupportAVX(mode):
|
||||
return 'avx' in flags and 'avx2' in flags
|
||||
return False
|
||||
|
||||
# Check if the test itself requires AVX
|
||||
def TestRequiresAVXSupport():
|
||||
# Check if the test itself requires AVX
|
||||
exe_path = sys.argv[len(sys.argv) - 1]
|
||||
json_path = os.path.dirname(os.path.dirname(exe_path)) + '/requirements/' + os.path.basename(exe_path) + '.json'
|
||||
|
||||
|
||||
Loaded 100 of 717 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user