mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-08 00:00:16 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9681559d56 | ||
|
|
b478e4845f | ||
|
|
1fa5104076 | ||
|
|
ce65f5376f | ||
|
|
0695249fc8 | ||
|
|
51144c99a7 | ||
|
|
251398a7cb | ||
|
|
2e6a7f869c | ||
|
|
f308162334 | ||
|
|
db4867839c | ||
|
|
73ffff7d22 | ||
|
|
dc48a4f73c | ||
|
|
efbccccdc0 | ||
|
|
3e7cd88dcc | ||
|
|
12cfe8fc37 | ||
|
|
c6d2ce043f | ||
|
|
5c34c574c8 | ||
|
|
d3cfdcb431 | ||
|
|
4018d23c39 | ||
|
|
81d4e8fe9d | ||
|
|
34b48c4069 | ||
|
|
01a3ab6ca7 | ||
|
|
476c242d7f | ||
|
|
ae3fa6a836 | ||
|
|
ba93bdd66d | ||
|
|
1fa0b37fac | ||
|
|
5b4a5969cc | ||
|
|
941a7934ef | ||
|
|
474439ab4b | ||
|
|
4428aea5ee | ||
|
|
e9a9cc5bc3 | ||
|
|
2291c5b230 | ||
|
|
f78e194cf7 | ||
|
|
b77ddcf1a7 | ||
|
|
3ef677537d | ||
|
|
1df1265ed1 | ||
|
|
f0854a16fe | ||
|
|
6bd476fb03 | ||
|
|
b9c0af7c3b | ||
|
|
5149ebc70e | ||
|
|
e92a6a5803 | ||
|
|
6da963a695 | ||
|
|
2a74489858 | ||
|
|
65a436ca98 | ||
|
|
bc533c8050 | ||
|
|
fd6cea4698 | ||
|
|
a1aa1658ec | ||
|
|
5c4c468d13 | ||
|
|
69ef1658cc | ||
|
|
8c72aa76a0 | ||
|
|
928a932a43 | ||
|
|
83601055dc | ||
|
|
68480f6e43 | ||
|
|
42291540ab | ||
|
|
194eb69838 | ||
|
|
c1d27fa453 | ||
|
|
9d5f7caa79 | ||
|
|
a1d78dceb0 | ||
|
|
2bcf435e0a | ||
|
|
30e853305d | ||
|
|
600f4bb2b6 | ||
|
|
5868814c91 | ||
|
|
e862f8f86c | ||
|
|
85c1ecd035 | ||
|
|
68ad448672 | ||
|
|
c18fb3cb78 | ||
|
|
53702f989c | ||
|
|
c547b1bec3 | ||
|
|
740350c8ea | ||
|
|
9f9b20eac0 | ||
|
|
c67ffb82a8 | ||
|
|
1ea24f3d6c | ||
|
|
226bd51afe | ||
|
|
293568be36 | ||
|
|
152fe81d16 | ||
|
|
494dd64c50 | ||
|
|
56de0d1ab4 | ||
|
|
73c1f4cc54 | ||
|
|
6a6a82385e | ||
|
|
fc8ef0e723 | ||
|
|
de11c05d2a | ||
|
|
addbc8cad8 | ||
|
|
63e37b7cbb | ||
|
|
9ee329034f | ||
|
|
f3e904207b | ||
|
|
e4ae6ce635 | ||
|
|
f7d76255ad | ||
|
|
ea45f9c694 | ||
|
|
1d449c0f58 | ||
|
|
9ecc991043 | ||
|
|
4ef834859c | ||
|
|
fa082bc5c4 | ||
|
|
67caab026a | ||
|
|
91c55facb2 | ||
|
|
321d4d84d7 | ||
|
|
22faa58e0b | ||
|
|
ebd559f662 | ||
|
|
d27c9d3f98 | ||
|
|
fbef482265 | ||
|
|
a57926ac57 | ||
|
|
70a7137e62 | ||
|
|
2d3a08a362 | ||
|
|
cc02edb3f6 | ||
|
|
f894cd90f3 | ||
|
|
24675969cc | ||
|
|
a0cba1194e | ||
|
|
0952fa95f4 | ||
|
|
6177ab957b | ||
|
|
5c1300a2b9 | ||
|
|
af9dd0827a | ||
|
|
dc0162122f | ||
|
|
ae491fb15b | ||
|
|
957c1fc420 | ||
|
|
86acfb35aa | ||
|
|
e27d12ee5e | ||
|
|
63d52c2a1a | ||
|
|
9d4a71b57a | ||
|
|
3462dc3e14 | ||
|
|
d21351e66e | ||
|
|
3204d20335 | ||
|
|
a519489d80 | ||
|
|
5558c3a35a | ||
|
|
58d9755314 | ||
|
|
498ba0a384 | ||
|
|
5a5477e895 | ||
|
|
a17d7ce6ba | ||
|
|
afc7248912 | ||
|
|
6bb578fea8 | ||
|
|
bed5f293dc |
No files matched your search
+14
-13
@@ -49,6 +49,7 @@ jobs:
|
||||
run: cmake --build build --target asm_files 32bit_asm_files JemallocLibs Catch2 vixl cephes_128bit
|
||||
|
||||
- name: Build
|
||||
id: build
|
||||
run: cmake --build build
|
||||
|
||||
- name: Install
|
||||
@@ -56,40 +57,40 @@ jobs:
|
||||
|
||||
# GCC tests
|
||||
- name: GCC64 Target Tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: gcc_target_tests_64
|
||||
|
||||
- name: GCC32 Target Tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: gcc_target_tests_32
|
||||
|
||||
# API tests
|
||||
- name: API Tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: api_tests
|
||||
|
||||
- name: FEXCore API Tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: fexcore_apitests
|
||||
|
||||
# ARM emission tests
|
||||
- name: ARM Emitter Tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: emitter_tests
|
||||
|
||||
# Linux tests
|
||||
- name: FEX Linux Tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: fex_linux_tests_all
|
||||
@@ -98,13 +99,13 @@ jobs:
|
||||
|
||||
# Thunking
|
||||
- name: Thunkgen tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: thunkgen_tests
|
||||
|
||||
- name: Test GL No-Thunks
|
||||
if: ${{ always() && matrix.arch[1] == 'x64' }}
|
||||
if: ${{ steps.build.outcome == 'success' && matrix.arch[1] == 'x64' }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: thunk_functional_tests_nothunks
|
||||
@@ -112,7 +113,7 @@ jobs:
|
||||
DISPLAY: ':0'
|
||||
|
||||
- name: Test GL Thunks
|
||||
if: ${{ always() && matrix.arch[1] == 'x64' }}
|
||||
if: ${{ steps.build.outcome == 'success' && matrix.arch[1] == 'x64' }}
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: thunk_functional_tests_thunks
|
||||
@@ -121,28 +122,28 @@ jobs:
|
||||
|
||||
# ASM tests
|
||||
- name: ASM Tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: asm_tests
|
||||
|
||||
# POSIX tests
|
||||
- name: POSIX Tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: posix_tests
|
||||
|
||||
# GVisor tests
|
||||
- name: GVisor Tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: gvisor_tests
|
||||
|
||||
# Struct verifier tests
|
||||
- name: Struct verifier tests
|
||||
if: ${{ always() }}
|
||||
if: steps.build.outcome == 'success'
|
||||
uses: ./.github/workflows/test
|
||||
with:
|
||||
target: struct_verifier
|
||||
|
||||
+40
-1
@@ -1,3 +1,17 @@
|
||||
spec:
|
||||
inputs:
|
||||
PROMOTE_BRANCH:
|
||||
description: "Branch to promote the build to. Empty means no promotion."
|
||||
default: "bleeding-edge"
|
||||
|
||||
---
|
||||
|
||||
workflow:
|
||||
rules:
|
||||
- when: always
|
||||
variables:
|
||||
PROMOTE_BRANCH: $[[ inputs.PROMOTE_BRANCH ]]
|
||||
|
||||
variables:
|
||||
DEBIAN_FRONTEND: noninteractive
|
||||
GIT_SUBMODULE_STRATEGY: recursive
|
||||
@@ -5,7 +19,8 @@ variables:
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
|
||||
aarch64:
|
||||
build:
|
||||
stage: build
|
||||
image: registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306
|
||||
tags:
|
||||
- docker
|
||||
@@ -30,3 +45,27 @@ aarch64:
|
||||
untracked: false
|
||||
paths:
|
||||
- install/
|
||||
|
||||
promote:
|
||||
stage: deploy
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
image: registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306
|
||||
tags:
|
||||
- docker
|
||||
- linux
|
||||
- arm64
|
||||
- aarch64
|
||||
rules:
|
||||
- if: '$PROMOTE_BRANCH'
|
||||
before_script:
|
||||
- apt-get -y update
|
||||
- apt-get install -y tmux curl
|
||||
script:
|
||||
# comment out to debug: SSH in via GCP, go down the container and attach to the session (with `tmux attach -t debug`)
|
||||
# - tmux new-session -d -s debug
|
||||
# - while tmux has-session -t debug 2>/dev/null; do sleep 1; done
|
||||
|
||||
# ref controls which fex-depot code runs the pipeline, while VERSION_PARAM controls which fex branch's artifacts that pipeline downloads.
|
||||
- >
|
||||
curl --fail --location --request POST --form token=${FEX_DEPOT_TRIGGER_TOKEN} --form ref=master --form "variables[PROMOTE_BRANCH]=${PROMOTE_BRANCH}" --form "variables[VERSION_PARAM]=${CI_COMMIT_REF_NAME}" "${CI_API_V4_URL}/projects/fex%2Ffex-depot/trigger/pipeline"
|
||||
+13
-1
@@ -172,6 +172,13 @@ set(TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
set(OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version")
|
||||
set(OVERRIDE_HASH "detect" CACHE STRING "Override the FEX git hash")
|
||||
|
||||
get_property(IS_MULTI_CONFIG GLOBAL PROPERTY GENERATOR_IS_MULTI_CONFIG)
|
||||
if (NOT IS_MULTI_CONFIG AND NOT CMAKE_BUILD_TYPE)
|
||||
set(CMAKE_BUILD_TYPE Release
|
||||
CACHE STRING "Choose the type of build." FORCE)
|
||||
message(STATUS "No build type set, defaulting to a Release build")
|
||||
endif()
|
||||
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
|
||||
if (CMAKE_BUILD_TYPE MATCHES "DEBUG")
|
||||
set(ENABLE_ASSERTIONS TRUE)
|
||||
@@ -446,6 +453,11 @@ if(ENUM_ENUM_WARNING)
|
||||
add_compile_options(-Wno-deprecated-enum-enum-conversion)
|
||||
endif()
|
||||
|
||||
# GCC enables -Wchanges-meaning by default and treats some cases as an error
|
||||
if(CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
add_compile_options(-Wno-error=changes-meaning)
|
||||
endif()
|
||||
|
||||
if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
add_compile_options(-Werror)
|
||||
if (NOT ENABLE_STRICT_WERROR)
|
||||
@@ -681,6 +693,6 @@ if (BUILD_THUNKS)
|
||||
add_dependencies(uninstall uninstall_guest-libs-32)
|
||||
endif()
|
||||
|
||||
if (BUILD_STEAM_SUPPORT)
|
||||
if (NOT MINGW AND BUILD_STEAM_SUPPORT)
|
||||
add_subdirectory(Source/Steam/)
|
||||
endif()
|
||||
@@ -311,7 +311,7 @@ class ExtendedMemOperand final {
|
||||
public:
|
||||
ExtendedMemOperand(XRegister rn, XRegister rm = XReg::zr, ExtendedType Option = ExtendedType::LSL_64, uint32_t Shift = 0)
|
||||
: rn {rn}
|
||||
, MetaType {.ExtendedType {
|
||||
, MetaType {.Extended {
|
||||
.Header = {.MemType = TYPE_EXTENDED},
|
||||
.rm = rm,
|
||||
.Option = Option,
|
||||
@@ -340,7 +340,7 @@ public:
|
||||
Register rm;
|
||||
ExtendedType Option;
|
||||
uint32_t Shift;
|
||||
} ExtendedType;
|
||||
} Extended;
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
IndexType Index;
|
||||
|
||||
@@ -3627,8 +3627,8 @@ public:
|
||||
|
||||
void strb(ARMEmitter::Register rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3650,8 +3650,8 @@ public:
|
||||
}
|
||||
void ldrb(ARMEmitter::Register rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3673,8 +3673,8 @@ public:
|
||||
}
|
||||
void ldrsb(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3696,8 +3696,8 @@ public:
|
||||
}
|
||||
void ldrsb(ARMEmitter::WRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3719,8 +3719,8 @@ public:
|
||||
}
|
||||
void strh(ARMEmitter::Register rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3742,8 +3742,8 @@ public:
|
||||
}
|
||||
void ldrh(ARMEmitter::Register rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3765,8 +3765,8 @@ public:
|
||||
}
|
||||
void ldrsh(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3788,8 +3788,8 @@ public:
|
||||
}
|
||||
void ldrsh(ARMEmitter::WRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3811,8 +3811,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::WRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3834,8 +3834,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::WRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3857,8 +3857,8 @@ public:
|
||||
}
|
||||
void ldrsw(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsw(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrsw(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrsw(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3880,8 +3880,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3903,8 +3903,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::XRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3926,8 +3926,8 @@ public:
|
||||
}
|
||||
void prfm(ARMEmitter::Prefetch prfop, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
prfm(prfop, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
prfm(prfop, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
prfm(prfop, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3946,9 +3946,9 @@ public:
|
||||
|
||||
void strb(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_A_FMT(MemSrc.MetaType.ExtendedType.Shift == false, "Can't shift byte");
|
||||
strb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_A_FMT(MemSrc.MetaType.Extended.Shift == false, "Can't shift byte");
|
||||
strb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3970,9 +3970,9 @@ public:
|
||||
}
|
||||
void ldrb(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_A_FMT(MemSrc.MetaType.ExtendedType.Shift == false, "Can't shift byte");
|
||||
ldrb(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
LOGMAN_THROW_A_FMT(MemSrc.MetaType.Extended.Shift == false, "Can't shift byte");
|
||||
ldrb(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrb(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -3994,8 +3994,8 @@ public:
|
||||
}
|
||||
void strh(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
strh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
strh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4017,8 +4017,8 @@ public:
|
||||
}
|
||||
void ldrh(ARMEmitter::VRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrh(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldrh(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldrh(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4040,8 +4040,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::SRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4063,8 +4063,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::SRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4086,8 +4086,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::DRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4109,8 +4109,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::DRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4132,8 +4132,8 @@ public:
|
||||
}
|
||||
void str(ARMEmitter::QRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
str(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
str(rt, MemSrc.rn);
|
||||
} else {
|
||||
@@ -4155,8 +4155,8 @@ public:
|
||||
}
|
||||
void ldr(ARMEmitter::QRegister rt, ARMEmitter::ExtendedMemOperand MemSrc) {
|
||||
if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED &&
|
||||
MemSrc.MetaType.ExtendedType.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.ExtendedType.rm, MemSrc.MetaType.ExtendedType.Option, MemSrc.MetaType.ExtendedType.Shift);
|
||||
MemSrc.MetaType.Extended.rm.Idx() != ARMEmitter::Reg::r31.Idx()) {
|
||||
ldr(rt, MemSrc.rn, MemSrc.MetaType.Extended.rm, MemSrc.MetaType.Extended.Option, MemSrc.MetaType.Extended.Shift);
|
||||
} else if (MemSrc.MetaType.Header.MemType == ARMEmitter::ExtendedMemOperand::Type::TYPE_EXTENDED) {
|
||||
ldr(rt, MemSrc.rn);
|
||||
} else {
|
||||
|
||||
+81
-30
@@ -4,29 +4,34 @@
|
||||
#
|
||||
# pip-compile --generate-hashes --output-file=requirements_formatting.txt --strip-extras requirements_formatting.txt.in
|
||||
#
|
||||
black==25.1.0 \
|
||||
--hash=sha256:030b9759066a4ee5e5aca28c3c77f9c64789cdd4de8ac1df642c40b708be6171 \
|
||||
--hash=sha256:055e59b198df7ac0b7efca5ad7ff2516bca343276c466be72eb04a3bcc1f82d7 \
|
||||
--hash=sha256:0e519ecf93120f34243e6b0054db49c00a35f84f195d5bce7e9f5cfc578fc2da \
|
||||
--hash=sha256:172b1dbff09f86ce6f4eb8edf9dede08b1fce58ba194c87d7a4f1a5aa2f5b3c2 \
|
||||
--hash=sha256:1e2978f6df243b155ef5fa7e558a43037c3079093ed5d10fd84c43900f2d8ecc \
|
||||
--hash=sha256:33496d5cd1222ad73391352b4ae8da15253c5de89b93a80b3e2c8d9a19ec2666 \
|
||||
--hash=sha256:3b48735872ec535027d979e8dcb20bf4f70b5ac75a8ea99f127c106a7d7aba9f \
|
||||
--hash=sha256:4b60580e829091e6f9238c848ea6750efed72140b91b048770b64e74fe04908b \
|
||||
--hash=sha256:759e7ec1e050a15f89b770cefbf91ebee8917aac5c20483bc2d80a6c3a04df32 \
|
||||
--hash=sha256:8f0b18a02996a836cc9c9c78e5babec10930862827b1b724ddfe98ccf2f2fe4f \
|
||||
--hash=sha256:95e8176dae143ba9097f351d174fdaf0ccd29efb414b362ae3fd72bf0f710717 \
|
||||
--hash=sha256:96c1c7cd856bba8e20094e36e0f948718dc688dba4a9d78c3adde52b9e6c2299 \
|
||||
--hash=sha256:a1ee0a0c330f7b5130ce0caed9936a904793576ef4d2b98c40835d6a65afa6a0 \
|
||||
--hash=sha256:a22f402b410566e2d1c950708c77ebf5ebd5d0d88a6a2e87c86d9fb48afa0d18 \
|
||||
--hash=sha256:a39337598244de4bae26475f77dda852ea00a93bd4c728e09eacd827ec929df0 \
|
||||
--hash=sha256:afebb7098bfbc70037a053b91ae8437c3857482d3a690fefc03e9ff7aa9a5fd3 \
|
||||
--hash=sha256:bacabb307dca5ebaf9c118d2d2f6903da0d62c9faa82bd21a33eecc319559355 \
|
||||
--hash=sha256:bce2e264d59c91e52d8000d507eb20a9aca4a778731a08cfff7e5ac4a4bb7096 \
|
||||
--hash=sha256:d9e6827d563a2c820772b32ce8a42828dc6790f095f441beef18f96aa6f8294e \
|
||||
--hash=sha256:db8ea9917d6f8fc62abd90d944920d95e73c83a5ee3383493e35d271aca872e9 \
|
||||
--hash=sha256:ea0213189960bda9cf99be5b8c8ce66bb054af5e9e861249cd23471bd7b0b3ba \
|
||||
--hash=sha256:f3df5f1bf91d36002b0a75389ca8663510cf0531cca8aa5c1ef695b46d98655f
|
||||
black==26.3.1 \
|
||||
--hash=sha256:0126ae5b7c09957da2bdbd91a9ba1207453feada9e9fe51992848658c6c8e01c \
|
||||
--hash=sha256:0f76ff19ec5297dd8e66eb64deda23631e642c9393ab592826fd4bdc97a4bce7 \
|
||||
--hash=sha256:28ef38aee69e4b12fda8dba75e21f9b4f979b490c8ac0baa7cb505369ac9e1ff \
|
||||
--hash=sha256:2bd5aa94fc267d38bb21a70d7410a89f1a1d318841855f698746f8e7f51acd1b \
|
||||
--hash=sha256:2c50f5063a9641c7eed7795014ba37b0f5fa227f3d408b968936e24bc0566b07 \
|
||||
--hash=sha256:2d6bfaf7fd0993b420bed691f20f9492d53ce9a2bcccea4b797d34e947318a78 \
|
||||
--hash=sha256:41cd2012d35b47d589cb8a16faf8a32ef7a336f56356babd9fcf70939ad1897f \
|
||||
--hash=sha256:474c27574d6d7037c1bc875a81d9be0a9a4f9ee95e62800dab3cfaadbf75acd5 \
|
||||
--hash=sha256:5602bdb96d52d2d0672f24f6ffe5218795736dd34807fd0fd55ccd6bf206168b \
|
||||
--hash=sha256:5e9d0d86df21f2e1677cc4bd090cd0e446278bcbbe49bf3659c308c3e402843e \
|
||||
--hash=sha256:5ed0ca58586c8d9a487352a96b15272b7fa55d139fc8496b519e78023a8dab0a \
|
||||
--hash=sha256:6c54a4a82e291a1fee5137371ab488866b7c86a3305af4026bdd4dc78642e1ac \
|
||||
--hash=sha256:6e131579c243c98f35bce64a7e08e87fb2d610544754675d4a0e73a070a5aa3a \
|
||||
--hash=sha256:855822d90f884905362f602880ed8b5df1b7e3ee7d0db2502d4388a954cc8c54 \
|
||||
--hash=sha256:86a8b5035fce64f5dcd1b794cf8ec4d31fe458cf6ce3986a30deb434df82a1d2 \
|
||||
--hash=sha256:8a33d657f3276328ce00e4d37fe70361e1ec7614da5d7b6e78de5426cb56332f \
|
||||
--hash=sha256:92c0ec1f2cc149551a2b7b47efc32c866406b6891b0ee4625e95967c8f4acfb1 \
|
||||
--hash=sha256:9a5e9f45e5d5e1c5b5c29b3bd4265dcc90e8b92cf4534520896ed77f791f4da5 \
|
||||
--hash=sha256:afc622538b430aa4c8c853f7f63bc582b3b8030fd8c80b70fb5fa5b834e575c2 \
|
||||
--hash=sha256:b07fc0dab849d24a80a29cfab8d8a19187d1c4685d8a5e6385a5ce323c1f015f \
|
||||
--hash=sha256:b5e6f89631eb88a7302d416594a32faeee9fb8fb848290da9d0a5f2903519fc1 \
|
||||
--hash=sha256:bf9bf162ed91a26f1adba8efda0b573bc6924ec1408a52cc6f82cb73ec2b142c \
|
||||
--hash=sha256:c7e72339f841b5a237ff14f7d3880ddd0fc7f98a1199e8c4327f9a4f478c1839 \
|
||||
--hash=sha256:ddb113db38838eb9f043623ba274cfaf7d51d5b0c22ecb30afe58b1bb8322983 \
|
||||
--hash=sha256:dfdd51fc3e64ea4f35873d1b3fb25326773d55d2329ff8449139ebaad7357efb \
|
||||
--hash=sha256:f1cd08e99d2f9317292a311dfe578fd2a24b15dbce97792f9c4d752275c1fa56 \
|
||||
--hash=sha256:f89f2ab047c76a9c03f78d0d66ca519e389519902fa27e7a91117ef7611c0568
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# darker
|
||||
@@ -290,9 +295,9 @@ packaging==23.1 \
|
||||
--hash=sha256:994793af429502c4ea2ebf6bf664629d07c1a9fe974af92966e4b8d2df7edc61 \
|
||||
--hash=sha256:a392980d2b6cffa644431898be54b0045151319d1e7ec34f0cfed48767dd334f
|
||||
# via black
|
||||
pathspec==0.11.2 \
|
||||
--hash=sha256:1d6ed233af05e679efb96b1851550ea95bbb64b7c490b0f5aa52996c11e92a20 \
|
||||
--hash=sha256:e0d8d0ac2f12da61956eb2306b69f9469b42f4deb0f3cb6ed47b9cce9996ced3
|
||||
pathspec==1.0.4 \
|
||||
--hash=sha256:0210e2ae8a21a9137c0d470578cb0e595af87edaa6ebf12ff176f14a02e0e645 \
|
||||
--hash=sha256:fb6ae2fd4e7c921a165808a552060e722767cfa526f99ca5156ed2ce45a5c723
|
||||
# via black
|
||||
platformdirs==3.10.0 \
|
||||
--hash=sha256:b45696dab2d7cc691a3226759c0d3b00c47c8b6e293d96f6436f733303f77f6d \
|
||||
@@ -306,10 +311,12 @@ pygithub==2.6.1 \
|
||||
--hash=sha256:6f2fa6d076ccae475f9fc392cc6cdbd54db985d4f69b8833a28397de75ed6ca3 \
|
||||
--hash=sha256:b5c035392991cca63959e9453286b41b54d83bf2de2daa7d7ff7e4312cebf3bf
|
||||
# via -r requirements_formatting.txt.in
|
||||
pyjwt==2.8.0 \
|
||||
--hash=sha256:57e28d156e3d5c10088e0c68abb90bfac3df82b40a71bd0daa20c65ccd5c23de \
|
||||
--hash=sha256:59127c392cc44c2da5bb3192169a91f429924e17aff6534d70fdc02ab3e04320
|
||||
# via pygithub
|
||||
pyjwt==2.12.1 \
|
||||
--hash=sha256:28ca37c070cad8ba8cd9790cd940535d40274d22f80ab87f3ac6a713e6e8454c \
|
||||
--hash=sha256:c74a7a2adf861c04d002db713dd85f84beb242228e671280bf709d765b03672b
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
pynacl==1.6.2 \
|
||||
--hash=sha256:018494d6d696ae03c7e656e5e74cdfd8ea1326962cc401bcf018f1ed8436811c \
|
||||
--hash=sha256:04316d1fc625d860b6c162fff704eb8426b1a8bcd3abacea11142cbd99a6b574 \
|
||||
@@ -339,6 +346,50 @@ pynacl==1.6.2 \
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
pytokens==0.4.1 \
|
||||
--hash=sha256:0fc71786e629cef478cbf29d7ea1923299181d0699dbe7c3c0f4a583811d9fc1 \
|
||||
--hash=sha256:11edda0942da80ff58c4408407616a310adecae1ddd22eef8c692fe266fa5009 \
|
||||
--hash=sha256:140709331e846b728475786df8aeb27d24f48cbcf7bcd449f8de75cae7a45083 \
|
||||
--hash=sha256:24afde1f53d95348b5a0eb19488661147285ca4dd7ed752bbc3e1c6242a304d1 \
|
||||
--hash=sha256:26cef14744a8385f35d0e095dc8b3a7583f6c953c2e3d269c7f82484bf5ad2de \
|
||||
--hash=sha256:27b83ad28825978742beef057bfe406ad6ed524b2d28c252c5de7b4a6dd48fa2 \
|
||||
--hash=sha256:292052fe80923aae2260c073f822ceba21f3872ced9a68bb7953b348e561179a \
|
||||
--hash=sha256:29d1d8fb1030af4d231789959f21821ab6325e463f0503a61d204343c9b355d1 \
|
||||
--hash=sha256:2a44ed93ea23415c54f3face3b65ef2b844d96aeb3455b8a69b3df6beab6acc5 \
|
||||
--hash=sha256:30f51edd9bb7f85c748979384165601d028b84f7bd13fe14d3e065304093916a \
|
||||
--hash=sha256:34bcc734bd2f2d5fe3b34e7b3c0116bfb2397f2d9666139988e7a3eb5f7400e3 \
|
||||
--hash=sha256:3ad72b851e781478366288743198101e5eb34a414f1d5627cdd585ca3b25f1db \
|
||||
--hash=sha256:3f901fe783e06e48e8cbdc82d631fca8f118333798193e026a50ce1b3757ea68 \
|
||||
--hash=sha256:42f144f3aafa5d92bad964d471a581651e28b24434d184871bd02e3a0d956037 \
|
||||
--hash=sha256:4a14d5f5fc78ce85e426aa159489e2d5961acf0e47575e08f35584009178e321 \
|
||||
--hash=sha256:4a58d057208cb9075c144950d789511220b07636dd2e4708d5645d24de666bdc \
|
||||
--hash=sha256:4e691d7f5186bd2842c14813f79f8884bb03f5995f0575272009982c5ac6c0f7 \
|
||||
--hash=sha256:5502408cab1cb18e128570f8d598981c68a50d0cbd7c61312a90507cd3a1276f \
|
||||
--hash=sha256:584c80c24b078eec1e227079d56dc22ff755e0ba8654d8383b2c549107528918 \
|
||||
--hash=sha256:5ad948d085ed6c16413eb5fec6b3e02fa00dc29a2534f088d3302c47eb59adf9 \
|
||||
--hash=sha256:670d286910b531c7b7e3c0b453fd8156f250adb140146d234a82219459b9640c \
|
||||
--hash=sha256:682fa37ff4d8e95f7df6fe6fe6a431e8ed8e788023c6bcc0f0880a12eab80ad1 \
|
||||
--hash=sha256:6d6c4268598f762bc8e91f5dbf2ab2f61f7b95bdc07953b602db879b3c8c18e1 \
|
||||
--hash=sha256:79fc6b8699564e1f9b521582c35435f1bd32dd06822322ec44afdeba666d8cb3 \
|
||||
--hash=sha256:8bdb9d0ce90cbf99c525e75a2fa415144fd570a1ba987380190e8b786bc6ef9b \
|
||||
--hash=sha256:8fcb9ba3709ff77e77f1c7022ff11d13553f3c30299a9fe246a166903e9091eb \
|
||||
--hash=sha256:941d4343bf27b605e9213b26bfa1c4bf197c9c599a9627eb7305b0defcfe40c1 \
|
||||
--hash=sha256:967cf6e3fd4adf7de8fc73cd3043754ae79c36475c1c11d514fc72cf5490094a \
|
||||
--hash=sha256:970b08dd6b86058b6dc07efe9e98414f5102974716232d10f32ff39701e841c4 \
|
||||
--hash=sha256:97f50fd18543be72da51dd505e2ed20d2228c74e0464e4262e4899797803d7fa \
|
||||
--hash=sha256:9bd7d7f544d362576be74f9d5901a22f317efc20046efe2034dced238cbbfe78 \
|
||||
--hash=sha256:add8bf86b71a5d9fb5b89f023a80b791e04fba57960aa790cc6125f7f1d39dfe \
|
||||
--hash=sha256:b35d7e5ad269804f6697727702da3c517bb8a5228afa450ab0fa787732055fc9 \
|
||||
--hash=sha256:b49750419d300e2b5a3813cf229d4e5a4c728dae470bcc89867a9ad6f25a722d \
|
||||
--hash=sha256:d31b97b3de0f61571a124a00ffe9a81fb9939146c122c11060725bd5aea79975 \
|
||||
--hash=sha256:d70e77c55ae8380c91c0c18dea05951482e263982911fc7410b1ffd1dadd3440 \
|
||||
--hash=sha256:d9907d61f15bf7261d7e775bd5d7ee4d2930e04424bab1972591918497623a16 \
|
||||
--hash=sha256:da5baeaf7116dced9c6bb76dc31ba04a2dc3695f3d9f74741d7910122b456edc \
|
||||
--hash=sha256:dc74c035f9bfca0255c1af77ddd2d6ae8419012805453e4b0e7513e17904545d \
|
||||
--hash=sha256:dcafc12c30dbaf1e2af0490978352e0c4041a7cde31f4f81435c2a5e8b9cabb6 \
|
||||
--hash=sha256:ee44d0f85b803321710f9239f335aafe16553b39106384cef8e6de40cb4ef2f6 \
|
||||
--hash=sha256:f66a6bbe741bd431f6d741e617e0f39ec7257ca1f89089593479347cc4d13324
|
||||
# via black
|
||||
requests==2.32.4 \
|
||||
--hash=sha256:27babd3cda2a6d50b30443204ee89830707d396671944c998b5975b031ac2b2c \
|
||||
--hash=sha256:27d0316682c8a29834d3264820024b62a36942083d52caf2f14c0591336d3422
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
black~=25.1
|
||||
black>=26.3.1
|
||||
darker==2.1.1
|
||||
PyGithub==2.6.1
|
||||
cryptography>=46.0.5
|
||||
@@ -7,3 +7,4 @@ requests>=2.32.4
|
||||
idna>=3.7
|
||||
certifi>=2024.7.4
|
||||
PyNaCl>=1.6.2
|
||||
PyJWT>=2.12.1
|
||||
@@ -6,7 +6,8 @@ set(FEXCORE_BASE_SRCS
|
||||
Utils/FileLoading.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
Utils/SpinWaitLock.cpp)
|
||||
Utils/SpinWaitLock.cpp
|
||||
Utils/WildcardMatcher.cpp)
|
||||
|
||||
if (NOT MINGW)
|
||||
list(APPEND FEXCORE_BASE_SRCS
|
||||
@@ -123,6 +124,11 @@ else()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
# GCC requires libatomic to use 128-bit atomics
|
||||
list(APPEND LIBS atomic)
|
||||
endif()
|
||||
|
||||
# Generate config
|
||||
configure_file(${CMAKE_CURRENT_SOURCE_DIR}/Interface/Config/Config.json.in
|
||||
${CMAKE_BINARY_DIR}/generated/Config/Config.json)
|
||||
|
||||
@@ -16,7 +16,11 @@ struct VectorScalarF64Pair {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
// Can't use uint8x16_t directly from arm_neon.h here.
|
||||
// Overrides softfloat-3e's defines which causes problems.
|
||||
#ifdef __clang__
|
||||
using VectorRegType = __attribute__((neon_vector_type(16))) uint8_t;
|
||||
#else
|
||||
using VectorRegType = __attribute__((vector_size(16))) uint8_t;
|
||||
#endif
|
||||
struct VectorRegPairType {
|
||||
VectorRegType val[2];
|
||||
};
|
||||
|
||||
@@ -75,7 +75,9 @@
|
||||
"ENABLE3DNOW": "enable3dnow",
|
||||
"DISABLE3DNOW": "disable3dnow",
|
||||
"ENABLESSE4A": "enablesse4a",
|
||||
"DISABLESSE4A": "disablesse4a"
|
||||
"DISABLESSE4A": "disablesse4a",
|
||||
"ENABLEMOPS": "enablemops",
|
||||
"DISABLEMOPS": "disablemops"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -99,7 +101,8 @@
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
|
||||
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it"
|
||||
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it",
|
||||
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
@@ -192,7 +195,7 @@
|
||||
},
|
||||
"DisableL2Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
@@ -200,7 +203,7 @@
|
||||
},
|
||||
"DynamicL1Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
|
||||
@@ -127,6 +127,7 @@ public:
|
||||
|
||||
void ExecuteThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
bool CheckIfBlockIsCacheable(FEXCore::Core::InternalThreadState&, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
@@ -208,7 +209,7 @@ public:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
|
||||
void InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
|
||||
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex& GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
|
||||
@@ -262,7 +263,7 @@ public:
|
||||
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
|
||||
} Config;
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex CodeInvalidationMutex {};
|
||||
|
||||
uint32_t StrictSplitLockMutex {};
|
||||
|
||||
|
||||
@@ -448,8 +448,10 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
return;
|
||||
}
|
||||
|
||||
if ((Constant >> 32) == 0) {
|
||||
if ((Constant >> 32) == 0 && !NOPPad) {
|
||||
// If the upper 32-bits is all zero, we can now switch to a 32-bit move.
|
||||
// NOTE: The NOP padding code does not appropriately adjust to this yet,
|
||||
// so we skip this optimization in that case
|
||||
s = ARMEmitter::Size::i32Bit;
|
||||
Is64Bit = false;
|
||||
Segments = std::min(Segments, 2);
|
||||
@@ -677,7 +679,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask, bool NZCV) {
|
||||
void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, SpillStaticRegOptions Options) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Disable AFP features when spilling registers.
|
||||
@@ -698,7 +700,7 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
}
|
||||
#endif
|
||||
|
||||
if (NZCV) {
|
||||
if (Options.NZCV) {
|
||||
// Regardless of what GPRs/FPRs we're spilling, we need to spill NZCV since it
|
||||
// is always static and almost certainly clobbered by the subsequent code.
|
||||
//
|
||||
@@ -710,25 +712,25 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
unsigned PFAFSpillMask = GPRSpillMask & PFAFMask;
|
||||
GPRSpillMask &= ~PFAFSpillMask;
|
||||
unsigned PFAFSpillMask = Options.GPRSpillMask & PFAFMask;
|
||||
Options.GPRSpillMask &= ~PFAFSpillMask;
|
||||
|
||||
str(REG_CALLRET_SP, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.callret_sp));
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i + 1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) && ((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
} else if (((1U << Reg1.Idx()) & GPRSpillMask)) {
|
||||
str(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
} else if (((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
str(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i + 1]));
|
||||
if (((1U << Reg1.Idx()) & Options.GPRSpillMask) && ((1U << Reg2.Idx()) & Options.GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i));
|
||||
} else if (((1U << Reg1.Idx()) & Options.GPRSpillMask)) {
|
||||
str(Reg1.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i));
|
||||
} else if (((1U << Reg2.Idx()) & Options.GPRSpillMask)) {
|
||||
str(Reg2.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i + 1));
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (NZCV && PFAFSpillMask) {
|
||||
if (Options.NZCV && PFAFSpillMask) {
|
||||
auto PFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
@@ -737,21 +739,21 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
stp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), PFOffset);
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
if (Options.FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
|
||||
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TmpReg, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
if (((1U << Reg.Idx()) & Options.FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TmpReg, ARRAY_OFFSETOF(Core::CpuStateFrame, State.xmm.avx.data, i));
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B, STATE.R(), TmpReg);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GPRSpillMask && FPRSpillMask == ~0U) {
|
||||
if (Options.GPRSpillMask && Options.FPRSpillMask == ~0U) {
|
||||
// Optimize the common case where we can spill four registers per instruction
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data));
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
@@ -764,12 +766,12 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) && ((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
} else if (((1U << Reg1.Idx()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
} else if (((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i + 1][0]));
|
||||
if (((1U << Reg1.Idx()) & Options.FPRSpillMask) && ((1U << Reg2.Idx()) & Options.FPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i));
|
||||
} else if (((1U << Reg1.Idx()) & Options.FPRSpillMask)) {
|
||||
str(Reg1.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i));
|
||||
} else if (((1U << Reg2.Idx()) & Options.FPRSpillMask)) {
|
||||
str(Reg2.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i + 1));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -777,8 +779,7 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask, std::optional<ARMEmitter::Register> OptionalReg,
|
||||
std::optional<ARMEmitter::Register> OptionalReg2, bool NZCV) {
|
||||
void Arm64Emitter::FillStaticRegs(FillStaticRegOptions Options) {
|
||||
auto FindTempReg = [this](uint32_t* GPRFillMask) -> std::optional<ARMEmitter::Register> {
|
||||
for (auto Reg : StaticRegisters) {
|
||||
if (((1U << Reg.Idx()) & *GPRFillMask)) {
|
||||
@@ -789,20 +790,21 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
return std::nullopt;
|
||||
};
|
||||
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 2 GPRs for a temp");
|
||||
uint32_t TempGPRFillMask = GPRFillMask;
|
||||
if (!OptionalReg.has_value()) {
|
||||
OptionalReg = FindTempReg(&TempGPRFillMask);
|
||||
LOGMAN_THROW_A_FMT(Options.GPRFillMask != 0, "Must fill at least 2 GPRs for a temp");
|
||||
uint32_t TempGPRFillMask = Options.GPRFillMask;
|
||||
if (!Options.OptionalReg.has_value()) {
|
||||
Options.OptionalReg = FindTempReg(&TempGPRFillMask);
|
||||
}
|
||||
|
||||
if (!OptionalReg2.has_value()) {
|
||||
OptionalReg2 = FindTempReg(&TempGPRFillMask);
|
||||
if (!Options.OptionalReg2.has_value()) {
|
||||
Options.OptionalReg2 = FindTempReg(&TempGPRFillMask);
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(OptionalReg.has_value() && OptionalReg2.has_value(), "Didn't have an SRA register to use as a temporary while "
|
||||
"spilling!");
|
||||
LOGMAN_THROW_A_FMT(Options.OptionalReg.has_value() && Options.OptionalReg2.has_value(), "Didn't have an SRA register to use as a "
|
||||
"temporary while "
|
||||
"spilling!");
|
||||
|
||||
auto TmpReg = *OptionalReg;
|
||||
auto TmpReg2 = *OptionalReg2;
|
||||
auto TmpReg = *Options.OptionalReg;
|
||||
auto TmpReg2 = *Options.OptionalReg2;
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
// Load STATE in from the CPU area as x28 is not callee saved in the ARM64EC ABI.
|
||||
@@ -812,7 +814,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
ldr(REG_CALLRET_SP, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.callret_sp));
|
||||
|
||||
if (NZCV) {
|
||||
if (Options.NZCV) {
|
||||
// Regardless of what GPRs/FPRs we're filling, we need to fill NZCV since it
|
||||
// is always static and was almost certainly clobbered.
|
||||
//
|
||||
@@ -822,23 +824,23 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
|
||||
}
|
||||
|
||||
FillSpecialRegs(TmpReg, TmpReg2, true, FPRs);
|
||||
FillSpecialRegs(TmpReg, TmpReg2, true, Options.FPRs);
|
||||
|
||||
if (FPRs) {
|
||||
if (Options.FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TmpReg, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
if (((1U << Reg.Idx()) & Options.FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TmpReg, ARRAY_OFFSETOF(Core::CpuStateFrame, State.xmm.avx.data, i));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B.Zeroing(), STATE.R(), TmpReg);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
if (Options.GPRFillMask && Options.FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data));
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
@@ -851,12 +853,12 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) && ((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
} else if (((1U << Reg1.Idx()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0]));
|
||||
} else if (((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i + 1][0]));
|
||||
if (((1U << Reg1.Idx()) & Options.FPRFillMask) && ((1U << Reg2.Idx()) & Options.FPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.Q(), Reg2.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i));
|
||||
} else if (((1U << Reg1.Idx()) & Options.FPRFillMask)) {
|
||||
ldr(Reg1.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i));
|
||||
} else if (((1U << Reg2.Idx()) & Options.FPRFillMask)) {
|
||||
ldr(Reg2.Q(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.xmm.sse.data, i + 1));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -865,23 +867,23 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
// PF/AF are special, remove them from the mask
|
||||
uint32_t PFAFMask = ((1u << REG_PF.Idx()) | ((1u << REG_AF.Idx())));
|
||||
uint32_t PFAFFillMask = GPRFillMask & PFAFMask;
|
||||
GPRFillMask &= ~PFAFMask;
|
||||
uint32_t PFAFFillMask = Options.GPRFillMask & PFAFMask;
|
||||
Options.GPRFillMask &= ~PFAFMask;
|
||||
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i += 2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i + 1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) && ((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
} else if ((1U << Reg1.Idx()) & GPRFillMask) {
|
||||
ldr(Reg1.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
} else if ((1U << Reg2.Idx()) & GPRFillMask) {
|
||||
ldr(Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i + 1]));
|
||||
if (((1U << Reg1.Idx()) & Options.GPRFillMask) && ((1U << Reg2.Idx()) & Options.GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i));
|
||||
} else if ((1U << Reg1.Idx()) & Options.GPRFillMask) {
|
||||
ldr(Reg1.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i));
|
||||
} else if ((1U << Reg2.Idx()) & Options.GPRFillMask) {
|
||||
ldr(Reg2.X(), STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, State.gregs, i + 1));
|
||||
}
|
||||
}
|
||||
|
||||
// Now handle PF/AF
|
||||
if (NZCV && PFAFFillMask) {
|
||||
if (Options.NZCV && PFAFFillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
|
||||
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
@@ -1056,7 +1058,10 @@ size_t Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, boo
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
// Spill the static registers.
|
||||
SpillStaticRegs(TmpReg, true, PreserveSRAMask, PreserveSRAFPRMask);
|
||||
SpillStaticRegs(TmpReg, {
|
||||
.GPRSpillMask = PreserveSRAMask,
|
||||
.FPRSpillMask = PreserveSRAFPRMask,
|
||||
});
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
@@ -1103,7 +1108,11 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
}
|
||||
|
||||
// Fill the static registers.
|
||||
FillStaticRegs(FPRs, PreserveSRAMask, PreserveSRAFPRMask);
|
||||
FillStaticRegs({
|
||||
.GPRFillMask = PreserveSRAMask,
|
||||
.FPRFillMask = PreserveSRAFPRMask,
|
||||
.FPRs = FPRs,
|
||||
});
|
||||
|
||||
// Pop the vector registers.
|
||||
PopVectorRegisters(CanUseSVE256, DynamicFPRs);
|
||||
|
||||
@@ -135,10 +135,35 @@ protected:
|
||||
// Returning REG_INVALID if there was no mapping.
|
||||
FEXCore::X86State::X86Reg GetX86RegRelationToARMReg(ARMEmitter::Register Reg);
|
||||
|
||||
void SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U, bool NZCV = true);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U,
|
||||
std::optional<ARMEmitter::Register> OptionalReg = std::nullopt,
|
||||
std::optional<ARMEmitter::Register> OptionalReg2 = std::nullopt, bool NZCV = true);
|
||||
struct SpillStaticRegOptions final {
|
||||
uint32_t GPRSpillMask {~0U};
|
||||
uint32_t FPRSpillMask {~0U};
|
||||
bool FPRs {true};
|
||||
bool NZCV {true};
|
||||
};
|
||||
|
||||
struct FillStaticRegOptions final {
|
||||
std::optional<ARMEmitter::Register> OptionalReg {std::nullopt};
|
||||
std::optional<ARMEmitter::Register> OptionalReg2 {std::nullopt};
|
||||
uint32_t GPRFillMask {~0U};
|
||||
uint32_t FPRFillMask {~0U};
|
||||
bool FPRs {true};
|
||||
bool NZCV {true};
|
||||
};
|
||||
|
||||
void SpillStaticRegs(ARMEmitter::Register TmpReg, SpillStaticRegOptions Options);
|
||||
void FillStaticRegs(FillStaticRegOptions Options);
|
||||
|
||||
|
||||
void SpillStaticRegs(ARMEmitter::Register TmpReg) {
|
||||
// Work around a clang bug: https://bugs.llvm.org/show_bug.cgi?id=36684
|
||||
SpillStaticRegs(TmpReg, {});
|
||||
}
|
||||
|
||||
void FillStaticRegs() {
|
||||
// Work around a clang bug: https://bugs.llvm.org/show_bug.cgi?id=36684
|
||||
FillStaticRegs({});
|
||||
}
|
||||
|
||||
// Register 0-18 + 29 + 30 are caller saved
|
||||
static constexpr uint32_t CALLER_GPR_MASK = 0b0110'0000'0000'0111'1111'1111'1111'1111U;
|
||||
@@ -178,7 +203,9 @@ protected:
|
||||
if (SupportsPreserveAllABI) {
|
||||
return SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
} else {
|
||||
SpillStaticRegs(TmpReg, FPRs);
|
||||
SpillStaticRegs(TmpReg, {
|
||||
.FPRs = FPRs,
|
||||
});
|
||||
return PushDynamicRegs(TmpReg);
|
||||
}
|
||||
}
|
||||
@@ -188,7 +215,7 @@ protected:
|
||||
FillForPreserveAllABICall(FPRs);
|
||||
} else {
|
||||
PopDynamicRegs();
|
||||
FillStaticRegs(FPRs);
|
||||
FillStaticRegs({.FPRs = FPRs});
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -363,6 +363,9 @@ namespace CPU {
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMemJIT", reinterpret_cast<void*>(Ptr), Size);
|
||||
|
||||
// Huge-pages reduce the amount of iTLB misses dramatically when it works.
|
||||
FEXCore::Allocator::VirtualTHPControl(reinterpret_cast<void*>(Ptr), Size, FEXCore::Allocator::THPControl::Enable);
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
|
||||
|
||||
@@ -141,7 +141,7 @@ constexpr uint32_t FAMILY_IDENTIFIER = GenerateFamily(CPUFamily {
|
||||
#endif
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
uint32_t GetCycleCounterFrequency() {
|
||||
uint64_t GetCycleCounterFrequency() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], CNTFRQ_EL0" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
@@ -408,7 +408,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
}
|
||||
|
||||
#else
|
||||
uint32_t GetCycleCounterFrequency() {
|
||||
uint64_t GetCycleCounterFrequency() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -756,6 +756,95 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 29) | // Arch capabilities - Speculative side channel mitigations
|
||||
(0 << 30) | // Arch capabilities - MSR module specific
|
||||
(0 << 31); // SSBD - Speculative Store Bypass Disable
|
||||
} else if (Leaf == 1) {
|
||||
Res.eax = (0U << 0) | // SHA512
|
||||
(0U << 1) | // SM3
|
||||
(0U << 2) | // SM4
|
||||
(0U << 3) | // RAO_INT
|
||||
(0U << 4) | // AVX_VNNI
|
||||
(0U << 5) | // AVX512_BF16
|
||||
(0U << 6) | // LASS (Linear Address Space Separation)
|
||||
(0U << 7) | // CMPCCXADD
|
||||
(0U << 8) | // ARCH_PERFMON_EXT
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // FAST_REP_MOVSB
|
||||
(0U << 11) | // FAST_REP_STOSB
|
||||
(0U << 12) | // FAST_REP_CMPSB_SCASB
|
||||
(0U << 13) | // Reserved
|
||||
(0U << 14) | // Reserved
|
||||
(0U << 15) | // Reserved
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // FRED (Flexible Return and Event Delivery)
|
||||
(0U << 18) | // LKGS (Load into Kernel GS Base)
|
||||
(0U << 19) | // WRMSRNS
|
||||
(0U << 20) | // NMI_SRC
|
||||
(0U << 21) | // AMX_FP16
|
||||
(0U << 22) | // HRESET
|
||||
(0U << 23) | // AVX_IFMA
|
||||
(0U << 24) | // Reserved
|
||||
(0U << 25) | // Reserved
|
||||
(0U << 26) | // LAM (Linear Address Masking)
|
||||
(0U << 27) | // MSRLIST
|
||||
(0U << 28) | // Reserved
|
||||
(0U << 29) | // Reserved
|
||||
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
|
||||
(0U << 31); // MOVRS
|
||||
|
||||
// Bits 4-31 currently reserved.
|
||||
Res.ebx = (0U << 0) | // PPIN
|
||||
(0U << 1) | // PBNDKB
|
||||
(0U << 2) | // Reserved
|
||||
(0U << 3); // CPUIDMAXVAL_LIM_RMV
|
||||
|
||||
// Bits 6-31 also reserved.
|
||||
Res.ecx = (0U << 0) | // RDT_M_ASYM
|
||||
(0U << 1) | // RDT_A_ASYM
|
||||
(0U << 2) | // Reserved
|
||||
(0U << 3) | // Reserved
|
||||
(0U << 4) | // Reserved
|
||||
(0U << 5); // MSR_IMM
|
||||
|
||||
// Bits 25-31 also reserved.
|
||||
Res.edx = (0U << 0) | // Reserved
|
||||
(0U << 1) | // Reserved
|
||||
(0U << 2) | // Reserved
|
||||
(0U << 3) | // Reserved
|
||||
(0U << 4) | // AVX_VNNI_INT8
|
||||
(0U << 5) | // AVX_NE_CONVERT
|
||||
(0U << 6) | // Reserved
|
||||
(0U << 7) | // Reserved
|
||||
(0U << 8) | // AMX_COMPLEX
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // AVX_VNNI_INT16
|
||||
(0U << 11) | // Reserved
|
||||
(0U << 12) | // Reserved
|
||||
(0U << 13) | // UTMR (User-timer events)
|
||||
(0U << 14) | // PREFETCHI
|
||||
(0U << 15) | // USER_MSR
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // UIRET_UIF
|
||||
(0U << 18) | // CET_SSS
|
||||
(0U << 19) | // AVX10
|
||||
(0U << 20) | // Reserved
|
||||
(0U << 21) | // APX_F
|
||||
(0U << 22) | // SEC-TEE_ATTESTATION
|
||||
(0U << 23) | // MWAIT
|
||||
(0U << 24); // SLSM (Static LSM)
|
||||
} else if (Leaf == 2) {
|
||||
// All bits are reserved except for EDX
|
||||
Res.eax = 0;
|
||||
Res.ebx = 0;
|
||||
Res.ecx = 0;
|
||||
|
||||
// Bits 8-31 are reserved.
|
||||
Res.edx = (0U << 0) | // PSFD
|
||||
(0U << 1) | // IPRED_CTRL
|
||||
(0U << 2) | // RRSBA_CTRL
|
||||
(0U << 3) | // DDPD_U
|
||||
(0U << 4) | // BHI_CTRL
|
||||
(0U << 5) | // MCDT_NO
|
||||
(0U << 6) | // UC_LOCK_DISABLE
|
||||
(0U << 7); // MONITOR_MITG_NO
|
||||
}
|
||||
|
||||
return Res;
|
||||
@@ -813,7 +902,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
// TSC frequency = ECX * EBX / EAX
|
||||
uint32_t FrequencyHz = GetCycleCounterFrequency();
|
||||
uint64_t FrequencyHz = GetCycleCounterFrequency();
|
||||
if (FrequencyHz) {
|
||||
Res.eax = 1;
|
||||
Res.ebx = 1U << CTX->Config.TSCScale;
|
||||
@@ -834,6 +923,27 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_1Ah(uint32_t Leaf) const {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_24h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
|
||||
if (Leaf == 0) {
|
||||
// EAX indicates the maximum number of subleaves.
|
||||
Res.eax = 0;
|
||||
|
||||
// Bits 19-31 reserved
|
||||
// NOTE: We return all zero here until we have a CPU with AVX10
|
||||
// even if some of the fields otherwise have fixed values.
|
||||
Res.ebx = (0U << 0) | // (bits 0-7 specify the vector ISA version)
|
||||
(0U << 16); // Defined as always 0b111
|
||||
|
||||
// All bits reserved
|
||||
Res.ecx = 0;
|
||||
Res.edx = 0;
|
||||
}
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
// Hypervisor CPUID information leaf
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0000h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
|
||||
@@ -14,7 +14,7 @@ namespace Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
uint32_t GetCycleCounterFrequency();
|
||||
uint64_t GetCycleCounterFrequency();
|
||||
|
||||
// Debugging define to switch what family of CPU we execute as.
|
||||
// Might be useful if an application makes an assumption about a CPU.
|
||||
@@ -176,6 +176,7 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_24h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0001h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf) const;
|
||||
@@ -200,7 +201,7 @@ private:
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
void SetupFeatures();
|
||||
static constexpr size_t PRIMARY_FUNCTION_COUNT = 27;
|
||||
static constexpr size_t PRIMARY_FUNCTION_COUNT = 37;
|
||||
static constexpr size_t HYPERVISOR_FUNCTION_COUNT = 2;
|
||||
static constexpr size_t EXTENDED_FUNCTION_COUNT = 32;
|
||||
static constexpr std::array<FunctionHandler, PRIMARY_FUNCTION_COUNT> Primary = {
|
||||
@@ -268,7 +269,48 @@ private:
|
||||
#ifndef CPUID_AMD
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
&CPUIDEmu::Function_1Ah,
|
||||
// 0x1B: PCONFIG info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1C: Last Branch Records (LBR) info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1D: Tile info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1E: TMUL info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1F: V2 Extended topology
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x20: Processor History Reset info
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x21: Unimplemented
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x22: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x23: Architectural Performance Monitoring Extended
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x24: Converged Vector ISA
|
||||
&CPUIDEmu::Function_24h,
|
||||
#else
|
||||
// 0x1A: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1B: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1C: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1D: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1E: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x1F: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x20: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x21: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x22: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x23: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
// 0x24: Reserved
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
#endif
|
||||
};
|
||||
@@ -340,9 +382,49 @@ private:
|
||||
#ifndef CPUID_AMD
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1B: PCONFIG info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1C: Last Branch Records (LBR) info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1D: Tile info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1E: TMUL info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1F: V2 Extended topology
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x20: Processor History Reset info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x21: Unimplemented/Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x22: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x23: Architectural Performance Monitoring Extended
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x24: Converged Vector ISA
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT},
|
||||
#else
|
||||
// 0x1A: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1B: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1C: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1D: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1E: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x1F: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x20: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x21: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x22: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x23: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 0x24: Reserved
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
#endif
|
||||
}};
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
|
||||
#include <Interface/Context/Context.h>
|
||||
#include <Interface/Core/ArchHelpers/Arm64Emitter.h>
|
||||
@@ -399,13 +399,13 @@ bool CodeCache::LoadData(Core::InternalThreadState* Thread, std::byte* MappedCac
|
||||
auto begin = ranges::lower_bound(BlockList, BinarySection.BeginVA - BinarySection.FileStartVA, std::less {}, &BlockListEntry::first);
|
||||
auto end =
|
||||
ranges::upper_bound(begin, BlockList.end(), BinarySection.EndVA - BinarySection.FileStartVA - 1, std::less {}, &BlockListEntry::first);
|
||||
BlockList.erase(end, BlockList.end());
|
||||
BlockList.erase(BlockList.begin(), begin);
|
||||
if (BlockList.empty()) {
|
||||
if (begin == end) {
|
||||
// Not an error since there is just no data to load
|
||||
LogMan::Msg::IFmt("No blocks cached in this range, aborting");
|
||||
return true;
|
||||
}
|
||||
BlockList.erase(end, BlockList.end());
|
||||
BlockList.erase(BlockList.begin(), begin);
|
||||
}
|
||||
|
||||
// Read relocations
|
||||
|
||||
@@ -30,7 +30,7 @@ $end_info$
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
#include "Utils/variable_length_integer.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -502,11 +502,14 @@ static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter*
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", NewIR.PostRA() ? "post" : "pre", GuestRIP, out.str());
|
||||
};
|
||||
|
||||
bool ContextImpl::CheckIfBlockIsCacheable(FEXCore::Core::InternalThreadState& Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
return Thread.FrontendDecoder->CheckIfCacheable(Thread, reinterpret_cast<const uint8_t*>(GuestRIP), GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
ContextImpl::GenerateIRResult
|
||||
ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
uint64_t TotalInstructions {0};
|
||||
@@ -706,8 +709,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return {{}, 0, 0, 0, 0};
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
return {std::nullopt, 0, 0, 0, 0};
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
@@ -774,6 +777,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, NeedsAddGuestCodeRanges] =
|
||||
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
if (!IRView) {
|
||||
// OpDispatcher IR already released in this case.
|
||||
return {{}, nullptr, 0, 0, false};
|
||||
}
|
||||
|
||||
@@ -784,6 +788,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
// as expensive and are easily reverted.
|
||||
if (MaxInst != 1) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
// Raced to compile, release the OpDispatcher IR.
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
|
||||
.DebugData = nullptr,
|
||||
|
||||
@@ -546,6 +546,10 @@ void Dispatcher::EmitDispatcher() {
|
||||
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.LUDIV));
|
||||
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.LDIV));
|
||||
|
||||
EmitF64Sin();
|
||||
EmitF64Cos();
|
||||
EmitF64Tan();
|
||||
|
||||
// Interpreter fallbacks
|
||||
{
|
||||
constexpr static std::array<FallbackABI, FABI_UNKNOWN> ABIS {{
|
||||
@@ -845,6 +849,403 @@ void Dispatcher::EmitF64ToExtF80() {
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
|
||||
void Dispatcher::EmitF64Sin() {
|
||||
F64SinHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
constexpr auto V2 = ARMEmitter::VReg::v2;
|
||||
constexpr auto V3 = ARMEmitter::VReg::v3;
|
||||
constexpr auto V4 = ARMEmitter::VReg::v4;
|
||||
constexpr auto V5 = ARMEmitter::VReg::v5;
|
||||
|
||||
ARMEmitter::ForwardLabel Fallback, NonZero;
|
||||
ARMEmitter::ForwardLabel InvPiPi1Label, Pi23Label;
|
||||
ARMEmitter::ForwardLabel C0Label, C1Label, C2Label, C3Label, C4Label, C5Label, C6Label;
|
||||
ARMEmitter::ForwardLabel RangeLabel;
|
||||
|
||||
// sin(+/-0) = +/-0
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP1, VTMP1.D());
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &NonZero);
|
||||
ret();
|
||||
(void)Bind(&NonZero);
|
||||
|
||||
// Save q2-q5.
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::QReg::q3, ARMEmitter::Reg::rsp, -64);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::QReg::q4, ARMEmitter::QReg::q5, ARMEmitter::Reg::rsp, 32);
|
||||
|
||||
// save nzcv
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
// Range check: fall back for |x| >= 2^23, NaN, and inf.
|
||||
fabs(VTMP2.D(), VTMP1.D());
|
||||
ldr(V2.D(), &RangeLabel);
|
||||
fcmp(VTMP2.D(), V2.D());
|
||||
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
|
||||
|
||||
// n = rint(x/pi).
|
||||
ldr(V2.Q(), &InvPiPi1Label); // q2 = {inv_pi, pi_1}
|
||||
fmul(VTMP2.D(), VTMP1.D(), V2.D());
|
||||
frinta(VTMP2.D(), VTMP2.D());
|
||||
|
||||
// odd = (int(n) & 1) << 63.
|
||||
fcvtzs(ARMEmitter::Size::i64Bit, TMP1, VTMP2.D());
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP1, TMP1, 63);
|
||||
|
||||
// r = x - n*pi (range reduction) via .2D lane-indexed FMLS.
|
||||
ldr(V3.Q(), &Pi23Label); // q3 = {pi_2, pi_3}
|
||||
fmov(V4.D(), VTMP1.D()); // r = x
|
||||
fmls(ARMEmitter::SubRegSize::i64Bit, V4.Q(), VTMP2.Q(), V2.Q(), 1); // r -= n * pi_1
|
||||
fmls(ARMEmitter::SubRegSize::i64Bit, V4.Q(), VTMP2.Q(), V3.Q(), 0); // r -= n * pi_2
|
||||
fmls(ARMEmitter::SubRegSize::i64Bit, V4.Q(), VTMP2.Q(), V3.Q(), 1); // r -= n * pi_3
|
||||
|
||||
// r^2, r^4.
|
||||
fmul(V5.D(), V4.D(), V4.D());
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, V4.D());
|
||||
fmul(V3.D(), V5.D(), V5.D());
|
||||
|
||||
// Estrin polynomial: p = c0 + r2*c1 + r4*(c2 + r2*c3) + r8*(c4 + r2*c5 + r4*c6).
|
||||
// Level 1 (independent FMAs).
|
||||
ldr(VTMP1.D(), &C0Label);
|
||||
ldr(VTMP2.D(), &C1Label);
|
||||
fmadd(VTMP1.D(), V5.D(), VTMP2.D(), VTMP1.D()); // p01 = c0 + r2*c1
|
||||
|
||||
ldr(VTMP2.D(), &C2Label);
|
||||
ldr(V2.D(), &C3Label);
|
||||
fmadd(VTMP2.D(), V5.D(), V2.D(), VTMP2.D()); // p23 = c2 + r2*c3
|
||||
|
||||
ldr(V2.D(), &C4Label);
|
||||
ldr(V4.D(), &C5Label);
|
||||
fmadd(V2.D(), V5.D(), V4.D(), V2.D()); // p45 = c4 + r2*c5
|
||||
|
||||
// Level 2 (serial).
|
||||
ldr(V4.D(), &C6Label);
|
||||
fmadd(V2.D(), V3.D(), V4.D(), V2.D()); // p46 = p45 + r4*c6
|
||||
fmadd(VTMP2.D(), V3.D(), V2.D(), VTMP2.D()); // p26 = p23 + r4*p46
|
||||
fmadd(VTMP1.D(), V3.D(), VTMP2.D(), VTMP1.D()); // p06 = p01 + r4*p26
|
||||
|
||||
// y = r + r^3 * p06.
|
||||
fmov(ARMEmitter::Size::i64Bit, V4.D(), TMP2);
|
||||
fmul(V5.D(), V5.D(), V4.D());
|
||||
fmadd(VTMP1.D(), V5.D(), VTMP1.D(), V4.D());
|
||||
|
||||
// result = y XOR odd.
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, VTMP1.D());
|
||||
eor(ARMEmitter::Size::i64Bit, TMP2, TMP2, TMP1);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP2);
|
||||
|
||||
// restore nzcv
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
// Restore q2-q5 and return.
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::QReg::q4, ARMEmitter::QReg::q5, ARMEmitter::Reg::rsp, 32);
|
||||
ldp<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::QReg::q3, ARMEmitter::Reg::rsp, 64);
|
||||
ret();
|
||||
|
||||
// Fallback path.
|
||||
(void)Bind(&Fallback);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::QReg::q4, ARMEmitter::QReg::q5, ARMEmitter::Reg::rsp, 32);
|
||||
ldp<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::QReg::q3, ARMEmitter::Reg::rsp, 64);
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64SIN].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64SIN].Func));
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
// Constant pool.
|
||||
Align(16);
|
||||
(void)Bind(&InvPiPi1Label);
|
||||
dc64(0x3FD4'5F30'6DC9'C883ULL); // inv_pi
|
||||
dc64(0x4009'21FB'5444'2D18ULL); // pi_1
|
||||
(void)Bind(&Pi23Label);
|
||||
dc64(0x3CA1'A626'3314'5C06ULL); // pi_2
|
||||
dc64(0x395C'1CD1'2902'4E09ULL); // pi_3
|
||||
(void)Bind(&C0Label);
|
||||
dc64(0xBFC5'5555'5555'547BULL); // c0
|
||||
(void)Bind(&C1Label);
|
||||
dc64(0x3F81'1111'1110'8A4DULL); // c1
|
||||
(void)Bind(&C2Label);
|
||||
dc64(0xBF2A'01A0'1993'6F27ULL); // c2
|
||||
(void)Bind(&C3Label);
|
||||
dc64(0x3EC7'1DE3'7A97'D93EULL); // c3
|
||||
(void)Bind(&C4Label);
|
||||
dc64(0xBE5A'E633'9199'87C6ULL); // c4
|
||||
(void)Bind(&C5Label);
|
||||
dc64(0x3DE6'0E27'7AE0'7CECULL); // c5
|
||||
(void)Bind(&C6Label);
|
||||
dc64(0xBD69'E954'0300'A100ULL); // c6
|
||||
(void)Bind(&RangeLabel);
|
||||
dc64(0x4160'0000'0000'0000ULL); // 2^23
|
||||
}
|
||||
|
||||
void Dispatcher::EmitF64Cos() {
|
||||
F64CosHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
constexpr auto Accum = ARMEmitter::VReg::v2;
|
||||
|
||||
ARMEmitter::ForwardLabel Fallback;
|
||||
ARMEmitter::ForwardLabel RangeLabel, InvPiLabel;
|
||||
ARMEmitter::ForwardLabel Pi1Label, Pi2Label, Pi3Label;
|
||||
ARMEmitter::ForwardLabel C0Label, C1Label, C2Label, C3Label, C4Label, C5Label, C6Label;
|
||||
|
||||
// Save q2 for use as accumulator
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
// save nzcv
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
// Range check: fall back for |x| >= 2^23, NaN, and inf.
|
||||
fabs(VTMP2.D(), VTMP1.D());
|
||||
ldr(Accum.D(), &RangeLabel);
|
||||
fcmp(VTMP2.D(), Accum.D());
|
||||
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
|
||||
|
||||
// n = rint(x * (1/pi) + 0.5).
|
||||
ldr(Accum.D(), &InvPiLabel);
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP2, 0.5f);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
frinta(VTMP2.D(), VTMP2.D());
|
||||
|
||||
// odd = (int(n) & 1) << 63.
|
||||
fcvtzs(ARMEmitter::Size::i64Bit, TMP1, VTMP2.D());
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP1, TMP1, 63);
|
||||
|
||||
// Save input to Accum before overwriting VTMP1.
|
||||
fmov(Accum.D(), VTMP1.D());
|
||||
|
||||
// n = n - 0.5.
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP1, 0.5f);
|
||||
fsub(VTMP2.D(), VTMP2.D(), VTMP1.D());
|
||||
|
||||
// r = x - n*pi (range reduction), in extended precision.
|
||||
ldr(VTMP1.D(), &Pi1Label);
|
||||
fmsub(Accum.D(), VTMP2.D(), VTMP1.D(), Accum.D());
|
||||
ldr(VTMP1.D(), &Pi2Label);
|
||||
fmsub(Accum.D(), VTMP2.D(), VTMP1.D(), Accum.D());
|
||||
ldr(VTMP1.D(), &Pi3Label);
|
||||
fmsub(Accum.D(), VTMP2.D(), VTMP1.D(), Accum.D());
|
||||
|
||||
// sin(r) poly approx.
|
||||
fmul(VTMP1.D(), Accum.D(), Accum.D());
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, Accum.D());
|
||||
|
||||
// Horner: p = c6 + r2*(c5 + r2*(... + r2*c0)).
|
||||
ldr(VTMP2.D(), &C6Label);
|
||||
ldr(Accum.D(), &C5Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C4Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C3Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C2Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C1Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C0Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
|
||||
// y = r + r^3 * p.
|
||||
fmov(ARMEmitter::Size::i64Bit, Accum.D(), TMP2);
|
||||
fmul(VTMP1.D(), VTMP1.D(), Accum.D());
|
||||
fmadd(Accum.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
|
||||
// result = y XOR odd.
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, Accum.D());
|
||||
eor(ARMEmitter::Size::i64Bit, TMP2, TMP2, TMP1);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP2);
|
||||
|
||||
// restore nzcv
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
// Restore q2 and return.
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
// Fallback path.
|
||||
(void)Bind(&Fallback);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64COS].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64COS].Func));
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
// Constant pool.
|
||||
Align(16);
|
||||
(void)Bind(&InvPiLabel);
|
||||
dc64(0x3FD4'5F30'6DC9'C883ULL); // inv_pi
|
||||
(void)Bind(&Pi1Label);
|
||||
dc64(0x4009'21FB'5444'2D18ULL); // pi_1
|
||||
(void)Bind(&Pi2Label);
|
||||
dc64(0x3CA1'A626'3314'5C06ULL); // pi_2
|
||||
(void)Bind(&Pi3Label);
|
||||
dc64(0x395C'1CD1'2902'4E09ULL); // pi_3
|
||||
(void)Bind(&C0Label);
|
||||
dc64(0xBFC5'5555'5555'547BULL); // c0
|
||||
(void)Bind(&C1Label);
|
||||
dc64(0x3F81'1111'1110'8A4DULL); // c1
|
||||
(void)Bind(&C2Label);
|
||||
dc64(0xBF2A'01A0'1993'6F27ULL); // c2
|
||||
(void)Bind(&C3Label);
|
||||
dc64(0x3EC7'1DE3'7A97'D93EULL); // c3
|
||||
(void)Bind(&C4Label);
|
||||
dc64(0xBE5A'E633'9199'87C6ULL); // c4
|
||||
(void)Bind(&C5Label);
|
||||
dc64(0x3DE6'0E27'7AE0'7CECULL); // c5
|
||||
(void)Bind(&C6Label);
|
||||
dc64(0xBD69'E954'0300'A100ULL); // c6
|
||||
(void)Bind(&RangeLabel);
|
||||
dc64(0x4160'0000'0000'0000ULL); // 2^23
|
||||
}
|
||||
|
||||
void Dispatcher::EmitF64Tan() {
|
||||
F64TanHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
constexpr auto Accum = ARMEmitter::VReg::v2;
|
||||
|
||||
ARMEmitter::ForwardLabel Fallback, NonZero;
|
||||
ARMEmitter::ForwardLabel RangeLabel, TwoOverPiLabel;
|
||||
ARMEmitter::ForwardLabel HalfPi0Label, HalfPi1Label;
|
||||
ARMEmitter::ForwardLabel C0Label, C1Label, C2Label, C3Label, C4Label, C5Label, C6Label, C7Label, C8Label;
|
||||
|
||||
// tan(+/-0) = +/-0
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP1, VTMP1.D());
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &NonZero);
|
||||
ret();
|
||||
(void)Bind(&NonZero);
|
||||
|
||||
// Save q2 for use as accumulator
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
// save nzcv
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
// Range check: fall back for |x| >= 2^23, NaN, and inf.
|
||||
fabs(VTMP2.D(), VTMP1.D());
|
||||
ldr(Accum.D(), &RangeLabel);
|
||||
fcmp(VTMP2.D(), Accum.D());
|
||||
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
|
||||
|
||||
// q = nearest integer to 2 * x / pi.
|
||||
ldr(VTMP2.D(), &TwoOverPiLabel);
|
||||
fmul(VTMP2.D(), VTMP1.D(), VTMP2.D());
|
||||
frinta(VTMP2.D(), VTMP2.D());
|
||||
|
||||
// qi = int(q).
|
||||
fcvtzs(ARMEmitter::Size::i64Bit, TMP1, VTMP2.D());
|
||||
|
||||
// r = x - q * pi/2 (range reduction), in extended precision.
|
||||
fmov(Accum.D(), VTMP1.D());
|
||||
ldr(VTMP1.D(), &HalfPi0Label);
|
||||
fmsub(Accum.D(), VTMP2.D(), VTMP1.D(), Accum.D());
|
||||
ldr(VTMP1.D(), &HalfPi1Label);
|
||||
fmsub(Accum.D(), VTMP2.D(), VTMP1.D(), Accum.D());
|
||||
|
||||
// Further reduce r to [-pi/8, pi/8].
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP1, 0.5f);
|
||||
fmul(Accum.D(), Accum.D(), VTMP1.D());
|
||||
|
||||
// Approximate tan(r) using order 8 polynomial.
|
||||
fmul(VTMP1.D(), Accum.D(), Accum.D());
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, Accum.D());
|
||||
|
||||
// Horner: p = C8 + r2*(C7 + r2*(... + r2*C0)).
|
||||
ldr(VTMP2.D(), &C8Label);
|
||||
ldr(Accum.D(), &C7Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C6Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C5Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C4Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C3Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C2Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C1Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
ldr(Accum.D(), &C0Label);
|
||||
fmadd(VTMP2.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
|
||||
// p = r + r^3 * p.
|
||||
fmov(ARMEmitter::Size::i64Bit, Accum.D(), TMP2);
|
||||
fmul(VTMP1.D(), VTMP1.D(), Accum.D());
|
||||
fmadd(Accum.D(), VTMP1.D(), VTMP2.D(), Accum.D());
|
||||
|
||||
// Double-angle reconstruction: tan(2x) = 2*tan(x) / (1 - tan^2(x)).
|
||||
fadd(VTMP1.D(), Accum.D(), Accum.D());
|
||||
fmul(VTMP2.D(), Accum.D(), Accum.D());
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, Accum, 1.0f);
|
||||
fsub(VTMP2.D(), VTMP2.D(), Accum.D());
|
||||
|
||||
ARMEmitter::ForwardLabel SkipSwap;
|
||||
(void)tbnz(TMP1, 0, &SkipSwap);
|
||||
|
||||
fneg(Accum.D(), VTMP1.D());
|
||||
fmov(VTMP1.D(), VTMP2.D());
|
||||
fmov(VTMP2.D(), Accum.D());
|
||||
|
||||
(void)Bind(&SkipSwap);
|
||||
|
||||
// result = numerator / denominator -> VTMP1.
|
||||
fdiv(VTMP1.D(), VTMP2.D(), VTMP1.D());
|
||||
|
||||
// restore nzcv
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
// Restore q2 and return.
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
// Fallback path.
|
||||
(void)Bind(&Fallback);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64TAN].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64TAN].Func));
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
// Constant pool.
|
||||
Align(16);
|
||||
(void)Bind(&TwoOverPiLabel);
|
||||
dc64(0x3FE4'5F30'6DC9'C883ULL); // two_over_pi
|
||||
(void)Bind(&HalfPi0Label);
|
||||
dc64(0x3FF9'21FB'5444'2D18ULL); // half_pi[0]
|
||||
(void)Bind(&HalfPi1Label);
|
||||
dc64(0x3C91'A626'3314'5C07ULL); // half_pi[1]
|
||||
(void)Bind(&C0Label);
|
||||
dc64(0x3FD5'5555'5555'5556ULL); // C0
|
||||
(void)Bind(&C1Label);
|
||||
dc64(0x3FC1'1111'1111'0A63ULL); // C1
|
||||
(void)Bind(&C2Label);
|
||||
dc64(0x3FAB'A1BA'1BB4'6414ULL); // C2
|
||||
(void)Bind(&C3Label);
|
||||
dc64(0x3F96'64F4'7E5B'5445ULL); // C3
|
||||
(void)Bind(&C4Label);
|
||||
dc64(0x3F82'26E5'E5EC'DFA3ULL); // C4
|
||||
(void)Bind(&C5Label);
|
||||
dc64(0x3F6D'6C7D'DBF8'7047ULL); // C5
|
||||
(void)Bind(&C6Label);
|
||||
dc64(0x3F57'EA75'D05B'583EULL); // C6
|
||||
(void)Bind(&C7Label);
|
||||
dc64(0x3F42'89F2'2964'A03CULL); // C7
|
||||
(void)Bind(&C8Label);
|
||||
dc64(0x3F34'E4FD'1414'7622ULL); // C8
|
||||
(void)Bind(&RangeLabel);
|
||||
dc64(0x4160'0000'0000'0000ULL); // 2^23
|
||||
}
|
||||
|
||||
uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
auto Address = GetCursorAddress<uint64_t>();
|
||||
constexpr static auto FallbackPointerReg = TMP4;
|
||||
@@ -1289,6 +1690,9 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
|
||||
Ptrs.SignalReturnHandlerRT = SignalHandlerReturnAddressRT;
|
||||
Ptrs.LUDIVHandler = LUDIVHandlerAddress;
|
||||
Ptrs.LDIVHandler = LDIVHandlerAddress;
|
||||
Ptrs.F64SinHandler = F64SinHandlerAddress;
|
||||
Ptrs.F64CosHandler = F64CosHandlerAddress;
|
||||
Ptrs.F64TanHandler = F64TanHandlerAddress;
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Ptrs.FallbackHandlerPointers, &ABIPointers[0]);
|
||||
|
||||
@@ -28,6 +28,10 @@ class ContextImpl;
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) STATE.R(), offsetof(FEXCore::Core::STATE_TYPE, FIELD)
|
||||
#define STATE_PTR_IDX(STATE_TYPE, FIELD, INDEX) STATE.R(), ARRAY_OFFSETOF(FEXCore::Core::STATE_TYPE, FIELD, INDEX)
|
||||
#define FALLBACK_HANDLER_OFFSET(INDEX, FIELD) \
|
||||
STATE.R(), \
|
||||
(ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, Pointers.FallbackHandlerPointers, INDEX) + offsetof(FEXCore::Core::FallbackABIInfo, FIELD))
|
||||
|
||||
class Dispatcher final : public Arm64Emitter {
|
||||
public:
|
||||
@@ -95,6 +99,11 @@ private:
|
||||
uint64_t LUDIVHandlerAddress {};
|
||||
uint64_t LDIVHandlerAddress {};
|
||||
|
||||
// F64 trig shared handlers
|
||||
uint64_t F64SinHandlerAddress {};
|
||||
uint64_t F64CosHandlerAddress {};
|
||||
uint64_t F64TanHandlerAddress {};
|
||||
|
||||
void EmitDispatcher();
|
||||
uint64_t GenerateABICall(FallbackABI ABI);
|
||||
|
||||
@@ -105,6 +114,10 @@ private:
|
||||
void EmitF32ToExtF80();
|
||||
void EmitF64ToExtF80();
|
||||
|
||||
void EmitF64Sin();
|
||||
void EmitF64Cos();
|
||||
void EmitF64Tan();
|
||||
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
|
||||
|
||||
@@ -1348,6 +1348,13 @@ const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
bool Decoder::CheckIfCacheable(FEXCore::Core::InternalThreadState& Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
DecodeInstructionsAtEntry(&Thread, InstStream, PC, MaxInst);
|
||||
bool Uncacheable = HitBadRelocation;
|
||||
DelayedDisownBuffer();
|
||||
return !Uncacheable;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
@@ -1465,6 +1472,13 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
}
|
||||
|
||||
BlockIt->BlockStatus = DecodeInstruction(OpAddress);
|
||||
if (HitBadRelocation) {
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks = {*BlockIt};
|
||||
BlockInfo.EntryPoints.clear();
|
||||
BlockInfo.CodePages.clear();
|
||||
return;
|
||||
}
|
||||
uint64_t OpEndAddress = OpAddress + DecodeInst->InstSize;
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, OpAddress);
|
||||
@@ -1483,7 +1497,7 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (BlockIt->BlockStatus != DecodedBlockStatus::SUCCESS) [[unlikely]] {
|
||||
if (!EntryBlock) {
|
||||
if (!EntryBlock && BlockIt->BlockStatus != DecodedBlockStatus::BAD_RELOCATION) {
|
||||
// In multiblock configurations, we can early terminate any non-entrypoint blocks with the expectation that this won't get hit.
|
||||
// Improves compile-times.
|
||||
// Just need to undo additions that this block decoding has caused.
|
||||
|
||||
@@ -54,6 +54,7 @@ public:
|
||||
};
|
||||
|
||||
Decoder(FEXCore::Core::InternalThreadState* Thread);
|
||||
bool CheckIfCacheable(FEXCore::Core::InternalThreadState&, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
const DecodedBlockInformation* GetDecodedBlockInfo() const {
|
||||
|
||||
@@ -329,7 +329,7 @@ DEF_OP(TelemetrySetValue) {
|
||||
auto Op = IROp->C<IR::IROp_TelemetrySetValue>();
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.TelemetryValueAddresses[Op->TelemetryValueIndex]));
|
||||
ldr(TMP2, STATE_PTR_IDX(CpuStateFrame, Pointers.TelemetryValueAddresses, Op->TelemetryValueIndex));
|
||||
|
||||
// Cortex fuses cmp+cset.
|
||||
cmp(ARMEmitter::Size::i32Bit, Src, 0);
|
||||
|
||||
@@ -265,7 +265,10 @@ DEF_OP(Syscall) {
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
|
||||
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
|
||||
SpillStaticRegs(TMP1, {
|
||||
.GPRSpillMask = GPRSpillMask,
|
||||
.FPRSpillMask = FPRSpillMask,
|
||||
});
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -299,7 +302,12 @@ DEF_OP(Syscall) {
|
||||
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r1,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
.GPRFillMask = GPRSpillMask,
|
||||
.FPRFillMask = FPRSpillMask,
|
||||
});
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
@@ -322,7 +330,10 @@ DEF_OP(Thunk) {
|
||||
// X0: CTX
|
||||
// X1: Args (from guest stack)
|
||||
|
||||
SpillStaticRegs(TMP1, true, ~0U, ~0U, false); // spill to ctx before ra64 spill
|
||||
// spill to ctx before ra64 spill
|
||||
SpillStaticRegs(TMP1, {
|
||||
.NZCV = false,
|
||||
});
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
@@ -337,7 +348,10 @@ DEF_OP(Thunk) {
|
||||
|
||||
PopDynamicRegs();
|
||||
|
||||
FillStaticRegs(true, ~0U, ~0U, std::nullopt, std::nullopt, false); // load from ctx after ra64 refill
|
||||
// load from ctx after ra64 refill
|
||||
FillStaticRegs({
|
||||
.NZCV = false,
|
||||
});
|
||||
}
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
|
||||
@@ -68,6 +68,10 @@ PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
static void PrintMsg(const char* Value) {
|
||||
LogMan::Msg::DFmt("{}", Value);
|
||||
}
|
||||
|
||||
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
|
||||
}
|
||||
@@ -133,8 +137,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.S(), Src1.S());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -151,8 +155,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -176,8 +180,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
mov(ARMEmitter::Size::i32Bit, TMP2, Src1);
|
||||
}
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -194,8 +198,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -212,8 +216,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -230,8 +234,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -254,8 +258,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -276,8 +280,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -294,8 +298,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -312,8 +316,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -330,8 +334,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -351,8 +355,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -369,8 +373,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -394,8 +398,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -416,8 +420,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -434,8 +438,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
// tmp2 (x1/x11): source 2
|
||||
// tmp3 (x2/x12): source 3
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP1, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
|
||||
stp<ARMEmitter::IndexType::PRE>(TMP1, ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
@@ -476,8 +480,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
movz(ARMEmitter::Size::i32Bit, TMP1, Control);
|
||||
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
ldr(TMP2, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, ABIHandler));
|
||||
ldr(TMP4, FALLBACK_HANDLER_OFFSET(Info.HandlerIndex, Func));
|
||||
blr(TMP2);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
@@ -636,6 +640,8 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
|
||||
Ptrs.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Ptrs.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Ptrs.PrintMsgValue = reinterpret_cast<uint64_t>(PrintMsg);
|
||||
|
||||
Ptrs.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
|
||||
Ptrs.MonoBackpatcherWrite = reinterpret_cast<uint64_t>(&Context::ContextImpl::MonoBackpatcherWrite);
|
||||
Ptrs.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
@@ -841,6 +847,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
CallReturnTargets.clear();
|
||||
PendingJumpThunks.clear();
|
||||
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
|
||||
Relocations.resize(PrevNumAllocations, FEXCore::CPU::Relocation::Default()); // Discard any relocations generated from a previous attempt
|
||||
|
||||
CodeData.EntryPoints.clear();
|
||||
|
||||
|
||||
@@ -563,7 +563,7 @@ DEF_OP(LoadDF) {
|
||||
auto Flag = X86State::RFLAG_DF_RAW_LOC;
|
||||
|
||||
// DF needs sign extension to turn 0x1/0xFF into 1/-1
|
||||
ldrsb(Dst.X(), STATE, offsetof(FEXCore::Core::CPUState, flags[Flag]));
|
||||
ldrsb(Dst.X(), STATE, ARRAY_OFFSETOF(FEXCore::Core::CPUState, flags, Flag));
|
||||
}
|
||||
|
||||
DEF_OP(ContextClear) {
|
||||
@@ -1849,13 +1849,6 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
|
||||
DEF_OP(MemSet) {
|
||||
// TODO: A future looking task would be to support this with ARM's MOPS instructions.
|
||||
// The 8-bit non-atomic forward path directly matches ARM's SETP/SETM/SETE instruction,
|
||||
// while the backward version needs some fixup to convert it to a forward direction.
|
||||
//
|
||||
// Assuming non-atomicity and non-faulting behaviour, this can accelerate this implementation.
|
||||
// Additionally: This is commonly used as a memset to zero. If we know up-front with an inline constant
|
||||
// that the value is zero, we can optimize any operation larger than 8-bit down to 8-bit to use the MOPS implementation.
|
||||
const auto Op = IROp->C<IR::IROp_MemSet>();
|
||||
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
@@ -1933,8 +1926,30 @@ DEF_OP(MemSet) {
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
auto EmitMemset = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
const bool IsBackwards = Direction == -1;
|
||||
|
||||
// Sets the result to the final address written depending on
|
||||
// whether or not the memset is forwards or backwards.
|
||||
const auto MakeFinalAddress = [&] {
|
||||
if (IsBackwards) {
|
||||
switch (Size) {
|
||||
case 1: sub(Dst.X(), MemReg.X(), Length.X()); break;
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: sub(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled MemSet size: {}", Size); break;
|
||||
}
|
||||
} else {
|
||||
switch (Size) {
|
||||
case 1: add(Dst.X(), MemReg.X(), Length.X()); break;
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: add(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled MemSet size: {}", Size); break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel AgainInternal {};
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
@@ -1943,12 +1958,56 @@ DEF_OP(MemSet) {
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
if (CTX->HostFeatures.SupportsMOPS) {
|
||||
const bool Is8Bit = SubRegSize == ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
// We can handle 8-bit memsets and any other size that happens
|
||||
// to be using an inlined zero value (resulting in the use of ZR).
|
||||
//
|
||||
// NOTE:
|
||||
// Strictly speaking, this can also be trivially expanded to handle other sizes
|
||||
// that happen to use any value that could fit inside a byte if the need
|
||||
// arises. This does increase branching and code generation, however, since
|
||||
// we'd still need to emit the fallback in the event a value for a larger size
|
||||
// falls outside the range of a byte instead of only generating the MOPS code.
|
||||
if (Is8Bit || Value == ARMEmitter::Reg::zr) {
|
||||
// If we're performing a non-byte-sized zeroing operation then we need to
|
||||
// scale the counter accordingly. (e.g. a 64-bit memset of size 2 needs to
|
||||
// be turned into an 8-bit memset of size 16)
|
||||
if (!Is8Bit) {
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP1, TMP1, FEXCore::ToUnderlying(SubRegSize));
|
||||
}
|
||||
|
||||
// If backwards, then we need to adjust the starting address because
|
||||
// set{p, m, e} memset forwards, so we need to slide this bad boy
|
||||
// back like: (address - count) + 1.
|
||||
//
|
||||
// This lets us offset the address such that we can treat a backwards
|
||||
// memset as if it were a forwards one.
|
||||
if (IsBackwards) {
|
||||
sub(TMP2, TMP2, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
}
|
||||
|
||||
// Unfortunately set operations fiddle with NZCV, so we need to preserve it.
|
||||
mrs(TMP3, ARMEmitter::SystemRegister::NZCV);
|
||||
setp(TMP2, TMP1, Value.X());
|
||||
setm(TMP2, TMP1, Value.X());
|
||||
sete(TMP2, TMP1, Value.X());
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
|
||||
|
||||
MakeFinalAddress();
|
||||
(void)Bind(&DoneInternal);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
ARMEmitter::ForwardLabel AgainInternal256Exit {};
|
||||
ARMEmitter::BackwardLabel AgainInternal256 {};
|
||||
ARMEmitter::ForwardLabel AgainInternal128Exit {};
|
||||
ARMEmitter::BackwardLabel AgainInternal128 {};
|
||||
|
||||
if (Direction == -1) {
|
||||
if (IsBackwards) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
}
|
||||
|
||||
@@ -1986,39 +2045,23 @@ DEF_OP(MemSet) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
if (IsBackwards) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
}
|
||||
}
|
||||
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemStoreTSO(Value, OpSize, SizeDirection);
|
||||
MemStoreTSO(Value, Size, SizeDirection);
|
||||
} else {
|
||||
MemStore(Value, OpSize, SizeDirection);
|
||||
MemStore(Value, Size, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
|
||||
(void)Bind(&DoneInternal);
|
||||
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
case 1: add(Dst.X(), MemReg.X(), Length.X()); break;
|
||||
case 2: add(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 1); break;
|
||||
case 4: add(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 2); break;
|
||||
case 8: add(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 3); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case 1: sub(Dst.X(), MemReg.X(), Length.X()); break;
|
||||
case 2: sub(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 1); break;
|
||||
case 4: sub(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 2); break;
|
||||
case 8: sub(Dst.X(), MemReg.X(), Length.X(), ARMEmitter::ShiftType::LSL, 3); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
}
|
||||
MakeFinalAddress();
|
||||
};
|
||||
|
||||
if (DirectionIsInline) {
|
||||
@@ -2041,10 +2084,6 @@ DEF_OP(MemSet) {
|
||||
}
|
||||
|
||||
DEF_OP(MemCpy) {
|
||||
// TODO: A future looking task would be to support this with ARM's MOPS instructions.
|
||||
// The 8-bit non-atomic path directly matches ARM's CPYP/CPYM/CPYE instruction,
|
||||
//
|
||||
// Assuming non-atomicity and non-faulting behaviour, this can accelerate this implementation.
|
||||
const auto Op = IROp->C<IR::IROp_MemCpy>();
|
||||
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
@@ -2175,8 +2214,40 @@ DEF_OP(MemCpy) {
|
||||
};
|
||||
|
||||
auto EmitMemcpy = [&](int32_t Direction) {
|
||||
const int32_t OpSize = Size;
|
||||
const int32_t SizeDirection = Size * Direction;
|
||||
const bool IsBackwards = Direction == -1;
|
||||
|
||||
const auto FinalizeAddresses = [&] {
|
||||
if (IsBackwards) {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
sub(Dst0.X(), TMP1, TMP3);
|
||||
sub(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
case 4:
|
||||
case 8:
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size));
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled MemCpy size: {}", Size); break;
|
||||
}
|
||||
} else {
|
||||
switch (Size) {
|
||||
case 1:
|
||||
add(Dst0.X(), TMP1, TMP3);
|
||||
add(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
case 4:
|
||||
case 8:
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size));
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Size));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled MemCpy size: {}", Size); break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel AgainInternal {};
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
@@ -2185,6 +2256,48 @@ DEF_OP(MemCpy) {
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
if (CTX->HostFeatures.SupportsMOPS) {
|
||||
// In the event we have an overlap (gross), we need to fall back
|
||||
// to the non-mops copy handler. Since the overlap check needs to
|
||||
// make use of NZCV, we need to save it. This can be avoided with
|
||||
// ARMv9.6+'s FEAT_CMPBR, but alas, we don't have access to that right now.
|
||||
//
|
||||
// NOTE: That we need to temporarily trash TMP1 and restore it after the
|
||||
// comparison.
|
||||
ARMEmitter::ForwardLabel OverlapCase;
|
||||
mrs(TMP4, ARMEmitter::SystemRegister::NZCV);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP2, TMP3);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP1, Length.X());
|
||||
mov(TMP1, Length.X());
|
||||
(void)bc(ARMEmitter::Condition::CC_LT, &OverlapCase);
|
||||
|
||||
// If doing something larger than a byte copy, then we need to scale
|
||||
// the counter value accordingly to convert it to bytes.
|
||||
if (Size > 1) {
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP1, TMP1, FEXCore::ilog2(Size));
|
||||
}
|
||||
|
||||
// Adjust addresses so that we treat the backward copy as a forward copy
|
||||
if (IsBackwards) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, TMP1);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP3, TMP3, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, Size);
|
||||
add(ARMEmitter::Size::i64Bit, TMP3, TMP3, Size);
|
||||
}
|
||||
|
||||
// Unfortunately copy operations fiddle with NZCV, so we need to preserve it.
|
||||
cpyfp(TMP2, TMP3, TMP1);
|
||||
cpyfm(TMP2, TMP3, TMP1);
|
||||
cpyfe(TMP2, TMP3, TMP1);
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP4);
|
||||
|
||||
(void)b(&DoneInternal);
|
||||
|
||||
// Turns out we overlap and need to fall back. Make sure to restore NZCV.
|
||||
(void)Bind(&OverlapCase);
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP4);
|
||||
}
|
||||
|
||||
ARMEmitter::ForwardLabel AbsPos {};
|
||||
ARMEmitter::ForwardLabel AgainInternal256Exit {};
|
||||
ARMEmitter::ForwardLabel AgainInternal128Exit {};
|
||||
@@ -2198,7 +2311,7 @@ DEF_OP(MemCpy) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
|
||||
(void)tbnz(TMP4, 63, &AgainInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
if (IsBackwards) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32 - Size);
|
||||
}
|
||||
@@ -2233,7 +2346,7 @@ DEF_OP(MemCpy) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
if (IsBackwards) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
add(ARMEmitter::Size::i64Bit, TMP3, TMP3, 32 - Size);
|
||||
}
|
||||
@@ -2241,9 +2354,9 @@ DEF_OP(MemCpy) {
|
||||
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemCpyTSO(OpSize, SizeDirection);
|
||||
MemCpyTSO(Size, SizeDirection);
|
||||
} else {
|
||||
MemCpy(OpSize, SizeDirection);
|
||||
MemCpy(Size, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
@@ -2255,54 +2368,14 @@ DEF_OP(MemCpy) {
|
||||
mov(TMP2, MemRegSrc.X());
|
||||
mov(TMP3, Length.X());
|
||||
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
add(Dst0.X(), TMP1, TMP3);
|
||||
add(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
add(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
add(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
sub(Dst0.X(), TMP1, TMP3);
|
||||
sub(Dst1.X(), TMP2, TMP3);
|
||||
break;
|
||||
case 2:
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 1);
|
||||
break;
|
||||
case 4:
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 2);
|
||||
break;
|
||||
case 8:
|
||||
sub(Dst0.X(), TMP1, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
sub(Dst1.X(), TMP2, TMP3, ARMEmitter::ShiftType::LSL, 3);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); break;
|
||||
}
|
||||
}
|
||||
FinalizeAddresses();
|
||||
};
|
||||
|
||||
if (DirectionIsInline) {
|
||||
LOGMAN_THROW_A_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
EmitMemcpy(DirectionConstant);
|
||||
} else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
// Emit forward direction memcpy then backward direction memcpy.
|
||||
for (int32_t Direction : {1, -1}) {
|
||||
EmitMemcpy(Direction);
|
||||
if (Direction == 1) {
|
||||
|
||||
@@ -210,6 +210,25 @@ DEF_OP(Print) {
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(PrintMsg) {
|
||||
auto Op = IROp->C<IR::IROp_PrintMsg>();
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, reinterpret_cast<uintptr_t>(Op->Value));
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.PrintMsgValue));
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
if (CTX->HostFeatures.SupportsCPUIndexInTPIDRRO) {
|
||||
mrs(GetReg(Node), ARMEmitter::SystemRegister::TPIDRRO_EL0);
|
||||
@@ -227,7 +246,10 @@ DEF_OP(ProcessorID) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
SpillStaticRegs(TMP1, {
|
||||
.GPRSpillMask = SpillMask,
|
||||
.FPRs = false,
|
||||
});
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -264,7 +286,13 @@ DEF_OP(ProcessorID) {
|
||||
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask, ~0U, ARMEmitter::Reg::r8, ARMEmitter::Reg::r2);
|
||||
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r8,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
.GPRFillMask = SpillMask,
|
||||
.FPRs = false,
|
||||
});
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
|
||||
@@ -977,7 +977,7 @@ DEF_OP(LoadNamedVectorConstant) {
|
||||
}
|
||||
// Load the pointer.
|
||||
auto GenerateMemOperand = [this](IR::OpSize OpSize, uint32_t NamedConstant, ARMEmitter::Register Base) {
|
||||
const auto ConstantOffset = offsetof(FEXCore::Core::CpuStateFrame, Pointers.NamedVectorConstants[NamedConstant]);
|
||||
const auto ConstantOffset = ARRAY_OFFSETOF(FEXCore::Core::CpuStateFrame, Pointers.NamedVectorConstants, NamedConstant);
|
||||
|
||||
if (ConstantOffset <= 255 || // Unscaled 9-bit signed
|
||||
((ConstantOffset & (IR::OpSizeToSize(OpSize) - 1)) == 0 &&
|
||||
@@ -985,13 +985,13 @@ DEF_OP(LoadNamedVectorConstant) {
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, ConstantOffset);
|
||||
}
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.NamedVectorConstantPointers[NamedConstant]));
|
||||
ldr(TMP1, STATE_PTR_IDX(CpuStateFrame, Pointers.NamedVectorConstantPointers, NamedConstant));
|
||||
return ARMEmitter::ExtendedMemOperand(TMP1, ARMEmitter::IndexType::OFFSET, 0);
|
||||
};
|
||||
|
||||
if (OpSize == IR::OpSize::i256Bit) {
|
||||
// Handle SVE 32-byte variant upfront.
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.NamedVectorConstantPointers[Op->Constant]));
|
||||
ldr(TMP1, STATE_PTR_IDX(CpuStateFrame, Pointers.NamedVectorConstantPointers, Op->Constant));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), TMP1, 0);
|
||||
return;
|
||||
}
|
||||
@@ -1013,7 +1013,7 @@ DEF_OP(LoadNamedVectorIndexedConstant) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
// Load the pointer.
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.IndexedNamedVectorConstantPointers[Op->Constant]));
|
||||
ldr(TMP1, STATE_PTR_IDX(CpuStateFrame, Pointers.IndexedNamedVectorConstantPointers, Op->Constant));
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: ldrb(Dst, TMP1, Op->Index); break;
|
||||
@@ -4611,4 +4611,44 @@ DEF_OP(VFCopySign) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F64SIN) {
|
||||
const auto Op = IROp->C<IR::IROp_F64SIN>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64SinHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64COS) {
|
||||
const auto Op = IROp->C<IR::IROp_F64COS>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64CosHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64TAN) {
|
||||
const auto Op = IROp->C<IR::IROp_F64TAN>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64TanHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -41,6 +41,9 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize, false, false));
|
||||
LOGMAN_THROW_A_FMT(PagePointer != -1ULL, "Failed to allocate PagePointer");
|
||||
|
||||
// Disable THP on the Lookup cache.
|
||||
FEXCore::Allocator::VirtualTHPControl(reinterpret_cast<const void*>(PagePointer), TotalCacheSize, FEXCore::Allocator::THPControl::Disable);
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Lookup", reinterpret_cast<void*>(PagePointer),
|
||||
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE);
|
||||
CTX->SyscallHandler->MarkOvercommitRange(PagePointer, TotalCacheSize);
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/SHMStats.h>
|
||||
#include "Utils/WritePriorityMutex.h"
|
||||
#include <FEXCore/Utils/WritePriorityMutex.h>
|
||||
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
|
||||
@@ -2643,7 +2643,10 @@ void OpDispatchBuilder::IMULOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
// 64-bit special cased to save a move
|
||||
Ref Result = Size < OpSize::i64Bit ? _Mul(OpSize::i64Bit, Src1, Src2) : nullptr;
|
||||
Ref Result {};
|
||||
if (Size < OpSize::i64Bit) {
|
||||
Result = _Mul(OpSize::i64Bit, Src1, Src2);
|
||||
}
|
||||
Ref ResultHigh {};
|
||||
if (Size == OpSize::i8Bit) {
|
||||
// Result is stored in AX
|
||||
@@ -4487,8 +4490,6 @@ void OpDispatchBuilder::StoreResult(RegClass Class, X86Tables::DecodedOp Op, Ref
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx)
|
||||
: IREmitter {ctx->OpDispatcherAllocator, ctx->HostFeatures.SupportsTSOImm9}
|
||||
, CTX {ctx} {
|
||||
ResetWorkingList();
|
||||
|
||||
if (CTX->HostFeatures.SupportsAVX && CTX->HostFeatures.SupportsSVE256) {
|
||||
SaveAVXStateFunc = &OpDispatchBuilder::SaveAVXState;
|
||||
RestoreAVXStateFunc = &OpDispatchBuilder::RestoreAVXState;
|
||||
@@ -4501,7 +4502,8 @@ OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx)
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ResetWorkingList() {
|
||||
IREmitter::ResetWorkingList();
|
||||
IREmitter::ReownOrClaimBuffer();
|
||||
|
||||
JumpTargets.clear();
|
||||
BlockSetRIP = false;
|
||||
DecodeFailure = false;
|
||||
@@ -4887,6 +4889,11 @@ void OpDispatchBuilder::CLZeroOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::Prefetch(OpcodeArgs, bool ForStore, bool Stream, uint8_t Level) {
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
// NOP instance.
|
||||
return;
|
||||
}
|
||||
|
||||
Ref DestMem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
_Prefetch(ForStore, Stream, Level, DestMem, Invalid(), MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
@@ -305,7 +305,9 @@ public:
|
||||
|
||||
OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
// Should only be called at the start of IR Emission.
|
||||
void ResetWorkingList();
|
||||
|
||||
void ResetDecodeFailure() {
|
||||
NeedsBlockEnd = DecodeFailure = false;
|
||||
}
|
||||
@@ -1666,7 +1668,7 @@ private:
|
||||
[[nodiscard]]
|
||||
static uint32_t GPROffset(X86State::X86Reg reg) {
|
||||
LOGMAN_THROW_A_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
return static_cast<uint32_t>(offsetof(Core::CPUState, gregs[static_cast<size_t>(reg)]));
|
||||
return static_cast<uint32_t>(ARRAY_OFFSETOF(Core::CPUState, gregs, reg));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -1885,15 +1887,15 @@ private:
|
||||
// For DF, we need to transform 0/1 into 1/-1
|
||||
StoreDF(_SubShift(OpSize::i64Bit, Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
|
||||
auto PackedTF = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
auto PackedTF = _LoadContextGPR(OpSize::i8Bit, ARRAY_OFFSETOF(FEXCore::Core::CPUState, flags, BitOffset));
|
||||
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
|
||||
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
|
||||
// unblocked bit before raising the exception.
|
||||
auto NewPackedTF =
|
||||
_Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
|
||||
_StoreContextGPR(OpSize::i8Bit, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
_StoreContextGPR(OpSize::i8Bit, NewPackedTF, ARRAY_OFFSETOF(FEXCore::Core::CPUState, flags, BitOffset));
|
||||
} else {
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, ARRAY_OFFSETOF(FEXCore::Core::CPUState, flags, BitOffset));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1948,8 +1950,8 @@ private:
|
||||
[[nodiscard]]
|
||||
static uint32_t CacheIndexToContextOffset(int Index) {
|
||||
switch (Index) {
|
||||
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
|
||||
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
|
||||
case MM0Index ... MM7Index: return ARRAY_OFFSETOF(FEXCore::Core::CPUState, mm, Index - MM0Index);
|
||||
case AVXHigh0Index ... AVXHigh15Index: return ARRAY_OFFSETOF(FEXCore::Core::CPUState, avx_high, Index - AVXHigh0Index);
|
||||
default: return ~0U;
|
||||
}
|
||||
}
|
||||
@@ -2149,7 +2151,7 @@ private:
|
||||
// Recover the sign bit, it is the logical DF value
|
||||
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63));
|
||||
} else {
|
||||
return _LoadContextGPR(OpSize::i8Bit, offsetof(Core::CPUState, flags[BitOffset]));
|
||||
return _LoadContextGPR(OpSize::i8Bit, ARRAY_OFFSETOF(Core::CPUState, flags, BitOffset));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1347,6 +1347,11 @@ Ref OpDispatchBuilder::SHUFOpImpl(OpcodeArgs, IR::OpSize DstSize, IR::OpSize Ele
|
||||
Shuffle >>= ShiftAmount;
|
||||
}
|
||||
} else {
|
||||
if (Src1 == Src2 && Shuffle == 0) {
|
||||
// TODO: We can optimize significantly more shuffles when we know the sources match.
|
||||
// Special case broadcast element 0.
|
||||
return _VDupElement(DstSize, ElementSize, Src1, Shuffle & SelectionMask);
|
||||
}
|
||||
if (ElementSize == OpSize::i32Bit) {
|
||||
// We can shuffle optimally in a lot of cases.
|
||||
// TODO: We can optimize more of these cases.
|
||||
@@ -2756,7 +2761,7 @@ void OpDispatchBuilder::SaveSSEState(Ref MemBase) {
|
||||
|
||||
void OpDispatchBuilder::SaveMXCSRState(Ref MemBase) {
|
||||
// Store MXCSR and the mask for all bits.
|
||||
_StoreMemPairGPR(OpSize::i32Bit, GetMXCSR(), Constant(0xFFFF), MemBase, 24);
|
||||
_StoreMemPairGPR(OpSize::i32Bit, GetMXCSR(), Constant(0xFFC0), MemBase, 24);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveAVXState(Ref MemBase) {
|
||||
|
||||
@@ -50,7 +50,7 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> SecondGroup_ArchSelect_LUT = {{
|
||||
},
|
||||
}};
|
||||
|
||||
constexpr auto SecondInstGroupOps = []() consteval {
|
||||
constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> Table{};
|
||||
constexpr U16U8InfoStruct SecondaryExtensionOpTable[] = {
|
||||
// GROUP 1
|
||||
@@ -402,37 +402,37 @@ constexpr auto SecondInstGroupOps = []() consteval {
|
||||
// GROUP 16
|
||||
// AMD documentation claims again that this entire group is n/a to prefix
|
||||
// Tooling once again fails to disassemble oens with the prefix. Disable until proven otherwise
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 0), 1, X86InstInfo{"PREFETCH NTA", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 1), 1, X86InstInfo{"PREFETCH T0", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 2), 1, X86InstInfo{"PREFETCH T1", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 3), 1, X86InstInfo{"PREFETCH T2", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 0), 1, X86InstInfo{"PREFETCH NTA", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 1), 1, X86InstInfo{"PREFETCH T0", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 2), 1, X86InstInfo{"PREFETCH T1", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 3), 1, X86InstInfo{"PREFETCH T2", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 4), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 5), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 6), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_NONE, 7), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 0), 1, X86InstInfo{"PREFETCH NTA", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 1), 1, X86InstInfo{"PREFETCH T0", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 2), 1, X86InstInfo{"PREFETCH T1", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 3), 1, X86InstInfo{"PREFETCH T2", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 0), 1, X86InstInfo{"PREFETCH NTA", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 1), 1, X86InstInfo{"PREFETCH T0", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 2), 1, X86InstInfo{"PREFETCH T1", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 3), 1, X86InstInfo{"PREFETCH T2", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 4), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 5), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 6), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F3, 7), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_16, PF_66, 0), 1, X86InstInfo{"PREFETCH NTA", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 1), 1, X86InstInfo{"PREFETCH T0", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 2), 1, X86InstInfo{"PREFETCH T1", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 3), 1, X86InstInfo{"PREFETCH T2", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 0), 1, X86InstInfo{"PREFETCH NTA", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 1), 1, X86InstInfo{"PREFETCH T0", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 2), 1, X86InstInfo{"PREFETCH T1", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 3), 1, X86InstInfo{"PREFETCH T2", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 4), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 5), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 6), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_66, 7), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 0), 1, X86InstInfo{"PREFETCH NTA", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 1), 1, X86InstInfo{"PREFETCH T0", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 2), 1, X86InstInfo{"PREFETCH T1", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 3), 1, X86InstInfo{"PREFETCH T2", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 0), 1, X86InstInfo{"PREFETCH NTA", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 1), 1, X86InstInfo{"PREFETCH T0", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 2), 1, X86InstInfo{"PREFETCH T1", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 3), 1, X86InstInfo{"PREFETCH T2", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 4), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 5), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(TYPE_GROUP_16, PF_F2, 6), 1, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
|
||||
@@ -136,6 +136,7 @@
|
||||
"u16": "uint16_t",
|
||||
"u32": "uint32_t",
|
||||
"u64": "uint64_t",
|
||||
"c_str": "const char*",
|
||||
"OpSize": "FEXCore::IR::OpSize",
|
||||
"SSA": "OrderedNode*",
|
||||
"GPR": "OrderedNode*",
|
||||
@@ -240,6 +241,12 @@
|
||||
"Desc": ["Debug operation that prints an SSA value to the console",
|
||||
"May only print 64bits of the value"]
|
||||
},
|
||||
"PrintMsg c_str:$Value": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Debug operation that prints an string to the console.",
|
||||
"This is for debug only! Will break code caching!"
|
||||
]
|
||||
},
|
||||
"GPR = AllocateGPR i1:$ForPair": {
|
||||
"Desc": ["Silly pseudo-instruction to allocate a register for a future destination",
|
||||
"Note: if an instruction uses allocated destinations-as-sources,",
|
||||
@@ -2775,15 +2782,15 @@
|
||||
},
|
||||
"FPR = F64TAN FPR:$Src": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
"JITDispatch": true
|
||||
},
|
||||
"FPR = F64SIN FPR:$Src": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
"JITDispatch": true
|
||||
},
|
||||
"FPR = F64COS FPR:$Src": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
"JITDispatch": true
|
||||
},
|
||||
"FPR:$Sin, FPR:$Cos = F64SINCOS FPR:$Src": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
|
||||
@@ -38,6 +38,10 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, uint64_t Arg)
|
||||
*out << fextl::fmt::format("#{:#x}", Arg);
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, const char* const Arg) {
|
||||
*out << fextl::fmt::format("'{}'", Arg);
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg) {
|
||||
if (Arg == CondClass::AL) {
|
||||
*out << "ALWAYS";
|
||||
@@ -356,8 +360,8 @@ void Dump(fextl::stringstream* out, const IRListView* IR) {
|
||||
|
||||
++CurrentIndent;
|
||||
AddIndent();
|
||||
*out << fextl::fmt::format("(%0) IRHeader %{}, #{:#x}, #{}, #{}\n", HeaderOp->Blocks.ID(), HeaderOp->OriginalRIP, HeaderOp->BlockCount,
|
||||
HeaderOp->NumHostInstructions);
|
||||
*out << fextl::fmt::format("(%0) IRHeader %{}, #{:#x}, #{}, #{}\n", HeaderOp->Blocks.ID(), +HeaderOp->OriginalRIP, +HeaderOp->BlockCount,
|
||||
+HeaderOp->NumHostInstructions);
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
{
|
||||
|
||||
@@ -21,15 +21,15 @@ class IREmitter {
|
||||
public:
|
||||
IREmitter(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, bool SupportsTSOImm9)
|
||||
: DualListData {ThreadAllocator, 8 * 1024 * 1024}
|
||||
, SupportsTSOImm9(SupportsTSOImm9) {
|
||||
ReownOrClaimBuffer();
|
||||
ResetWorkingList();
|
||||
}
|
||||
, SupportsTSOImm9(SupportsTSOImm9) {}
|
||||
|
||||
virtual ~IREmitter() = default;
|
||||
|
||||
void ReownOrClaimBuffer() {
|
||||
DualListData.ReownOrClaimBuffer();
|
||||
|
||||
// Reset the working list on new buffer.
|
||||
ResetWorkingList();
|
||||
}
|
||||
|
||||
void DelayedDisownBuffer() {
|
||||
@@ -39,7 +39,6 @@ public:
|
||||
IRListView ViewIR() {
|
||||
return IRListView(&DualListData);
|
||||
}
|
||||
void ResetWorkingList();
|
||||
|
||||
/**
|
||||
* @name IR allocation routines
|
||||
@@ -512,6 +511,9 @@ protected:
|
||||
fextl::vector<Ref> CodeBlocks;
|
||||
uint64_t Entry {};
|
||||
bool SupportsTSOImm9 {};
|
||||
|
||||
private:
|
||||
void ResetWorkingList();
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -119,11 +119,7 @@ class DualIntrusiveAllocatorThreadPool final : public DualIntrusiveAllocator {
|
||||
public:
|
||||
DualIntrusiveAllocatorThreadPool(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, size_t Size)
|
||||
: DualIntrusiveAllocator {Size}
|
||||
, PoolObject {ThreadAllocator, Size * 2} {
|
||||
// Claim a buffer on allocation
|
||||
PoolObject.ReownOrClaimBuffer();
|
||||
}
|
||||
|
||||
, PoolObject {ThreadAllocator, Size * 2} {}
|
||||
void ReownOrClaimBuffer() {
|
||||
Data = PoolObject.ReownOrClaimBuffer();
|
||||
List = Data + MemorySize;
|
||||
|
||||
@@ -51,7 +51,7 @@ void IRDumper::Run(IREmitter* IREmit) {
|
||||
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpToFile) {
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIR(), HeaderOp->OriginalRIP, IR.PostRA() ? "-post.ir" : "-pre.ir");
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIR(), +HeaderOp->OriginalRIP, IR.PostRA() ? "-post.ir" : "-pre.ir");
|
||||
FD = FEXCore::File::File(fileName.c_str(),
|
||||
FEXCore::File::FileModes::WRITE | FEXCore::File::FileModes::CREATE | FEXCore::File::FileModes::TRUNCATE);
|
||||
}
|
||||
@@ -60,9 +60,9 @@ void IRDumper::Run(IREmitter* IREmit) {
|
||||
fextl::stringstream out;
|
||||
FEXCore::IR::Dump(&out, &IR);
|
||||
if (FD.IsValid()) {
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", +HeaderOp->OriginalRIP, out.str());
|
||||
} else {
|
||||
LogMan::Msg::IFmt("IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
LogMan::Msg::IFmt("IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", +HeaderOp->OriginalRIP, out.str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -862,25 +862,20 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
Ref SinValue {};
|
||||
Ref CosValue {};
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
SinValue = IREmit->_F64SIN(St0);
|
||||
CosValue = IREmit->_F64COS(St0);
|
||||
}
|
||||
#ifdef VIXL_SIMULATOR
|
||||
if (DisableVixlIndirectCalls() == 0) {
|
||||
if (ReducedPrecisionMode) {
|
||||
SinValue = IREmit->_F64SIN(St0);
|
||||
CosValue = IREmit->_F64COS(St0);
|
||||
} else {
|
||||
SinValue = IREmit->_F80SIN(St0);
|
||||
CosValue = IREmit->_F80COS(St0);
|
||||
}
|
||||
} else
|
||||
else if (DisableVixlIndirectCalls() == 0) {
|
||||
SinValue = IREmit->_F80SIN(St0);
|
||||
CosValue = IREmit->_F80COS(St0);
|
||||
}
|
||||
#endif
|
||||
{
|
||||
else {
|
||||
SinValue = IREmit->_AllocateFPR(OpSize::i128Bit, OpSize::i128Bit);
|
||||
CosValue = IREmit->_AllocateFPR(OpSize::i128Bit, OpSize::i128Bit);
|
||||
if (ReducedPrecisionMode) {
|
||||
IREmit->_F64SINCOS(St0, SinValue, CosValue);
|
||||
} else {
|
||||
IREmit->_F80SINCOS(St0, SinValue, CosValue);
|
||||
}
|
||||
IREmit->_F80SINCOS(St0, SinValue, CosValue);
|
||||
}
|
||||
|
||||
// Push values
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -32,8 +33,8 @@ std::pmr::memory_resource* get_default_resource() {
|
||||
}
|
||||
} // namespace fextl::pmr
|
||||
|
||||
#ifndef _WIN32
|
||||
namespace FEXCore::Allocator {
|
||||
#ifndef _WIN32
|
||||
MMAP_Hook mmap {::mmap};
|
||||
MUNMAP_Hook munmap {::munmap};
|
||||
|
||||
@@ -304,5 +305,18 @@ void UnlockAfterFork(FEXCore::Core::InternalThreadState* Thread, bool Child) {
|
||||
Alloc64->UnlockAfterFork(Thread, Child);
|
||||
}
|
||||
}
|
||||
} // namespace FEXCore::Allocator
|
||||
#else
|
||||
|
||||
void VirtualNameNOP(const char*, const void*, size_t) {}
|
||||
void VirtualTHPNOP(const void* Ptr, size_t Size, THPControl Control) {}
|
||||
|
||||
VirtualNamePtr VirtualName {VirtualNameNOP};
|
||||
VirtualTHPPtr VirtualTHPControl {VirtualTHPNOP};
|
||||
|
||||
void SetupHooks(size_t PageSize, HookPtrs Ptrs) {
|
||||
VirtualName = Ptrs.VirtualName;
|
||||
VirtualTHPControl = Ptrs.VirtualTHPControl;
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::Allocator
|
||||
@@ -1,11 +1,13 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <cstddef>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
void InitializeAllocator(size_t PageSize);
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState* Thread);
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState* Thread, bool Child);
|
||||
} // namespace FEXCore::Allocator
|
||||
@@ -109,6 +109,9 @@ static void* FEX_rp_mmap(size_t size, size_t alignment, size_t* offset, size_t*
|
||||
#define PR_SET_VMA_ANON_NAME 0
|
||||
#endif
|
||||
prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, ptr, map_size, global_config.page_name);
|
||||
|
||||
// Disable HUGEPAGE on allocation from rpmalloc.
|
||||
madvise(ptr, map_size, MADV_NOHUGEPAGE);
|
||||
}
|
||||
|
||||
if (ptr == nullptr) {
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
@@ -32,7 +32,7 @@ public:
|
||||
// Differs from Itanium specification
|
||||
LOGMAN_THROW_A_FMT(PMF.adj == 0, "C++ Pointer-To-Member representation didn't have adj == 0. Are you trying to cast a virtual member?");
|
||||
#else
|
||||
#error Don't know how to cast Member to function here. Likely just Itanium
|
||||
#error "Don't know how to cast Member to function here. Likely just Itanium"
|
||||
#endif
|
||||
return PMF.ptr;
|
||||
}
|
||||
@@ -54,7 +54,7 @@ public:
|
||||
"members.");
|
||||
return PMF.ptr;
|
||||
#else
|
||||
#error Don't know how to cast Member to function here. Likely just Itanium
|
||||
#error "Don't know how to cast Member to function here. Likely just Itanium"
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
|
||||
namespace FEXCore::Utils::SpinWaitLock {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
constexpr uint64_t NanosecondsInSecond = 1'000'000'000ULL;
|
||||
|
||||
static uint32_t GetCycleCounterFrequency() {
|
||||
static uint64_t GetCycleCounterFrequency() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], CNTFRQ_EL0" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
@@ -21,7 +21,7 @@ static uint64_t CalculateCyclesPerNanosecond() {
|
||||
return NanosecondsInSecond / CounterFrequency;
|
||||
}
|
||||
|
||||
uint32_t CycleCounterFrequency = GetCycleCounterFrequency();
|
||||
uint64_t CycleCounterFrequency = GetCycleCounterFrequency();
|
||||
uint64_t CyclesPerNanosecond = CalculateCyclesPerNanosecond();
|
||||
#endif
|
||||
} // namespace FEXCore::Utils::SpinWaitLock
|
||||
@@ -0,0 +1,23 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include <FEXCore/Utils/WildcardMatcher.h>
|
||||
|
||||
namespace FEXCore::Utils::Wildcard {
|
||||
static bool matchHelper(std::string_view pattern, std::string_view text, size_t p_idx, size_t t_idx) {
|
||||
if (p_idx == pattern.size()) {
|
||||
// Pattern exhausted
|
||||
return (t_idx == text.size());
|
||||
} else if (pattern[p_idx] == '*') {
|
||||
// Wildcard: Try matching zero characters, or one or more characters
|
||||
return matchHelper(pattern, text, p_idx + 1, t_idx) || (t_idx < text.size() && matchHelper(pattern, text, p_idx, t_idx + 1));
|
||||
} else {
|
||||
// Match normally
|
||||
return (t_idx < text.size() && pattern[p_idx] == text[t_idx] && matchHelper(pattern, text, p_idx + 1, t_idx + 1));
|
||||
}
|
||||
}
|
||||
|
||||
bool Matches(std::string_view pattern, std::string_view text) {
|
||||
return matchHelper(pattern, text, 0, 0);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Utils::Wildcard
|
||||
@@ -27,7 +27,12 @@ namespace HLE {
|
||||
struct SourcecodeMap;
|
||||
} // namespace HLE
|
||||
|
||||
enum class GuestRelocationType : uint32_t { Rel32, Rel64 };
|
||||
enum class GuestRelocationType : uint32_t {
|
||||
Rel32,
|
||||
Rel64,
|
||||
// Skip blocks containing this relocation
|
||||
Skip,
|
||||
};
|
||||
|
||||
// Generic information associated with an executable file.
|
||||
struct ExecutableFileInfo {
|
||||
|
||||
@@ -13,10 +13,10 @@
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/WritePriorityMutex.h>
|
||||
|
||||
namespace FEXCore {
|
||||
struct HostFeatures;
|
||||
class ForkableSharedMutex;
|
||||
class ThunkHandler;
|
||||
} // namespace FEXCore
|
||||
|
||||
@@ -73,6 +73,7 @@ public:
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void ExecuteThread(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual bool CheckIfBlockIsCacheable(FEXCore::Core::InternalThreadState& Thread, uint64_t GuestRIP, uint64_t MaxInst) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void CompileRIPCount(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) = 0;
|
||||
|
||||
@@ -143,7 +144,7 @@ public:
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void
|
||||
InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::Utils::WritePriorityMutex::Mutex& GetCodeInvalidationMutex() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void
|
||||
ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) = 0;
|
||||
|
||||
@@ -337,6 +337,7 @@ struct JITPointers {
|
||||
// Process specific
|
||||
uint64_t PrintValue {};
|
||||
uint64_t PrintVectorValue {};
|
||||
uint64_t PrintMsgValue {};
|
||||
uint64_t ThreadRemoveCodeEntryFromJIT {};
|
||||
uint64_t CPUIDObj {};
|
||||
uint64_t CPUIDFunction {};
|
||||
@@ -375,6 +376,9 @@ struct JITPointers {
|
||||
uint64_t L2Pointer {};
|
||||
uint64_t LUDIVHandler {};
|
||||
uint64_t LDIVHandler {};
|
||||
uint64_t F64SinHandler {};
|
||||
uint64_t F64CosHandler {};
|
||||
uint64_t F64TanHandler {};
|
||||
/** @} */
|
||||
|
||||
// Copy of process-wide named vector constants data.
|
||||
|
||||
@@ -41,6 +41,7 @@ struct HostFeatures {
|
||||
bool SupportsWFXT {};
|
||||
bool Supports3DNow {};
|
||||
bool SupportsSSE4a {};
|
||||
bool SupportsMOPS {};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
|
||||
@@ -12,9 +12,6 @@ struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
FEX_DEFAULT_VISIBILITY void SetupHooks(size_t PageSize);
|
||||
FEX_DEFAULT_VISIBILITY void ClearHooks();
|
||||
|
||||
FEX_DEFAULT_VISIBILITY size_t DetermineVASize();
|
||||
|
||||
#ifdef GLIBC_ALLOCATOR_FAULT
|
||||
|
||||
@@ -27,6 +27,24 @@ enum class ProtectOptions : uint32_t {
|
||||
};
|
||||
FEX_DEF_NUM_OPS(ProtectOptions)
|
||||
|
||||
enum class THPControl {
|
||||
Enable,
|
||||
Disable,
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
FEX_DEFAULT_VISIBILITY void SetupHooks(size_t PageSize);
|
||||
#else
|
||||
using VirtualNamePtr = void (*)(const char*, const void*, size_t);
|
||||
using VirtualTHPPtr = void (*)(const void*, size_t, THPControl);
|
||||
struct HookPtrs {
|
||||
VirtualNamePtr VirtualName;
|
||||
VirtualTHPPtr VirtualTHPControl;
|
||||
};
|
||||
FEX_DEFAULT_VISIBILITY void SetupHooks(size_t PageSize, HookPtrs Ptrs);
|
||||
#endif
|
||||
FEX_DEFAULT_VISIBILITY void ClearHooks();
|
||||
|
||||
#ifdef _WIN32
|
||||
inline void* VirtualAlloc(void* Base, size_t Size, bool Execute = false, bool Commit = true) {
|
||||
// Allocate top-down to avoid polluting the lower VA space, as even on 64-bit some programs (i.e. LuaJIT) require allocations below 4GB.
|
||||
@@ -82,8 +100,8 @@ inline bool VirtualProtect(void* Ptr, size_t Size, ProtectOptions options) {
|
||||
return ::VirtualProtect(Ptr, Size, prot, nullptr) == 0;
|
||||
}
|
||||
|
||||
inline void VirtualName(const char*, void*, size_t) {}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY extern VirtualNamePtr VirtualName;
|
||||
FEX_DEFAULT_VISIBILITY extern VirtualTHPPtr VirtualTHPControl;
|
||||
#else
|
||||
using MMAP_Hook = void* (*)(void*, size_t, int, int, int, off_t);
|
||||
using MUNMAP_Hook = int (*)(void*, size_t);
|
||||
@@ -123,6 +141,10 @@ inline bool VirtualProtect(void* Ptr, size_t Size, ProtectOptions options) {
|
||||
return ::mprotect(Ptr, Size, prot) == 0;
|
||||
}
|
||||
|
||||
inline void VirtualTHPControl(const void* Ptr, size_t Size, THPControl Control) {
|
||||
::madvise(const_cast<void*>(Ptr), Size, Control == THPControl::Enable ? MADV_HUGEPAGE : MADV_NOHUGEPAGE);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
// Memory allocation routines to be defined externally.
|
||||
@@ -142,7 +164,6 @@ void aligned_free(void* ptr);
|
||||
FEX_DEFAULT_VISIBILITY extern void InitializeThread();
|
||||
|
||||
#ifndef _WIN32
|
||||
void InitializeAllocator(size_t PageSize);
|
||||
void SetupAllocatorHooks(void* (*)(void* addr, size_t length, int prot, int flags, int fd, off_t offset), int (*)(void* addr, size_t length));
|
||||
#endif
|
||||
|
||||
|
||||
@@ -28,6 +28,9 @@
|
||||
// then program behavior is undefined.
|
||||
#define FEX_UNREACHABLE __builtin_unreachable()
|
||||
|
||||
// Like offsetof but for array members with a dynamic element index
|
||||
#define ARRAY_OFFSETOF(Type, ArrayMember, Index) (offsetof(Type, ArrayMember) + sizeof(Type::ArrayMember[0]) * (Index))
|
||||
|
||||
namespace FEXCore::Assert {
|
||||
// This function can not be inlined
|
||||
[[noreturn]]
|
||||
|
||||
@@ -50,8 +50,9 @@ namespace FEXCore::Utils::SpinWaitLock {
|
||||
#define SPINLOOP_32BIT SPINLOOP_BODY(ldar, w)
|
||||
#define SPINLOOP_64BIT SPINLOOP_BODY(ldar, x)
|
||||
|
||||
extern uint32_t CycleCounterFrequency;
|
||||
extern uint64_t CyclesPerNanosecond;
|
||||
FEX_DEFAULT_VISIBILITY extern uint64_t CycleCounterFrequency;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY extern uint64_t CyclesPerNanosecond;
|
||||
|
||||
///< Get the raw cycle counter which is synchronizing.
|
||||
/// `CNTVCTSS_EL0` also does the same thing, but requires the FEAT_ECV feature.
|
||||
@@ -0,0 +1,8 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
#include <string_view>
|
||||
|
||||
namespace FEXCore::Utils::Wildcard {
|
||||
bool Matches(std::string_view pattern, std::string_view text);
|
||||
} // namespace FEXCore::Utils::Wildcard
|
||||
+5
-1
@@ -9,11 +9,15 @@
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <synchapi.h>
|
||||
// Don't pull in all WIN32 headers for INFINITE. Causes too many problems.
|
||||
#ifndef INFINITE
|
||||
#define INFINITE 0xffffffff
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
|
||||
namespace FEXCore::Utils::WritePriorityMutex {
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <chrono>
|
||||
#include <thread>
|
||||
|
||||
@@ -59,6 +59,7 @@ class HostFeatures(Flag) :
|
||||
FEATURE_LRCPC = (1 << 14)
|
||||
FEATURE_LRCPC2 = (1 << 15)
|
||||
FEATURE_FRINTTS = (1 << 16)
|
||||
FEATURE_MOPS = (1 << 17)
|
||||
|
||||
HostFeaturesLookup = {
|
||||
"SVE128" : HostFeatures.FEATURE_SVE128,
|
||||
@@ -78,6 +79,7 @@ HostFeaturesLookup = {
|
||||
"LRCPC" : HostFeatures.FEATURE_LRCPC,
|
||||
"LRCPC2" : HostFeatures.FEATURE_LRCPC2,
|
||||
"FRINTTS" : HostFeatures.FEATURE_FRINTTS,
|
||||
"MOPS" : HostFeatures.FEATURE_MOPS,
|
||||
}
|
||||
|
||||
def GetHostFeatures(data):
|
||||
|
||||
+57
-36
@@ -7,11 +7,12 @@
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/WildcardMatcher.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
#include <FEXHeaderUtils/SymlinkChecks.h>
|
||||
|
||||
#include <cstring>
|
||||
#include <fmt/format.h>
|
||||
#include <functional>
|
||||
@@ -28,7 +29,8 @@
|
||||
|
||||
namespace FEX::Config {
|
||||
namespace JSON {
|
||||
static void LoadJSonConfig(const fextl::string& Config, std::function<void(const char* Name, const char* ConfigSring)> Func) {
|
||||
static void LoadJSonConfig(const fextl::string& Config, std::optional<fextl::string> AppName,
|
||||
std::function<void(const char* Name, const char* ConfigString)> Func) {
|
||||
fextl::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
@@ -48,21 +50,35 @@ namespace JSON {
|
||||
return;
|
||||
}
|
||||
|
||||
for (const json_t* ConfigItem = json_getChild(ConfigList); ConfigItem != nullptr; ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
fextl::vector<const json_t*> ConfigBlocks;
|
||||
ConfigBlocks.push_back(ConfigList);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("JSON file '{}': Couldn't get config name for an item", Config);
|
||||
return;
|
||||
if (AppName) {
|
||||
const json_t* OverrideList = json_getProperty(json, "AppOverrides");
|
||||
if (OverrideList) {
|
||||
for (const json_t* Item = json_getChild(OverrideList); Item != nullptr; Item = json_getSibling(Item)) {
|
||||
const char* AppPattern = json_getName(Item);
|
||||
|
||||
// Find the first match, then break
|
||||
if (FEXCore::Utils::Wildcard::Matches(AppPattern, *AppName)) {
|
||||
ConfigBlocks.push_back(Item);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("JSON file '{}': Couldn't get value for config item '{}'", Config, ConfigName);
|
||||
return;
|
||||
for (auto ConfigBlock : ConfigBlocks) {
|
||||
for (const json_t* ConfigItem = json_getChild(ConfigBlock); ConfigItem != nullptr; ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("JSON file '{}': Couldn't get value for config item '{}'", Config, ConfigName);
|
||||
return;
|
||||
}
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
} // namespace JSON
|
||||
@@ -167,22 +183,24 @@ protected:
|
||||
|
||||
class MainLoader final : public OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(fextl::string ConfigFile);
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type, std::optional<fextl::string> AppName = std::nullopt);
|
||||
explicit MainLoader(fextl::string ConfigFile, std::optional<fextl::string> AppName = std::nullopt);
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigFile);
|
||||
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
std::optional<fextl::string> AppName;
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
explicit AppLoader(const fextl::string& AppName, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
const fextl::string AppName;
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
@@ -221,12 +239,14 @@ void OptionMapper::MapNameToOption(const char* ConfigName, const char* ConfigStr
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type, std::optional<fextl::string> AppName)
|
||||
: OptionMapper(Type)
|
||||
, AppName {AppName}
|
||||
, Config {FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {}
|
||||
|
||||
MainLoader::MainLoader(fextl::string ConfigFile)
|
||||
MainLoader::MainLoader(fextl::string ConfigFile, std::optional<fextl::string> AppName)
|
||||
: OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, AppName {AppName}
|
||||
, Config {std::move(ConfigFile)} {}
|
||||
|
||||
|
||||
@@ -236,13 +256,14 @@ MainLoader::MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigF
|
||||
|
||||
void MainLoader::Load() {
|
||||
SetCurrentConfigFile(Config);
|
||||
JSON::LoadJSonConfig(Config, [this](const char* Name, const char* ConfigString) { MapNameToOption(Name, ConfigString); });
|
||||
JSON::LoadJSonConfig(Config, AppName, [this](const char* Name, const char* ConfigString) { MapNameToOption(Name, ConfigString); });
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: OptionMapper(Type) {
|
||||
AppLoader::AppLoader(const fextl::string& AppName, FEXCore::Config::LayerType Type)
|
||||
: OptionMapper(Type)
|
||||
, AppName {AppName} {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP || Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
Config = FEXCore::Config::GetApplicationConfig(AppName, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
@@ -250,7 +271,7 @@ AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType T
|
||||
|
||||
void AppLoader::Load() {
|
||||
SetCurrentConfigFile(Config);
|
||||
JSON::LoadJSonConfig(Config, [this](const char* Name, const char* ConfigString) { MapNameToOption(Name, ConfigString); });
|
||||
JSON::LoadJSonConfig(Config, AppName, [this](const char* Name, const char* ConfigString) { MapNameToOption(Name, ConfigString); });
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char* const _envp[])
|
||||
@@ -320,11 +341,11 @@ fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(const fextl::string* File) {
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(const fextl::string* File, std::optional<fextl::string> AppName) {
|
||||
if (File) {
|
||||
return fextl::make_unique<MainLoader>(*File);
|
||||
return fextl::make_unique<MainLoader>(*File, std::move(AppName));
|
||||
} else {
|
||||
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN, std::move(AppName));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -461,7 +482,7 @@ void LoadConfig(fextl::string ProgramName, char** const envp, const PortableInfo
|
||||
if (!IsPortable) {
|
||||
FEXCore::Config::AddLayer(CreateGlobalMainLayer());
|
||||
}
|
||||
FEXCore::Config::AddLayer(CreateMainLayer());
|
||||
FEXCore::Config::AddLayer(CreateMainLayer(nullptr, ProgramName.empty() ? std::nullopt : std::optional {ProgramName}));
|
||||
|
||||
if (!ProgramName.empty()) {
|
||||
if (!IsPortable) {
|
||||
@@ -633,17 +654,10 @@ fextl::string GetDataDirectory(bool Global, const PortableInformation& PortableI
|
||||
}
|
||||
|
||||
fextl::string GetConfigDirectory(bool Global, const PortableInformation& PortableInfo) {
|
||||
#ifdef FEX_STEAM_SUPPORT
|
||||
const char* SteamDataPath = getenv("STEAM_COMPAT_DATA_PATH");
|
||||
if (SteamDataPath) {
|
||||
return fextl::fmt::format("{}/fex-emu/", SteamDataPath);
|
||||
}
|
||||
#endif
|
||||
|
||||
const char* ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (PortableInfo.IsPortable && (Global || !ConfigOverride)) {
|
||||
if (PortableInfo.IsPortable && Global) {
|
||||
return fextl::fmt::format("{}/fex-emu/", PortableInfo.InterpreterPath);
|
||||
} else if (PortableInfo.IsPortable && ConfigOverride && !Global) {
|
||||
} else if (ConfigOverride && !Global) {
|
||||
fextl::string AppConfigStr = ConfigOverride;
|
||||
if (FHU::Filesystem::IsRelative(AppConfigStr)) {
|
||||
AppConfigStr = PortableInfo.InterpreterPath + AppConfigStr;
|
||||
@@ -652,6 +666,13 @@ fextl::string GetConfigDirectory(bool Global, const PortableInformation& Portabl
|
||||
return AppConfigStr;
|
||||
}
|
||||
|
||||
#ifdef FEX_STEAM_SUPPORT
|
||||
const char* SteamDataPath = getenv("STEAM_COMPAT_DATA_PATH");
|
||||
if (SteamDataPath) {
|
||||
return fextl::fmt::format("{}/fex-emu/", SteamDataPath);
|
||||
}
|
||||
#endif
|
||||
|
||||
fextl::string ConfigDir;
|
||||
if (Global) {
|
||||
return GLOBAL_DATA_DIRECTORY;
|
||||
|
||||
@@ -81,7 +81,7 @@ fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer();
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(const fextl::string* File = nullptr);
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(const fextl::string* File = nullptr, std::optional<fextl::string> AppName = std::nullopt);
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(std::string_view AppConfig);
|
||||
|
||||
/**
|
||||
|
||||
@@ -502,6 +502,7 @@ static void OverrideFeatures(FEXCore::HostFeatures* Features, uint64_t ForceSVEW
|
||||
ENABLE_DISABLE_OPTION(SupportsWFXT, WFXT, WFXT);
|
||||
ENABLE_DISABLE_OPTION(Supports3DNow, 3DNOW, 3DNOW);
|
||||
ENABLE_DISABLE_OPTION(SupportsSSE4a, SSE4A, SSE4A);
|
||||
ENABLE_DISABLE_OPTION(SupportsMOPS, MOPS, MOPS);
|
||||
GET_SINGLE_OPTION(Crypto, CRYPTO);
|
||||
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
@@ -625,11 +626,49 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
// Hardcode enable SVE with 256-bit wide registers.
|
||||
HostFeatures.SupportsSVE128 = ForceSVEWidth() ? ForceSVEWidth() >= 128 : true;
|
||||
HostFeatures.SupportsSVE256 = ForceSVEWidth() ? ForceSVEWidth() >= 256 : true;
|
||||
HostFeatures.SupportsMOPS = true;
|
||||
|
||||
// Simulator has a hardcoded ZVA size of 64-bytes.
|
||||
HostFeatures.SupportsCLZERO = true;
|
||||
HostFeatures.SupportsAES = true;
|
||||
HostFeatures.SupportsCRC = true;
|
||||
HostFeatures.SupportsAVX = true;
|
||||
HostFeatures.SupportsSHA = true;
|
||||
HostFeatures.SupportsPMULL_128Bit = true;
|
||||
HostFeatures.SupportsAES256 = true;
|
||||
|
||||
// Simulator doesn't support these
|
||||
HostFeatures.SupportsRPRES = false;
|
||||
HostFeatures.SupportsAFP = false;
|
||||
#else
|
||||
HostFeatures.SupportsSVE128 = Features.Supports(CPUFeatures::Feature::SVE2);
|
||||
HostFeatures.SupportsSVE256 = Features.Supports(CPUFeatures::Feature::SVE2) && Features.GetSVEVectorLengthInBits() >= 256;
|
||||
HostFeatures.SupportsMOPS = Features.Supports(CPUFeatures::Feature::MOPS);
|
||||
|
||||
// Check if we can support cacheline clears
|
||||
if (Features.GetDCZID().SupportsDCZVA()) {
|
||||
// If the DC ZVA size matches the emulated cache line size
|
||||
// This means we can use the instruction
|
||||
constexpr static uint64_t CACHELINE_SIZE = 64;
|
||||
HostFeatures.SupportsCLZERO = Features.GetDCZID().BlockSizeInBytes() == CACHELINE_SIZE;
|
||||
}
|
||||
#endif
|
||||
|
||||
HostFeatures.SupportsAVX = true;
|
||||
HostFeatures.SupportsAES256 = HostFeatures.SupportsAVX && HostFeatures.SupportsAES;
|
||||
HostFeatures.SupportsPreserveAllABI = FEX_HAS_PRESERVE_ALL_ATTR;
|
||||
|
||||
if (CTR) {
|
||||
HostFeatures.DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
HostFeatures.ICacheLineSize = 4 << (CTR & 0xF);
|
||||
} else {
|
||||
HostFeatures.DCacheLineSize = 64;
|
||||
HostFeatures.ICacheLineSize = 64;
|
||||
}
|
||||
|
||||
if (!HostFeatures.SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
#ifdef _WIN32
|
||||
// Disable 3DNow! by default to better match the set of extensions exposed on modern CPUs.
|
||||
@@ -640,12 +679,6 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
HostFeatures.Supports3DNow = true;
|
||||
#endif
|
||||
|
||||
HostFeatures.SupportsAES256 = HostFeatures.SupportsAVX && HostFeatures.SupportsAES;
|
||||
|
||||
if (!HostFeatures.SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
@@ -666,36 +699,6 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
SetFPCR(OriginalFPCR);
|
||||
#endif
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// simulator has a hardcoded ZVA size of 64-bytes.
|
||||
HostFeatures.SupportsCLZERO = true;
|
||||
HostFeatures.SupportsAES = true;
|
||||
HostFeatures.SupportsCRC = true;
|
||||
HostFeatures.SupportsAVX = true;
|
||||
HostFeatures.SupportsSHA = true;
|
||||
HostFeatures.SupportsPMULL_128Bit = true;
|
||||
HostFeatures.SupportsAES256 = true;
|
||||
|
||||
// Simulator doesn't support these
|
||||
HostFeatures.SupportsRPRES = false;
|
||||
HostFeatures.SupportsAFP = false;
|
||||
#else
|
||||
// Check if we can support cacheline clears
|
||||
if (Features.GetDCZID().SupportsDCZVA()) {
|
||||
// If the DC ZVA size matches the emulated cache line size
|
||||
// This means we can use the instruction
|
||||
constexpr static uint64_t CACHELINE_SIZE = 64;
|
||||
HostFeatures.SupportsCLZERO = Features.GetDCZID().BlockSizeInBytes() == CACHELINE_SIZE;
|
||||
}
|
||||
#endif
|
||||
|
||||
if (CTR) {
|
||||
HostFeatures.DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
HostFeatures.ICacheLineSize = 4 << (CTR & 0xF);
|
||||
} else {
|
||||
HostFeatures.DCacheLineSize = HostFeatures.ICacheLineSize = 64;
|
||||
}
|
||||
|
||||
#if defined(ARCHITECTURE_x86_64) && !defined(VIXL_SIMULATOR)
|
||||
FEX::X86::Features Feature {};
|
||||
HostFeatures.SupportsAES = Feature.Feat_aes;
|
||||
@@ -712,7 +715,6 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
HostFeatures.SupportsAFP = true;
|
||||
HostFeatures.SupportsFloatExceptions = true;
|
||||
#endif
|
||||
HostFeatures.SupportsPreserveAllABI = FEX_HAS_PRESERVE_ALL_ATTR;
|
||||
|
||||
HandleErrata(&HostFeatures, MIDR);
|
||||
OverrideFeatures(&HostFeatures, ForceSVEWidth());
|
||||
|
||||
@@ -519,6 +519,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
FEATURE_LRCPC = (1U << 14),
|
||||
FEATURE_LRCPC2 = (1U << 15),
|
||||
FEATURE_FRINTTS = (1U << 16),
|
||||
FEATURE_MOPS = (1U << 17),
|
||||
};
|
||||
|
||||
uint64_t SVEWidth = 0;
|
||||
@@ -569,6 +570,9 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_FRINTTS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEFRINTTS);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_MOPS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEMOPS);
|
||||
}
|
||||
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_TSO) {
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "1");
|
||||
@@ -624,6 +628,9 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_FRINTTS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEFRINTTS);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_MOPS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEMOPS);
|
||||
}
|
||||
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_TSO) {
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0");
|
||||
|
||||
@@ -1,11 +1,13 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <elf.h>
|
||||
#include <fcntl.h>
|
||||
#include <optional>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "Linux/Utils/ELFContainer.h"
|
||||
@@ -19,6 +21,7 @@
|
||||
struct ELFParser {
|
||||
Elf64_Ehdr ehdr;
|
||||
fextl::vector<Elf64_Phdr> phdrs;
|
||||
std::optional<fextl::vector<Elf64_Shdr>> shdrs;
|
||||
::ELFLoader::ELFContainer::ELFType type {::ELFLoader::ELFContainer::TYPE_NONE};
|
||||
|
||||
fextl::string InterpreterElf;
|
||||
@@ -30,6 +33,7 @@ struct ELFParser {
|
||||
|
||||
fd = NewFD;
|
||||
type = ::ELFLoader::ELFContainer::TYPE_NONE;
|
||||
shdrs.reset();
|
||||
|
||||
if (fd == -1) {
|
||||
// Likely just doesn't exist
|
||||
@@ -236,6 +240,168 @@ struct ELFParser {
|
||||
return ReadElf(NewFD);
|
||||
}
|
||||
|
||||
/**
|
||||
* Checks if DT_TEXTREL/DF_TEXTREL exist in the PT_DYNAMIC segment.
|
||||
*
|
||||
* These indicate that the ELF has relocations that cover to read-only code
|
||||
* pages. The dynamic loader will temporarily map these pages as writeable
|
||||
* to apply the relocations.
|
||||
*/
|
||||
bool HasCodeRelocations() const {
|
||||
if (fd == -1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto phdr_it = std::ranges::find_if(phdrs, [](auto& phdr) { return phdr.p_type == PT_DYNAMIC; });
|
||||
if (phdr_it == phdrs.end()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (type == ::ELFLoader::ELFContainer::TYPE_X86_32) {
|
||||
return HasCodeRelocations<Elf32_Dyn>(*phdr_it);
|
||||
} else {
|
||||
return HasCodeRelocations<Elf64_Dyn>(*phdr_it);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename Elf_Dyn>
|
||||
bool HasCodeRelocations(const Elf64_Phdr& phdr) const {
|
||||
const size_t EntryCount = phdr.p_filesz / sizeof(Elf_Dyn);
|
||||
fextl::vector<Elf_Dyn> Entries(EntryCount);
|
||||
|
||||
if (pread(fd, Entries.data(), phdr.p_filesz, phdr.p_offset) == -1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
for (auto& Entry : Entries) {
|
||||
if (Entry.d_tag == DT_NULL) {
|
||||
break;
|
||||
}
|
||||
if (Entry.d_tag == DT_TEXTREL) {
|
||||
return true;
|
||||
}
|
||||
if (Entry.d_tag == DT_FLAGS && (Entry.d_un.d_val & DF_TEXTREL)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parses relocation sections (SHT_REL/SHT_RELA) and returns a map of
|
||||
* offsets to relocations that FEX's JIT must know about.
|
||||
*/
|
||||
fextl::robin_map<uint32_t, FEXCore::GuestRelocationType> PopulateRelocations() {
|
||||
if (fd == -1 || !EnsureSectionHeadersLoaded()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
fextl::robin_map<uint32_t, FEXCore::GuestRelocationType> Relocations;
|
||||
bool Is32Bit = (type == ::ELFLoader::ELFContainer::TYPE_X86_32);
|
||||
|
||||
for (const auto& shdr : *shdrs) {
|
||||
if (shdr.sh_entsize == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const size_t EntryCount = shdr.sh_size / shdr.sh_entsize;
|
||||
|
||||
if (!Is32Bit) {
|
||||
if (shdr.sh_type == SHT_REL) {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected relocation section type");
|
||||
} else if (shdr.sh_type == SHT_RELA) {
|
||||
fextl::vector<Elf64_Rela> Entries(EntryCount);
|
||||
if (pread(fd, Entries.data(), shdr.sh_size, shdr.sh_offset) == -1) {
|
||||
LOGMAN_THROW_A_FMT(false, "Failed to read RELA section");
|
||||
}
|
||||
for (auto& Entry : Entries) {
|
||||
auto RelocType = ClassifyRelocation64(ELF64_R_TYPE(Entry.r_info));
|
||||
if (RelocType) {
|
||||
Relocations.emplace(static_cast<uint32_t>(Entry.r_offset), *RelocType);
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (shdr.sh_type == SHT_REL) {
|
||||
fextl::vector<Elf32_Rel> Entries(EntryCount);
|
||||
if (pread(fd, Entries.data(), shdr.sh_size, shdr.sh_offset) == -1) {
|
||||
LOGMAN_THROW_A_FMT(false, "Failed to read REL section");
|
||||
}
|
||||
for (auto& Entry : Entries) {
|
||||
auto RelocType = ClassifyRelocation32(ELF32_R_TYPE(Entry.r_info));
|
||||
if (RelocType) {
|
||||
Relocations.emplace(static_cast<uint32_t>(Entry.r_offset), *RelocType);
|
||||
}
|
||||
}
|
||||
} else if (shdr.sh_type == SHT_RELA) {
|
||||
fextl::vector<Elf32_Rela> Entries(EntryCount);
|
||||
if (pread(fd, Entries.data(), shdr.sh_size, shdr.sh_offset) == -1) {
|
||||
LOGMAN_THROW_A_FMT(false, "Failed to read RELA section");
|
||||
}
|
||||
for (auto& Entry : Entries) {
|
||||
auto RelocType = ClassifyRelocation32(ELF32_R_TYPE(Entry.r_info));
|
||||
if (RelocType) {
|
||||
Relocations.emplace(static_cast<uint32_t>(Entry.r_offset), *RelocType);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Relocations;
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns underlying 32-bit relocation entries.
|
||||
* SHT_REL entries are implicitly converted to Elf32_Rela.
|
||||
*/
|
||||
fextl::vector<Elf32_Rela> ReadRawRelocations32() {
|
||||
if (fd == -1 || type != ::ELFLoader::ELFContainer::TYPE_X86_32 || !EnsureSectionHeadersLoaded()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
// Load dynamic symbol table (find SHT_DYNSYM section)
|
||||
fextl::vector<Elf32_Sym> DynSyms;
|
||||
auto DynsymHeader = std::ranges::find_if(*shdrs, [](auto& shdr) { return shdr.sh_type == SHT_DYNSYM; });
|
||||
if (DynsymHeader != shdrs->end()) {
|
||||
size_t SymCount = DynsymHeader->sh_size / sizeof(Elf32_Sym);
|
||||
DynSyms.resize(SymCount);
|
||||
if (pread(fd, DynSyms.data(), DynsymHeader->sh_size, DynsymHeader->sh_offset) == -1) {
|
||||
LOGMAN_MSG_A_FMT("Could not load DYNSYM section");
|
||||
}
|
||||
}
|
||||
|
||||
fextl::vector<Elf32_Rela> Result;
|
||||
for (const auto& shdr : *shdrs) {
|
||||
if (shdr.sh_entsize == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const size_t EntryCount = shdr.sh_size / shdr.sh_entsize;
|
||||
|
||||
if (shdr.sh_type == SHT_REL) {
|
||||
fextl::vector<Elf32_Rel> Entries(EntryCount);
|
||||
if (pread(fd, Entries.data(), shdr.sh_size, shdr.sh_offset) == -1) {
|
||||
LOGMAN_MSG_A_FMT("Could not load REL section");
|
||||
}
|
||||
for (auto& Entry : Entries) {
|
||||
auto Sym = ELF32_R_SYM(Entry.r_info);
|
||||
int32_t Addend = (Sym < DynSyms.size()) ? static_cast<int32_t>(DynSyms[Sym].st_value) : 0;
|
||||
Result.push_back(Elf32_Rela {Entry.r_offset, Entry.r_info, Addend});
|
||||
}
|
||||
} else if (shdr.sh_type == SHT_RELA) {
|
||||
fextl::vector<Elf32_Rela> Entries(EntryCount);
|
||||
if (pread(fd, Entries.data(), shdr.sh_size, shdr.sh_offset) == -1) {
|
||||
LOGMAN_MSG_A_FMT("Could not load RELA section");
|
||||
}
|
||||
Result.insert(Result.end(), Entries.begin(), Entries.end());
|
||||
}
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
void Closefd() {
|
||||
if (fd != -1) {
|
||||
close(fd);
|
||||
@@ -246,4 +412,71 @@ struct ELFParser {
|
||||
~ELFParser() {
|
||||
Closefd();
|
||||
}
|
||||
|
||||
private:
|
||||
/// Returns true if loading section headers succeeded
|
||||
bool EnsureSectionHeadersLoaded() {
|
||||
if (shdrs.has_value()) {
|
||||
return !shdrs->empty();
|
||||
}
|
||||
|
||||
if (fd == -1 || ehdr.e_shoff == 0 || ehdr.e_shnum == 0) {
|
||||
shdrs.emplace();
|
||||
return false;
|
||||
}
|
||||
|
||||
if (type == ::ELFLoader::ELFContainer::TYPE_X86_64) {
|
||||
shdrs.emplace(ehdr.e_shnum);
|
||||
if (pread(fd, shdrs->data(), sizeof(Elf64_Shdr) * ehdr.e_shnum, ehdr.e_shoff) == -1) {
|
||||
shdrs->clear();
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
fextl::vector<Elf32_Shdr> shdrs32(ehdr.e_shnum);
|
||||
if (pread(fd, shdrs32.data(), sizeof(Elf32_Shdr) * ehdr.e_shnum, ehdr.e_shoff) == -1) {
|
||||
shdrs.emplace();
|
||||
return false;
|
||||
}
|
||||
|
||||
shdrs.emplace(ehdr.e_shnum);
|
||||
for (int i = 0; i < ehdr.e_shnum; i++) {
|
||||
#define COPY(name) (*shdrs)[i].name = shdrs32[i].name
|
||||
COPY(sh_name);
|
||||
COPY(sh_type);
|
||||
COPY(sh_flags);
|
||||
COPY(sh_addr);
|
||||
COPY(sh_offset);
|
||||
COPY(sh_size);
|
||||
COPY(sh_link);
|
||||
COPY(sh_info);
|
||||
COPY(sh_addralign);
|
||||
COPY(sh_entsize);
|
||||
#undef COPY
|
||||
}
|
||||
}
|
||||
|
||||
return !shdrs->empty();
|
||||
}
|
||||
|
||||
static std::optional<FEXCore::GuestRelocationType> ClassifyRelocation32(uint32_t Type) {
|
||||
if (Type == R_386_RELATIVE || Type == R_386_32) {
|
||||
return FEXCore::GuestRelocationType::Rel32;
|
||||
} else if (Type == R_386_PC32) {
|
||||
// Currently not handled
|
||||
return FEXCore::GuestRelocationType::Skip;
|
||||
} else if (Type == R_386_TLS_TPOFF) {
|
||||
// Currently not handled
|
||||
return FEXCore::GuestRelocationType::Skip;
|
||||
}
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
static std::optional<FEXCore::GuestRelocationType> ClassifyRelocation64(uint32_t Type) {
|
||||
if (Type == R_X86_64_RELATIVE || Type == R_X86_64_64) {
|
||||
return FEXCore::GuestRelocationType::Rel64;
|
||||
} else if (Type == R_X86_64_32) {
|
||||
return FEXCore::GuestRelocationType::Rel32;
|
||||
}
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
@@ -14,6 +14,8 @@
|
||||
#include <filesystem>
|
||||
#include <string>
|
||||
#include <sys/prctl.h>
|
||||
#include <sys/signal.h>
|
||||
#include <ucontext.h>
|
||||
|
||||
namespace {
|
||||
struct TSOEmulationFacts {
|
||||
@@ -94,6 +96,103 @@ TSOEmulationFacts GetTSOEmulationFacts() {
|
||||
#endif
|
||||
} // namespace
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
namespace SIGBUSTest {
|
||||
static bool* FaultArray {};
|
||||
|
||||
__attribute__((naked)) void atomic_store_u16(std::byte* Data, uint16_t Value) {
|
||||
asm volatile(R"(
|
||||
stlrh w1, [x0];
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) void atomic_store_u32(std::byte* Data, uint32_t Value) {
|
||||
asm volatile(R"(
|
||||
stlr w1, [x0];
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) void atomic_store_u64(std::byte* Data, uint64_t Value) {
|
||||
asm volatile(R"(
|
||||
stlr x1, [x0];
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
__attribute__((naked)) void atomic_store_u128(std::byte* Data, uint64_t Value) {
|
||||
asm volatile(R"(
|
||||
stlxp w3, x1, x1, [x0];
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
static void HandleSIGBUS(int, siginfo_t* info, void* context) {
|
||||
FaultArray[reinterpret_cast<uintptr_t>(info->si_addr) & 63] = true;
|
||||
|
||||
ucontext_t* ucontext = (ucontext_t*)context;
|
||||
mcontext_t* mcontext = &ucontext->uc_mcontext;
|
||||
// Skip the stlr.
|
||||
mcontext->pc += 4;
|
||||
}
|
||||
|
||||
void TestSIGBUS() {
|
||||
struct sigaction act {};
|
||||
act.sa_sigaction = HandleSIGBUS;
|
||||
act.sa_flags = SA_SIGINFO;
|
||||
sigaction(SIGBUS, &act, &act);
|
||||
auto ptr = reinterpret_cast<std::byte*>(mmap(nullptr, 4096, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
|
||||
auto test_fault = [](bool* FaultOffsets, auto AccessFunction, std::byte* AccessArray) {
|
||||
FaultArray = FaultOffsets;
|
||||
for (size_t i = 0; i < 64; ++i) {
|
||||
AccessFunction(AccessArray + i, 1);
|
||||
}
|
||||
};
|
||||
|
||||
auto print_granule = [](const char* size, bool* FaultArray) {
|
||||
std::string output {};
|
||||
for (size_t i = 0; i < 64; ++i) {
|
||||
if (i && (i % 16 == 0)) {
|
||||
output += " ";
|
||||
}
|
||||
|
||||
if (FaultArray[i]) {
|
||||
output += "\e[31m■\e[0m";
|
||||
} else {
|
||||
output += "\e[32m■\e[0m";
|
||||
}
|
||||
}
|
||||
|
||||
fprintf(stdout, "%s: %s\n", size, output.c_str());
|
||||
};
|
||||
|
||||
bool FaultOffset_16bit[64] {};
|
||||
bool FaultOffset_32bit[64] {};
|
||||
bool FaultOffset_64bit[64] {};
|
||||
bool FaultOffset_128bit[64] {};
|
||||
|
||||
test_fault(FaultOffset_16bit, atomic_store_u16, ptr);
|
||||
test_fault(FaultOffset_32bit, atomic_store_u32, ptr);
|
||||
test_fault(FaultOffset_64bit, atomic_store_u64, ptr);
|
||||
test_fault(FaultOffset_128bit, atomic_store_u128, ptr);
|
||||
|
||||
munmap(ptr, 4096);
|
||||
sigaction(SIGBUS, &act, nullptr);
|
||||
|
||||
fprintf(stdout, "Fault Granularity: Split every 16 bytes\n");
|
||||
print_granule(" 16-bit", FaultOffset_16bit);
|
||||
print_granule(" 32-bit", FaultOffset_32bit);
|
||||
print_granule(" 64-bit", FaultOffset_64bit);
|
||||
print_granule("128-bit", FaultOffset_128bit);
|
||||
}
|
||||
} // namespace SIGBUSTest
|
||||
#endif
|
||||
|
||||
int main(int argc, char** argv, char** envp) {
|
||||
FEX::Config::InitializeConfigs(FEX::Config::PortableInformation {});
|
||||
FEXCore::Config::Initialize();
|
||||
@@ -114,6 +213,7 @@ int main(int argc, char** argv, char** envp) {
|
||||
Parser.add_option("--tso-emulation-info").action("store_true").help("Print how FEX is emulating the x86-TSO memory model.");
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
Parser.add_option("--test-fault-granularity").action("store_true").help("Show SIGBUS fault granularity");
|
||||
Parser.add_option("--identification-reg-info").action("store_true").help("Print identification registers");
|
||||
#endif
|
||||
|
||||
@@ -146,6 +246,12 @@ int main(int argc, char** argv, char** envp) {
|
||||
fprintf(stdout, GIT_DESCRIBE_STRING "\n");
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
if (Options.is_set_by_user("test_fault_granularity")) {
|
||||
SIGBUSTest::TestSIGBUS();
|
||||
}
|
||||
#endif
|
||||
|
||||
if (Options.is_set_by_user("install_prefix")) {
|
||||
char SelfPath[PATH_MAX];
|
||||
auto Result = readlink("/proc/self/exe", SelfPath, PATH_MAX);
|
||||
|
||||
@@ -411,7 +411,7 @@ public:
|
||||
|
||||
// Set the process personality here
|
||||
// Also, what about ADDR_LIMIT_3GB & co ?
|
||||
uint32_t Personality = personality(~0ULL);
|
||||
uint32_t Personality = personality(~0U);
|
||||
Personality |= ExecuteAll ? READ_IMPLIES_EXEC : 0;
|
||||
if (-1 == personality(Personality)) {
|
||||
LogMan::Msg::EFmt("Setting personality failed");
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include <PortabilityInfo.h>
|
||||
#include <Thunks.h>
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
@@ -50,7 +51,8 @@ public:
|
||||
}
|
||||
|
||||
void* GuestMmap(FEXCore::Core::InternalThreadState*, void* addr, size_t Size, int prot, int Flags, int fd, off_t offset) override {
|
||||
auto Ret = mmap(addr, Size, prot, Flags, fd, offset);
|
||||
// Force writeable to allow applying relocations
|
||||
auto Ret = mmap(addr, Size, prot | PROT_WRITE, Flags, fd, offset);
|
||||
if (Ret != MAP_FAILED && VAFileStart == 0) {
|
||||
VAFileStart = reinterpret_cast<uintptr_t>(Ret);
|
||||
}
|
||||
@@ -113,8 +115,7 @@ static FEXCore::Core::InternalThreadState* SetupCompileThread(FEXCore::Context::
|
||||
}
|
||||
|
||||
// Returns filename of generated cache on success
|
||||
static std::optional<std::string>
|
||||
GenerateSingleCache(const FEXCore::ExecutableFileInfo& Binary, fextl::set<uintptr_t> BlockList, std::string_view OutDir) {
|
||||
static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInfo& Binary, fextl::set<uintptr_t> BlockList, std::string_view OutDir) {
|
||||
uint64_t CodeCacheConfigId = 0; // TODO: Make unique to active configuration
|
||||
|
||||
ELFCodeLoader Loader(Binary.Filename.c_str(), -1, "", fextl::vector<fextl::string> {Binary.Filename.c_str()},
|
||||
@@ -123,7 +124,19 @@ GenerateSingleCache(const FEXCore::ExecutableFileInfo& Binary, fextl::set<uintpt
|
||||
fmt::print("Invalid or unsupported ELF file.\n");
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
const bool Is64Bit = Loader.Is64BitMode();
|
||||
auto SyscallOSABI = Is64Bit ? FEXCore::HLE::SyscallOSABI::OS_LINUX64 : FEXCore::HLE::SyscallOSABI::OS_LINUX32;
|
||||
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(SyscallOSABI);
|
||||
|
||||
// Populate relocations from ELF file
|
||||
{
|
||||
ELFParser RelocParser;
|
||||
RelocParser.ReadElf(Binary.Filename);
|
||||
Binary.Relocations = RelocParser.PopulateRelocations();
|
||||
SyscallHandler->FileInfo.Relocations = Binary.Relocations;
|
||||
}
|
||||
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS64BIT_MODE, Is64Bit ? "1" : "0");
|
||||
|
||||
// Load HostFeatures
|
||||
@@ -137,13 +150,9 @@ GenerateSingleCache(const FEXCore::ExecutableFileInfo& Binary, fextl::set<uintpt
|
||||
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext(HostFeatures);
|
||||
|
||||
auto SignalDelegation = std::make_unique<FEX::DummyHandlers::DummySignalDelegator>();
|
||||
|
||||
auto SyscallOSABI = Is64Bit ? FEXCore::HLE::SyscallOSABI::OS_LINUX64 : FEXCore::HLE::SyscallOSABI::OS_LINUX32;
|
||||
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(SyscallOSABI);
|
||||
|
||||
Loader.CalculateHWCaps(CTX.get());
|
||||
|
||||
auto SignalDelegation = std::make_unique<FEX::DummyHandlers::DummySignalDelegator>();
|
||||
CTX->SetSignalDelegator(SignalDelegation.get());
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
auto ThunkHandler = FEX::HLE::CreateThunkHandler();
|
||||
@@ -166,6 +175,25 @@ GenerateSingleCache(const FEXCore::ExecutableFileInfo& Binary, fextl::set<uintpt
|
||||
if (!ElfBase.has_value()) {
|
||||
ERROR_AND_DIE_FMT("Failed to load ELF file {} ({})", Binary.Filename, Binary.FileId);
|
||||
}
|
||||
|
||||
{
|
||||
ELFParser RelocParser;
|
||||
RelocParser.ReadElf(Binary.Filename);
|
||||
auto relocs32 = RelocParser.ReadRawRelocations32();
|
||||
|
||||
for (auto& reloc : relocs32) {
|
||||
if (ELF32_R_TYPE(reloc.r_info) == R_386_RELATIVE) {
|
||||
// The FEX-relocation is applied on top of this during cache serialization, so this must be countered
|
||||
uint32_t val = *reinterpret_cast<uint32_t*>(SyscallHandler->VAFileStart + reloc.r_offset) + SyscallHandler->VAFileStart;
|
||||
memcpy(reinterpret_cast<uint32_t*>(SyscallHandler->VAFileStart + reloc.r_offset), &val, sizeof(val));
|
||||
} else if (ELF32_R_TYPE(reloc.r_info) == R_386_32) {
|
||||
// The FEX-relocation is applied on top of this during cache serialization, so this must be countered
|
||||
uint32_t* orig = reinterpret_cast<uint32_t*>(SyscallHandler->VAFileStart + reloc.r_offset);
|
||||
uint32_t val = *orig + reloc.r_addend + SyscallHandler->VAFileStart;
|
||||
memcpy(orig, &val, sizeof(val));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
CTX->GetCodeCache().InitiateCacheGeneration();
|
||||
@@ -174,7 +202,12 @@ GenerateSingleCache(const FEXCore::ExecutableFileInfo& Binary, fextl::set<uintpt
|
||||
std::vector<std::unique_ptr<ELFCodeLoader>> LoaderMem;
|
||||
|
||||
fmt::print(stderr, "Compiling code...\n");
|
||||
FEX_CONFIG_OPT(MaxInst, MAXINST);
|
||||
for (auto Addr : BlockList) {
|
||||
if (!CTX->CheckIfBlockIsCacheable(*Thread, Addr + SyscallHandler->VAFileStart, MaxInst)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
CTX->CompileRIP(Thread, Addr + SyscallHandler->VAFileStart);
|
||||
}
|
||||
|
||||
|
||||
@@ -1031,6 +1031,10 @@ int32_t AskForDistroSelection(const std::span<const WebFileFetcher::FileTargets>
|
||||
if (!ArgOptions::DistroVersion.empty()) {
|
||||
Info.DistroVersion = ArgOptions::DistroVersion;
|
||||
}
|
||||
// explicit CLI selection must still run exact-match logic.
|
||||
if (!ArgOptions::DistroName.empty() || !ArgOptions::DistroVersion.empty()) {
|
||||
Info.Unknown = false;
|
||||
}
|
||||
|
||||
return _AskForDistroSelection(Info, Targets);
|
||||
}
|
||||
|
||||
@@ -5,12 +5,74 @@
|
||||
#include <fcntl.h>
|
||||
#include <fmt/format.h>
|
||||
#include <unistd.h>
|
||||
#include <vector>
|
||||
#include <xxhash.h>
|
||||
#include <functional>
|
||||
|
||||
namespace XXFileHash {
|
||||
// 32MB blocks
|
||||
constexpr static size_t BLOCK_SIZE = 32 * 1024 * 1024;
|
||||
class Reader {
|
||||
public:
|
||||
Reader(int fd, size_t Size)
|
||||
: fd {fd}
|
||||
, Size {Size} {}
|
||||
|
||||
virtual ~Reader() = default;
|
||||
|
||||
bool Initialized() const {
|
||||
return IsInitialized;
|
||||
}
|
||||
|
||||
using Callback = std::function<bool(const void* Data, size_t Size)>;
|
||||
virtual bool Read(Callback cb) = 0;
|
||||
|
||||
protected:
|
||||
int fd {};
|
||||
size_t Size {};
|
||||
bool IsInitialized {};
|
||||
};
|
||||
|
||||
class MemoryReader final : public Reader {
|
||||
public:
|
||||
MemoryReader(int fd, size_t Size)
|
||||
: Reader(fd, Size) {
|
||||
Ptr = reinterpret_cast<std::byte*>(mmap(nullptr, Size, PROT_READ, MAP_SHARED, fd, 0));
|
||||
IsInitialized = Ptr != MAP_FAILED;
|
||||
}
|
||||
|
||||
~MemoryReader() {
|
||||
munmap(reinterpret_cast<void*>(Ptr), Size);
|
||||
}
|
||||
|
||||
bool Read(Callback cb) override {
|
||||
auto ReadPtr = Ptr;
|
||||
const auto ReadEndPtr = Ptr + Size;
|
||||
size_t ReadSize {};
|
||||
|
||||
// Claim sequential access.
|
||||
::madvise(reinterpret_cast<void*>(ReadPtr), Size, MADV_SEQUENTIAL);
|
||||
|
||||
while (ReadPtr < ReadEndPtr) {
|
||||
ReadSize = std::min<size_t>(READ_BLOCK_SIZE, ReadEndPtr - ReadPtr);
|
||||
|
||||
if (!cb(ReadPtr, ReadSize)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Only allow a single block read to be resident.
|
||||
::madvise(reinterpret_cast<void*>(ReadPtr), ReadSize, MADV_DONTNEED);
|
||||
|
||||
ReadPtr += ReadSize;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
private:
|
||||
std::byte* Ptr {};
|
||||
|
||||
// Only allow 128MB in flight.
|
||||
constexpr static size_t READ_BLOCK_SIZE = 128 * 1024 * 1024;
|
||||
};
|
||||
|
||||
std::optional<uint64_t> HashFile(const fextl::string& Filepath) {
|
||||
int fd = open(Filepath.c_str(), O_RDONLY);
|
||||
if (fd == -1) {
|
||||
@@ -48,33 +110,39 @@ std::optional<uint64_t> HashFile(const fextl::string& Filepath) {
|
||||
return HadError();
|
||||
}
|
||||
|
||||
MemoryReader Read(fd, Size);
|
||||
|
||||
if (!Read.Initialized()) {
|
||||
return HadError();
|
||||
}
|
||||
|
||||
const auto Start = std::chrono::high_resolution_clock::now();
|
||||
auto Now = Start;
|
||||
const double SizeD = Size;
|
||||
std::vector<char> Data(BLOCK_SIZE);
|
||||
off_t CurrentOffset = 0;
|
||||
auto Now = std::chrono::high_resolution_clock::now();
|
||||
size_t CurrentOffset {};
|
||||
|
||||
// Let the kernel know that we will be reading linearly
|
||||
posix_fadvise(fd, 0, Size, POSIX_FADV_SEQUENTIAL);
|
||||
while (CurrentOffset < Size) {
|
||||
|
||||
ssize_t Result = pread(fd, Data.data(), BLOCK_SIZE, CurrentOffset);
|
||||
if (Result == -1) {
|
||||
return HadError();
|
||||
auto CB_XXH = [&](const void* Data, size_t BlockSize) -> bool {
|
||||
if (XXH3_64bits_update(State, Data, BlockSize) == XXH_ERROR) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (XXH3_64bits_update(State, Data.data(), Result) == XXH_ERROR) {
|
||||
return HadError();
|
||||
}
|
||||
auto Cur = std::chrono::high_resolution_clock::now();
|
||||
auto Dur = Cur - Now;
|
||||
if (Dur >= std::chrono::seconds(1)) {
|
||||
fmt::print("{:.2}% hashed\n", (double)CurrentOffset / SizeD * 100.0);
|
||||
Now = Cur;
|
||||
}
|
||||
CurrentOffset += Result;
|
||||
|
||||
CurrentOffset += BlockSize;
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
if (!Read.Read(CB_XXH)) {
|
||||
return HadError();
|
||||
}
|
||||
|
||||
const XXH64_hash_t Hash = XXH3_64bits_digest(State);
|
||||
const auto Hash = XXH3_64bits_digest(State);
|
||||
XXH3_freeState(State);
|
||||
|
||||
close(fd);
|
||||
|
||||
@@ -202,16 +202,20 @@ void UnmountRootFS() {
|
||||
|
||||
if (pid == 0) {
|
||||
const char* argv[5];
|
||||
argv[0] = "fusermount";
|
||||
argv[0] = "fusermount3";
|
||||
argv[1] = "-u";
|
||||
argv[2] = "-q";
|
||||
argv[3] = MountFolder.c_str();
|
||||
argv[4] = nullptr;
|
||||
|
||||
if (execvp(argv[0], (char* const*)argv) == -1) {
|
||||
fprintf(stderr, "fusermount failed to execute. You may have an mount living at '%s' to clean up now\n", MountFolder.c_str());
|
||||
fprintf(stderr, "Try `%s %s %s %s`\n", argv[0], argv[1], argv[2], argv[3]);
|
||||
exit(1);
|
||||
// Try again with `fusermount`
|
||||
argv[0] = "fusermount";
|
||||
if (execvp(argv[0], (char* const*)argv) == -1) {
|
||||
fprintf(stderr, "fusermount{3,} failed to execute. You may have an mount living at '%s' to clean up now\n", MountFolder.c_str());
|
||||
fprintf(stderr, "Try `%s %s %s %s`\n", argv[0], argv[1], argv[2], argv[3]);
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Wait for fusermount to leave
|
||||
|
||||
@@ -279,7 +279,7 @@ FileManager::FileManager(FEXCore::Context::Context* ctx)
|
||||
// Using a local struct for this is slightly less ugly than using self-capturing lambdas
|
||||
struct {
|
||||
decltype(FileManager::ThunkOverlays)& ThunkOverlays;
|
||||
decltype(ThunkDB)& ThunkDB;
|
||||
decltype(ThunkDB)& DB;
|
||||
const fextl::string& ThunkGuestPath;
|
||||
bool Is64BitMode;
|
||||
|
||||
@@ -301,7 +301,7 @@ FileManager::FileManager(FEXCore::Context::Context* ctx)
|
||||
|
||||
void InsertDependencies(const fextl::unordered_set<fextl::string>& Depends) {
|
||||
for (const auto& Depend : Depends) {
|
||||
auto& DBDepend = ThunkDB.at(Depend);
|
||||
auto& DBDepend = DB.at(Depend);
|
||||
if (DBDepend.Enabled) {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -196,14 +196,7 @@ restart: {
|
||||
|
||||
if (MappedPtr == MAP_FAILED && errno != EEXIST) {
|
||||
return reinterpret_cast<void*>(-errno);
|
||||
} else if (MappedPtr == MAP_FAILED || MappedPtr >= reinterpret_cast<void*>(TOP_KEY << FEXCore::Utils::FEX_PAGE_SHIFT)) {
|
||||
// Handles the case where MAP_FIXED_NOREPLACE failed with MAP_FAILED
|
||||
// or if the host system's kernel isn't new enough then it returns the wrong pointer
|
||||
if (MappedPtr != MAP_FAILED && MappedPtr >= reinterpret_cast<void*>(TOP_KEY << FEXCore::Utils::FEX_PAGE_SHIFT)) {
|
||||
// Make sure to munmap this so we don't leak memory
|
||||
::munmap(MappedPtr, length);
|
||||
}
|
||||
|
||||
} else if (MappedPtr == MAP_FAILED) {
|
||||
if (UpperPage == TOP_KEY) {
|
||||
BottomPage = BASE_KEY;
|
||||
Wrapped = true;
|
||||
@@ -240,13 +233,7 @@ restart: {
|
||||
void* MappedPtr = ::mmap(reinterpret_cast<void*>(PageAddr << FEXCore::Utils::FEX_PAGE_SHIFT),
|
||||
PagesLength << FEXCore::Utils::FEX_PAGE_SHIFT, prot, flags, fd, offset);
|
||||
|
||||
if (MappedPtr >= reinterpret_cast<void*>(TOP_KEY << FEXCore::Utils::FEX_PAGE_SHIFT) && (flags & FEX_MAP_FIXED_NOREPLACE)) {
|
||||
// Handles the case where MAP_FIXED_NOREPLACE isn't handled by the host system's
|
||||
// kernel and returns the wrong pointer
|
||||
// Make sure to munmap this so we don't leak memory
|
||||
::munmap(MappedPtr, length);
|
||||
return reinterpret_cast<void*>(-EEXIST);
|
||||
} else if (MappedPtr != MAP_FAILED) {
|
||||
if (MappedPtr != MAP_FAILED) {
|
||||
SetUsedPages(PageAddr, PagesLength);
|
||||
return MappedPtr;
|
||||
} else {
|
||||
|
||||
@@ -90,7 +90,7 @@ uint64_t BPFEmitter::HandleLoad(uint32_t BPFIP, const sock_filter* Inst) {
|
||||
// Must be smaller than scratch space size.
|
||||
VALIDATE(Inst->k < 16);
|
||||
|
||||
EMIT_INST(ldr(DestReg, REG_SECCOMP_DATA, offsetof(WorkingBuffer, ScratchMemory[Inst->k])));
|
||||
EMIT_INST(ldr(DestReg, REG_SECCOMP_DATA, ARRAY_OFFSETOF(WorkingBuffer, ScratchMemory, Inst->k)));
|
||||
break;
|
||||
case BPF_LEN:
|
||||
// Just returns the length of seccomp_data.
|
||||
@@ -114,7 +114,7 @@ uint64_t BPFEmitter::HandleStore(uint32_t BPFIP, const sock_filter* Inst) {
|
||||
// Must be smaller than scratch space size.
|
||||
VALIDATE(Inst->k < 16);
|
||||
|
||||
EMIT_INST(str(SrcReg, REG_SECCOMP_DATA, offsetof(WorkingBuffer, ScratchMemory[Inst->k])));
|
||||
EMIT_INST(str(SrcReg, REG_SECCOMP_DATA, ARRAY_OFFSETOF(WorkingBuffer, ScratchMemory, Inst->k)));
|
||||
|
||||
RETURN_SUCCESS();
|
||||
}
|
||||
|
||||
@@ -894,7 +894,7 @@ SignalDelegator::SignalDelegator(FEXCore::Context::Context* _CTX, const std::str
|
||||
|
||||
// Most signals default to termination
|
||||
// These ones are slightly different
|
||||
static constexpr std::array<std::pair<int, SignalDelegator::DefaultBehaviour>, 14> SignalDefaultBehaviours = {{
|
||||
static constexpr std::array<std::pair<int, SignalDelegator::DefaultBehaviourType>, 14> SignalDefaultBehaviours = {{
|
||||
{SIGQUIT, DEFAULT_COREDUMP},
|
||||
{SIGILL, DEFAULT_COREDUMP},
|
||||
{SIGTRAP, DEFAULT_COREDUMP},
|
||||
|
||||
@@ -174,7 +174,7 @@ private:
|
||||
|
||||
FEXCore::ArchHelpers::Arm64::UnalignedHandlerType UnalignedHandlerType {FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::HalfBarrier};
|
||||
|
||||
enum DefaultBehaviour {
|
||||
enum DefaultBehaviourType {
|
||||
DEFAULT_TERM,
|
||||
// Core dump based signals are supposed to have a coredump appear
|
||||
// For FEX's behaviour we don't really care right now
|
||||
@@ -201,7 +201,7 @@ private:
|
||||
kernel_sigaction OldAction {};
|
||||
FEX::HLE::HostSignalDelegatorFunctionForGuest GuestHandler {};
|
||||
GuestSigAction GuestAction {};
|
||||
DefaultBehaviour DefaultBehaviour {DEFAULT_TERM};
|
||||
DefaultBehaviourType DefaultBehaviour {DEFAULT_TERM};
|
||||
|
||||
// Callbacks
|
||||
fextl::vector<HostSignalDelegatorFunction> Handlers {};
|
||||
|
||||
@@ -180,6 +180,10 @@ void SignalDelegator::RestoreFrame_x64(FEXCore::Core::InternalThreadState* Threa
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// Technically if mxcsr contains invalid bits then rt_sigreturn should return -EINVAL.
|
||||
// TODO: FEX doesn't support this today.
|
||||
Frame->State.mxcsr = fpstate->mxcsr & 0xFFC0;
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.AbridgedFTW = fpstate->ftw;
|
||||
@@ -254,6 +258,9 @@ void SignalDelegator::RestoreFrame_ia32(FEXCore::Core::InternalThreadState* Thre
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// Invalid bits are silently masked off in 32-bit.
|
||||
Frame->State.mxcsr = fpstate->mxcsr & 0xFFC0;
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.AbridgedFTW = FEXCore::FPState::ConvertToAbridgedFTW(fpstate->ftw);
|
||||
@@ -330,6 +337,9 @@ void SignalDelegator::RestoreRTFrame_ia32(FEXCore::Core::InternalThreadState* Th
|
||||
CTX->SetXMMRegistersFromState(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// Invalid bits are silently masked off in 32-bit.
|
||||
Frame->State.mxcsr = fpstate->mxcsr & 0xFFC0;
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.AbridgedFTW = FEXCore::FPState::ConvertToAbridgedFTW(fpstate->ftw);
|
||||
@@ -471,6 +481,10 @@ uint64_t SignalDelegator::SetupFrame_x64(FEXCore::Core::InternalThreadState* Thr
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
// Save mxcsr and the default mask.
|
||||
fpstate->mxcsr = Frame->State.mxcsr;
|
||||
fpstate->mxcsr_mask = 0xFFC0;
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
fpstate->ftw = Frame->State.AbridgedFTW;
|
||||
@@ -594,6 +608,8 @@ uint64_t SignalDelegator::SetupFrame_ia32(FEXCore::Core::InternalThreadState* Th
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
fpstate->mxcsr = Frame->State.mxcsr;
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
// Reconstruct FSW
|
||||
@@ -729,6 +745,8 @@ uint64_t SignalDelegator::SetupRTFrame_ia32(FEXCore::Core::InternalThreadState*
|
||||
CTX->ReconstructXMMRegisters(Thread, fpstate->_xmm, nullptr);
|
||||
}
|
||||
|
||||
fpstate->mxcsr = Frame->State.mxcsr;
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
// Reconstruct FSW
|
||||
|
||||
@@ -202,7 +202,13 @@ FEXCore::HLE::ExecutableRangeInfo SyscallHandler::QueryGuestExecutableRange(FEXC
|
||||
return {Entry->first, Entry->second.Length, Entry->second.Prot.Writable};
|
||||
}
|
||||
|
||||
static fextl::vector<Elf64_Phdr> ReadELFHeaders(int FD, std::span<std::byte> HeaderData = {}) {
|
||||
struct ReadELFHeadersResult {
|
||||
fextl::vector<Elf64_Phdr> ProgramHeaders;
|
||||
fextl::robin_map<uint32_t, FEXCore::GuestRelocationType> Relocations;
|
||||
bool HasCodeRelocations;
|
||||
};
|
||||
|
||||
static ReadELFHeadersResult ReadELFHeaders(int FD, std::span<std::byte> HeaderData = {}) {
|
||||
std::string_view ELFMagic = ELFMAG;
|
||||
if (HeaderData.data()) {
|
||||
if (HeaderData.size_bytes() < ELFMagic.size() || std::memcmp(ELFMagic.data(), HeaderData.data(), ELFMagic.size()) != 0) {
|
||||
@@ -213,9 +219,18 @@ static fextl::vector<Elf64_Phdr> ReadELFHeaders(int FD, std::span<std::byte> Hea
|
||||
// Read from FD in case the caller didn't have a mapped header available
|
||||
}
|
||||
|
||||
// Re-open the file with a fresh file descriptor (and let ELFParser close it on return).
|
||||
// NOTE: FDs returned by dup() share the same cursor state, so reading from them would have observable side effects.
|
||||
auto NewFD = open(fextl::fmt::format("/proc/self/fd/{}", FD).c_str(), O_RDONLY);
|
||||
|
||||
ELFParser Parser;
|
||||
Parser.ReadElf(dup(FD));
|
||||
return std::move(Parser.phdrs);
|
||||
if (!Parser.ReadElf(NewFD)) {
|
||||
return {};
|
||||
}
|
||||
|
||||
auto Relocations = Parser.PopulateRelocations();
|
||||
auto HasCodeRelocations = Parser.HasCodeRelocations();
|
||||
return ReadELFHeadersResult {std::move(Parser.phdrs), std::move(Relocations), HasCodeRelocations};
|
||||
}
|
||||
|
||||
static void LoadCodeCache(FEXCore::Core::InternalThreadState& Thread, FEXCore::ExecutableFileSectionInfo& Section, uint64_t CodeCacheConfigId) {
|
||||
@@ -280,7 +295,7 @@ void* SyscallHandler::GuestMmap(bool Is64Bit, FEXCore::Core::InternalThreadState
|
||||
InvalidateCodeRangeIfNecessary(Thread, Result, Size);
|
||||
|
||||
if (LateMetadata) {
|
||||
auto CodeInvalidationlk = GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), Thread);
|
||||
auto CodeInvalidationlk = FEXCore::GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), Thread);
|
||||
CTX->AddForceTSOInformation(LateMetadata->VolatileValidRanges, std::move(LateMetadata->VolatileInstructions));
|
||||
}
|
||||
|
||||
@@ -320,7 +335,7 @@ uint64_t SyscallHandler::GuestMunmap(bool Is64Bit, FEXCore::Core::InternalThread
|
||||
InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uint64_t>(addr), Size);
|
||||
|
||||
if (length) {
|
||||
auto CodeInvalidationlk = GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), Thread);
|
||||
auto CodeInvalidationlk = FEXCore::GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), Thread);
|
||||
CTX->RemoveForceTSOInformation(reinterpret_cast<uint64_t>(addr), length);
|
||||
}
|
||||
|
||||
@@ -399,6 +414,34 @@ uint64_t SyscallHandler::GuestMprotect(FEXCore::Core::InternalThreadState* Threa
|
||||
}
|
||||
|
||||
InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uint64_t>(addr), len);
|
||||
|
||||
// Prepare for delayed code cache load after ld/Wine is done applying relocations.
|
||||
// Hooking into mprotect is a reliable heuristic that matches behavior of ld (for ELF) and Wine (for PE).
|
||||
// False-positives are avoided by setting RequiresDelayedCacheLoad in TrackMmap only for
|
||||
// binaries that we know will go through this path.
|
||||
fextl::vector<FEXCore::ExecutableFileSectionInfo> CachedSections;
|
||||
if (EnableCodeCaching && (prot & PROT_EXEC) && (prot & PROT_WRITE) == 0) {
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
|
||||
auto VMAEntry = VMATracking.FindVMAEntry(reinterpret_cast<uint64_t>(addr));
|
||||
auto Resource = VMAEntry != VMATracking.VMAs.end() ? VMAEntry->second.Resource : nullptr;
|
||||
if (Resource && Resource->MappedFile && Resource->RequiresDelayedCacheLoad) {
|
||||
Resource->RequiresDelayedCacheLoad = false;
|
||||
LogMan::Msg::IFmt("Triggering delayed cache load for {} after mprotect of {:#x}-{:#x}", Resource->MappedFile->Filename,
|
||||
VMAEntry->first, VMAEntry->first + VMAEntry->second.Length);
|
||||
|
||||
for (auto VMA = Resource->FirstVMA; VMA; VMA = VMA->ResourceNextVMA) {
|
||||
CachedSections.push_back(BuildSectionInfo(*Resource, VMA->Base, VMA->Length));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Trigger delayed cache load. This must be done separately since
|
||||
// LoadCodeCache will call interfaces that acquire the VMATracking mutex.
|
||||
for (auto& CachedSection : CachedSections) {
|
||||
LoadCodeCache(*Thread, CachedSection, CodeCacheConfigId);
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
@@ -484,7 +527,7 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
const bool MappedELFHeaderAgain = ResourceIt != ResourceEnd && offset == 0 && !ResourceIt->second.ProgramHeaders.empty();
|
||||
if (ResourceIt == ResourceEnd || MappedELFHeaderAgain) {
|
||||
// Create a new MappedResource for previously unseen file and for re-mappings of an ELF header
|
||||
ResourceIt = VMATracking.InsertMappedResource(mrid, {nullptr, nullptr, 0});
|
||||
ResourceIt = VMATracking.InsertMappedResource(mrid, VMATracking::MappedResource {nullptr, nullptr, 0, {}, {}});
|
||||
ResourceIt->second.Iterator = ResourceIt;
|
||||
Inserted = true;
|
||||
}
|
||||
@@ -493,19 +536,34 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
// Only handle FDs that are backed by regular files that are executable
|
||||
if (PathLength != -1 && S_ISREG(buf.st_mode) && (buf.st_mode & S_IXUSR)) {
|
||||
// ELF files that are mapped multiple times get a separate MappedResource for each base virtual address
|
||||
if (Inserted) {
|
||||
if ((prot & PROT_READ) && Inserted) {
|
||||
Resource->MappedFile = fextl::make_unique<FEXCore::ExecutableFileInfo>();
|
||||
Resource->MappedFile->Filename = fextl::string(Tmp, PathLength);
|
||||
Resource->MappedFile->FileId = CTX->GetCodeCache().ComputeCodeMapId(Resource->MappedFile->Filename, fd);
|
||||
|
||||
// Read ELF headers if applicable.
|
||||
// Read ELF headers if applicable and needed for code caching.
|
||||
// For performance, skip ELF checks if we're not mapping the file header
|
||||
bool CheckForElfFile = (offset == 0);
|
||||
bool CheckForElfFile = (offset == 0) && EnableCodeCaching;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
CheckForElfFile = true;
|
||||
#endif
|
||||
if (CheckForElfFile) {
|
||||
Resource->ProgramHeaders = ReadELFHeaders(fd, std::span {reinterpret_cast<std::byte*>(addr), length});
|
||||
auto ELFResult = ReadELFHeaders(fd, std::span {reinterpret_cast<std::byte*>(addr), length});
|
||||
Resource->ProgramHeaders = std::move(ELFResult.ProgramHeaders);
|
||||
Resource->MappedFile->Relocations = std::move(ELFResult.Relocations);
|
||||
Resource->RequiresDelayedCacheLoad = ELFResult.HasCodeRelocations;
|
||||
|
||||
// GuestRelocationType::Skip indicates to FEXOfflineCompiler that
|
||||
// any blocks covered by the relocation may not be cached.
|
||||
// At runtime, we can safely drop these relocations.
|
||||
for (auto it = Resource->MappedFile->Relocations.begin(); it != Resource->MappedFile->Relocations.end();) {
|
||||
if (it->second == FEXCore::GuestRelocationType::Skip) {
|
||||
it = Resource->MappedFile->Relocations.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Resource->ProgramHeaders.empty() || offset == 0, "Expected file offset 0 for the first mapping of an ELF "
|
||||
"file");
|
||||
}
|
||||
@@ -555,7 +613,7 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
auto [Iter, IterEnd] = VMATracking.FindResources(mrid);
|
||||
LOGMAN_THROW_A_FMT(Iter == IterEnd, "VMA tracking error");
|
||||
|
||||
Iter = VMATracking.InsertMappedResource(mrid, {nullptr, nullptr, 0});
|
||||
Iter = VMATracking.InsertMappedResource(mrid, VMATracking::MappedResource {nullptr, nullptr, 0, {}, {}});
|
||||
Resource = &Iter->second;
|
||||
Resource->Iterator = Iter;
|
||||
}
|
||||
@@ -566,7 +624,11 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
// FEXServer was requested to generate library caches on program launch.
|
||||
if (EnableCodeCaching && Resource && Resource->MappedFile && VMATracking::VMAProt::fromProt(prot).Executable) {
|
||||
if (Thread) {
|
||||
CachedSection.emplace(BuildSectionInfo(*Resource, addr, Size));
|
||||
if (!Resource->RequiresDelayedCacheLoad) {
|
||||
CachedSection.emplace(BuildSectionInfo(*Resource, addr, Size));
|
||||
} else {
|
||||
LogMan::Msg::IFmt("Delaying code cache load for {} until mprotect {:#x}-{:#x}", Resource->MappedFile->Filename, addr, addr + Size);
|
||||
}
|
||||
} else {
|
||||
// Cache can't be loaded with a thread; skip this for now
|
||||
LogMan::Msg::DFmt("Oops, tried caching without a thread: {}", Resource->MappedFile->Filename);
|
||||
@@ -627,7 +689,7 @@ void SyscallHandler::TrackShmat(FEXCore::Core::InternalThreadState* Thread, int
|
||||
|
||||
auto [Iter, IterEnd] = VMATracking.FindResources(mrid);
|
||||
if (Iter == IterEnd) {
|
||||
Iter = VMATracking.InsertMappedResource(mrid, {nullptr, nullptr, Length});
|
||||
Iter = VMATracking.InsertMappedResource(mrid, VMATracking::MappedResource {nullptr, nullptr, Length, {}, {}});
|
||||
Iter->second.Iterator = Iter;
|
||||
}
|
||||
auto Resource = &Iter->second;
|
||||
|
||||
@@ -281,6 +281,11 @@ void VMATracking::DeleteVMARange(FEXCore::Context::Context* CTX, uintptr_t Base,
|
||||
void VMATracking::ChangeProtectionFlags(uintptr_t Base, uintptr_t Length, VMAProt NewProt) {
|
||||
Mutex.check_lock_owned_by_self_as_write();
|
||||
|
||||
// Handle 0 size as no-op like the kernel
|
||||
if (Length == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
// This needs to handle multiple split-merge strategies:
|
||||
// 1) Exact overlap - No Split, no Merge. Only protection tracking changes.
|
||||
// 2) Exact base overlap - Single insert, can never fail.
|
||||
|
||||
@@ -49,6 +49,7 @@ struct MappedResource {
|
||||
uint64_t Length; // 0 if not fixed size
|
||||
ContainerType::iterator Iterator;
|
||||
|
||||
bool RequiresDelayedCacheLoad = false;
|
||||
fextl::vector<Elf64_Phdr> ProgramHeaders;
|
||||
};
|
||||
|
||||
|
||||
@@ -189,6 +189,9 @@ FEX::HLE::ThreadStateObject* ThreadManager::CreateThread(uint64_t InitialRIP, ui
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_CallRetStacks", reinterpret_cast<void*>(AllocBase), CALLRET_STACK_ALLOC_SIZE);
|
||||
|
||||
// Disable HUGEPAGE on callret stacks.
|
||||
FEXCore::Allocator::VirtualTHPControl(reinterpret_cast<void*>(AllocBase), CALLRET_STACK_ALLOC_SIZE, FEXCore::Allocator::THPControl::Disable);
|
||||
|
||||
// Set the base used for invalidation to the start past the guard pages
|
||||
ThreadStateObject->Thread->CallRetStackBase = reinterpret_cast<void*>(AllocBase + FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
::mprotect(ThreadStateObject->Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE, PROT_READ | PROT_WRITE);
|
||||
|
||||
@@ -221,7 +221,7 @@ public:
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
auto CodeInvalidationlk = GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), CallingThread);
|
||||
auto CodeInvalidationlk = FEXCore::GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), CallingThread);
|
||||
CTX->InvalidateCodeBuffersCodeRange(Start, Length);
|
||||
for (auto& Thread : Threads) {
|
||||
CTX->InvalidateThreadCachedCodeRange(Thread->Thread, Start, Length);
|
||||
@@ -235,7 +235,7 @@ public:
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
auto CodeInvalidationlk = GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), CallingThread);
|
||||
auto CodeInvalidationlk = FEXCore::GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), CallingThread);
|
||||
CTX->InvalidateCodeBuffersCodeRange(Start, Length);
|
||||
for (auto& Thread : Threads) {
|
||||
CTX->InvalidateThreadCachedCodeRange(Thread->Thread, Start, Length);
|
||||
|
||||
@@ -208,7 +208,7 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(
|
||||
futex, [](FEXCore::Core::CpuStateFrame* Frame, int* uaddr, int futex_op, int val, const timespec32* timeout, int* uaddr2, uint32_t val3) -> uint64_t {
|
||||
void* timeout_ptr = (void*)timeout;
|
||||
const void* timeout_ptr = (const void*)timeout;
|
||||
struct timespec tp64 {};
|
||||
int cmd = futex_op & FUTEX_CMD_MASK;
|
||||
if (timeout && (cmd == FUTEX_WAIT || cmd == FUTEX_LOCK_PI || cmd == FUTEX_WAIT_BITSET || cmd == FUTEX_WAIT_REQUEUE_PI)) {
|
||||
|
||||
@@ -325,7 +325,7 @@ MakeHostTrampolineForGuestFunction(void* HostPacker, uintptr_t GuestTarget, uint
|
||||
LOGMAN_THROW_A_FMT(ThunkHandler->HostTrampolineInstanceDataPtr != MAP_FAILED, "Failed to mmap HostTrampolineInstanceDataPtr");
|
||||
}
|
||||
|
||||
auto HostTrampoline = reinterpret_cast<HostToGuestTrampolinePtr* const>(ThunkHandler->HostTrampolineInstanceDataPtr);
|
||||
auto HostTrampoline = reinterpret_cast<HostToGuestTrampolinePtr*>(ThunkHandler->HostTrampolineInstanceDataPtr);
|
||||
ThunkHandler->HostTrampolineInstanceDataAvailable -= HostToGuestTrampolineSize;
|
||||
ThunkHandler->HostTrampolineInstanceDataPtr += HostToGuestTrampolineSize;
|
||||
memcpy(HostTrampoline, (void*)&HostToGuestTrampolineTemplate, HostToGuestTrampolineSize);
|
||||
|
||||
@@ -703,23 +703,12 @@ void LoadFEXGeneratedCode(FEXCore::Core::InternalThreadState* Thread, bool Is64B
|
||||
Mapping->X86GeneratedCodePtr = Result;
|
||||
}
|
||||
} else {
|
||||
// First 64bit page
|
||||
constexpr uintptr_t LOCATION_MAX = 0x1'0000'0000;
|
||||
|
||||
// We need to have the sigret handler in the lower 32bits of memory space
|
||||
// Scan top down and try to allocate a location
|
||||
for (size_t Location = 0xFFFF'E000; Location != 0x0; Location -= PageSize) {
|
||||
auto Ptr = Handler->GuestMmap(Is64Bit, Thread, reinterpret_cast<void*>(Location), PageSize, PROT_READ | PROT_WRITE,
|
||||
MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
if (!FEX::HLE::HasSyscallError(Ptr) && reinterpret_cast<uintptr_t>(Ptr) >= LOCATION_MAX) {
|
||||
// Failed to map in the lower 32bits
|
||||
// Try again
|
||||
// Can happen in the case that host kernel ignores MAP_FIXED_NOREPLACE
|
||||
Handler->GuestMunmap(Thread, Ptr, PageSize);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!FEX::HLE::HasSyscallError(Ptr)) {
|
||||
Mapping->X86GeneratedCodePtr = Ptr;
|
||||
break;
|
||||
|
||||
@@ -147,7 +147,8 @@ static void IteratePids() {
|
||||
auto ExePath = Entry.path() / "exe";
|
||||
|
||||
// If cmdline doesn't exist then skip.
|
||||
if (!std::filesystem::exists(CMDLinePath)) {
|
||||
std::error_code ec;
|
||||
if (!std::filesystem::exists(CMDLinePath, ec) || ec) {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -177,7 +178,6 @@ static void IteratePids() {
|
||||
}
|
||||
}
|
||||
|
||||
std::error_code ec;
|
||||
std::string exe_link = std::filesystem::read_symlink(ExePath, ec);
|
||||
|
||||
auto deleted_pos = exe_link.find(" (deleted)");
|
||||
|
||||
@@ -64,12 +64,10 @@ ExitFunctionEC:
|
||||
strb wzr, [x17, #0x0] // ChpeV2CpuAreaInfo->InSimulation
|
||||
ldr x17, [x17, #0x20] // ChpeV2CpuAreaInfo->SuspendDoorbell
|
||||
ldr w17, [x17]
|
||||
cbz w17, no_suspend
|
||||
.global ExitFunctionSuspendPoint
|
||||
ExitFunctionSuspendPoint:
|
||||
brk #0xCAFE
|
||||
// Will resume here
|
||||
no_suspend:
|
||||
cbnz w17, ExitFunctionSuspendPoint
|
||||
|
||||
.global ExitFunctionSuspendResumePoint
|
||||
ExitFunctionSuspendResumePoint:
|
||||
// Either return to an exit thunk (return to ARM64EC function) or call an entry thunk (call to ARM64EC function).
|
||||
// It is assumed that a 'blr x16' instruction is only ever used to call into x86 code from an exit thunk, and that all
|
||||
// exported ARM64EC functions have a 4-byte offset to their entry thunk immediately before their first instruction.
|
||||
@@ -99,6 +97,11 @@ ret_sp_misaligned:
|
||||
ldr lr, [lr, #:lo12:X64ReturnInstr]
|
||||
br x17
|
||||
|
||||
.global ExitFunctionSuspendPoint
|
||||
ExitFunctionSuspendPoint:
|
||||
brk #0xCAFE
|
||||
// Resume will jump back to `ExitFunctionSuspendResumePoint`
|
||||
|
||||
// Makes a wrapper for calling a system call directly, skipping the usual ntdll thunks
|
||||
#define HASH #
|
||||
#define DIRECT_SYSCALL_WRAPPER(Name, WineIdName, WindowsId) \
|
||||
|
||||
@@ -27,6 +27,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
|
||||
#include "Windows/Common/Allocator.h"
|
||||
#include "Common/CallRetStack.h"
|
||||
#include "Common/JITGuardPage.h"
|
||||
#include "Common/Config.h"
|
||||
@@ -68,6 +69,7 @@ extern IMAGE_DOS_HEADER __ImageBase; // Provided by the linker
|
||||
extern void* ExitFunctionEC;
|
||||
extern void* CheckCall;
|
||||
extern void* ExitFunctionSuspendPoint;
|
||||
extern void* ExitFunctionSuspendResumePoint;
|
||||
|
||||
void* X64ReturnInstr; // See Module.S
|
||||
uintptr_t NtDllBase;
|
||||
@@ -593,6 +595,8 @@ NTSTATUS ProcessInit() {
|
||||
const bool IsWine = !!GetProcAddress(NtDll, "wine_get_version");
|
||||
OvercommitTracker.emplace(IsWine);
|
||||
|
||||
FEX::Windows::Allocator::SetupHooks(NtDll);
|
||||
|
||||
{
|
||||
auto HostFeatures = FEX::Windows::CPUFeatures::FetchHostFeatures(IsWine);
|
||||
CTX = FEXCore::Context::Context::CreateNewContext(HostFeatures);
|
||||
@@ -673,7 +677,7 @@ bool ResetToConsistentStateImpl(const ThreadCPUArea CPUArea, EXCEPTION_RECORD* E
|
||||
// A suspend interrupt can occur in ExitFunctionEC before InSimulation is unset and set SuspendDoorbell. If this
|
||||
// occurs then it is still our duty to cooperatively suspend with an appropriate context. To support this, after
|
||||
// unsetting InSimulation a brk #0xCAFE instruction will be raised that we can handle here.
|
||||
NativeContext->Pc += 4; // Skip over the brk instruction when we resume
|
||||
NativeContext->Pc = reinterpret_cast<uintptr_t>(ExitFunctionSuspendResumePoint); // Jump to the suspend resume point.
|
||||
*CPUArea.Area->SuspendDoorbell = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,243 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
|
||||
#include <array>
|
||||
#include <chrono>
|
||||
#include <libloaderapi.h>
|
||||
#include <sysinfoapi.h>
|
||||
#include <synchapi.h>
|
||||
#include <windef.h>
|
||||
#include <winternl.h>
|
||||
#include <winnt.h>
|
||||
#include <wine/debug.h>
|
||||
|
||||
namespace FEX::Windows::Allocator {
|
||||
#define PR_SET_VMA 0x53564d41
|
||||
#define PR_SET_VMA_ANON_NAME 0
|
||||
|
||||
#define MADV_HUGEPAGE 14
|
||||
#define MADV_NOHUGEPAGE 15
|
||||
|
||||
namespace Trampoline {
|
||||
struct madvise_data {
|
||||
const void* addr;
|
||||
size_t size;
|
||||
int advise;
|
||||
};
|
||||
|
||||
struct prctl_data {
|
||||
int op;
|
||||
uint64_t attr;
|
||||
const void* addr;
|
||||
size_t size;
|
||||
const char* name;
|
||||
uint64_t ret;
|
||||
};
|
||||
|
||||
__attribute__((naked)) uint64_t wine_prctl(prctl_data* d) {
|
||||
asm volatile(
|
||||
R"(
|
||||
.globl wine_prctl_begin
|
||||
wine_prctl_begin:
|
||||
mov x19, x0;
|
||||
mov x8, 167; // prctl
|
||||
ldr x0, [x19]; // op
|
||||
ldp x1, x2, [x19, %[attr_offset]]; // {attr, addr}
|
||||
ldp x3, x4, [x19, %[size_offset]]; // {size, name}
|
||||
svc #0;
|
||||
str x0, [x19, %[ret_offset]];
|
||||
// Tell wine it was all groovy.
|
||||
mov x0, 0;
|
||||
ret;
|
||||
|
||||
.globl wine_prctl_end
|
||||
wine_prctl_end:
|
||||
)"
|
||||
:
|
||||
: [attr_offset] "i"(offsetof(prctl_data, attr)), [size_offset] "i"(offsetof(prctl_data, size)), [ret_offset] "i"(offsetof(prctl_data, ret))
|
||||
: "memory");
|
||||
};
|
||||
|
||||
__attribute__((naked)) uint64_t wine_madvise(madvise_data* d) {
|
||||
asm volatile(R"(
|
||||
.globl wine_madvise_begin
|
||||
wine_madvise_begin:
|
||||
mov x8, 233; // madvise
|
||||
ldr x2, [x0, %[advise_offset]]; // advise
|
||||
ldp x0, x1, [x0]; // {addr, size}
|
||||
svc #0;
|
||||
// Tell wine it was all groovy.
|
||||
mov x0, 0;
|
||||
ret;
|
||||
|
||||
.globl wine_madvise_end
|
||||
wine_madvise_end:
|
||||
)" ::[advise_offset] "i"(offsetof(madvise_data, advise))
|
||||
: "memory");
|
||||
}
|
||||
|
||||
extern "C" uint64_t wine_madvise_begin;
|
||||
extern "C" uint64_t wine_madvise_end;
|
||||
|
||||
void* const wine_madvise_begin_loc = &wine_madvise_begin;
|
||||
void* const wine_madvise_end_loc = &wine_madvise_end;
|
||||
|
||||
extern "C" uint64_t wine_prctl_begin;
|
||||
extern "C" uint64_t wine_prctl_end;
|
||||
|
||||
void* const wine_prctl_begin_loc = &wine_prctl_begin;
|
||||
void* const wine_prctl_end_loc = &wine_prctl_end;
|
||||
|
||||
extern NTSTATUS(WINAPI* __wine_unix_call_dispatcher)(uint64_t, unsigned int, void*);
|
||||
decltype(__wine_unix_call_dispatcher) WineUnixCall;
|
||||
|
||||
enum unix_function_indexes {
|
||||
INDEX_PRCTL = 0,
|
||||
INDEX_MADVISE = 1,
|
||||
INDEX_MAX,
|
||||
};
|
||||
static std::array<void*, INDEX_MAX> unix_functions {};
|
||||
|
||||
static uint64_t wine_prctl_trampoline(int op, uint64_t attr, const void* addr, size_t size, const char* name) {
|
||||
prctl_data d {
|
||||
.op = op,
|
||||
.attr = attr,
|
||||
.addr = addr,
|
||||
.size = size,
|
||||
.name = name,
|
||||
};
|
||||
WineUnixCall(reinterpret_cast<uint64_t>(unix_functions.data()), INDEX_PRCTL, &d);
|
||||
return d.ret;
|
||||
}
|
||||
|
||||
static uint64_t wine_madvise_trampoline(const void* addr, size_t size, int advice) {
|
||||
madvise_data d {
|
||||
.addr = addr,
|
||||
.size = size,
|
||||
.advise = advice,
|
||||
};
|
||||
WineUnixCall(reinterpret_cast<uint64_t>(unix_functions.data()), INDEX_MADVISE, &d);
|
||||
return 0;
|
||||
}
|
||||
|
||||
void VirtualName(const char* Name, const void* Ptr, size_t Size) {
|
||||
static bool Supports {true};
|
||||
if (Supports) {
|
||||
auto Result = wine_prctl_trampoline(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, Name);
|
||||
if (Result != 0) {
|
||||
// Disable any additional attempts.
|
||||
Supports = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void VirtualTHPControl(const void* Ptr, size_t Size, FEXCore::Allocator::THPControl Control) {
|
||||
wine_madvise_trampoline(Ptr, Size, Control == FEXCore::Allocator::THPControl::Disable ? MADV_NOHUGEPAGE : MADV_HUGEPAGE);
|
||||
}
|
||||
} // namespace Trampoline
|
||||
|
||||
// This code path will eventually crash once Wine and the kernel implements `userspace syscall dispatch`.
|
||||
// FEX will need to switch over to using WINE's unixlib syscall approach then.
|
||||
// See `SetupHooks` for why unixlib doesn't work today.
|
||||
namespace Illegal {
|
||||
__attribute__((naked)) uint64_t prctl(int op, uint64_t attr, const void* addr, size_t size, const char* Name) {
|
||||
asm volatile(R"(
|
||||
mov x8, 167; // prctl
|
||||
svc #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) uint64_t madvise(const void* addr, size_t size, int advice) {
|
||||
asm volatile(R"(
|
||||
mov x8, 233; // madvise
|
||||
svc #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
void VirtualName(const char* Name, const void* Ptr, size_t Size) {
|
||||
static bool Supports {true};
|
||||
if (Supports) {
|
||||
auto Result = prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, Name);
|
||||
if (Result != 0) {
|
||||
// Disable any additional attempts.
|
||||
Supports = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void VirtualTHPControl(const void* Ptr, size_t Size, FEXCore::Allocator::THPControl Control) {
|
||||
madvise(Ptr, Size, Control == FEXCore::Allocator::THPControl::Disable ? MADV_NOHUGEPAGE : MADV_HUGEPAGE);
|
||||
}
|
||||
} // namespace Illegal
|
||||
|
||||
void SetupHooks(HMODULE ntdll) {
|
||||
// If this symbol doesn't exist, then we aren't running under WINE.
|
||||
const auto Sym = GetProcAddress(ntdll, "__wine_unix_call_dispatcher");
|
||||
|
||||
if (!Sym) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::Allocator::HookPtrs Ptrs {};
|
||||
|
||||
// Wine will soon require us to use unixlib for calling helper routines that use Linux syscalls.
|
||||
// It needs this for `userspace syscall dispatch` to capture rogue applications doing raw Windows syscalls.
|
||||
// If FEX doesn't use unixlib, then when WINE and a kernel implements this, then our "illegal" path will start crashing.
|
||||
//
|
||||
// This code currently conflicts with FEX's `Call Checker` since the latter captures uses of `unixlib`.
|
||||
// We hence have to stick to the `Illegal::` functions until a workaround is implemented.
|
||||
if constexpr (false) {
|
||||
// NTSTATUS __wine_unix_call_dispatcher( unixlib_handle_t, unsigned int, void * );
|
||||
// - unixlib_handle_t is just an array of functions
|
||||
// - uint32_t is just an index in to that
|
||||
// - void* is the user provided pointer, gets loaded in to x0 in for the unix_function called.
|
||||
// - Return value - SUCCESS or other error.
|
||||
Trampoline::WineUnixCall = *reinterpret_cast<decltype(Trampoline::WineUnixCall)*>(Sym);
|
||||
|
||||
// This code must be copied over to allocated memory from top down allocations apparently.
|
||||
auto Code = reinterpret_cast<uint8_t*>(
|
||||
::VirtualAlloc(nullptr, FEXCore::Utils::FEX_PAGE_SIZE, MEM_RESERVE | MEM_COMMIT | MEM_TOP_DOWN, PAGE_EXECUTE_READWRITE));
|
||||
if (!Code) {
|
||||
return;
|
||||
}
|
||||
|
||||
size_t CurrentOffset {};
|
||||
const size_t prctl_size =
|
||||
reinterpret_cast<uintptr_t>(Trampoline::wine_prctl_end_loc) - reinterpret_cast<uintptr_t>(Trampoline::wine_prctl_begin_loc);
|
||||
const size_t madvise_size =
|
||||
reinterpret_cast<uintptr_t>(Trampoline::wine_madvise_end_loc) - reinterpret_cast<uintptr_t>(Trampoline::wine_madvise_begin_loc);
|
||||
|
||||
// Copy prctl.
|
||||
memcpy(Code + CurrentOffset, Trampoline::wine_prctl_begin_loc, prctl_size);
|
||||
Trampoline::unix_functions[Trampoline::INDEX_PRCTL] = Code + CurrentOffset;
|
||||
CurrentOffset += prctl_size;
|
||||
|
||||
// Copy madvise.
|
||||
memcpy(Code + CurrentOffset, Trampoline::wine_madvise_begin_loc, madvise_size);
|
||||
Trampoline::unix_functions[Trampoline::INDEX_MADVISE] = Code + CurrentOffset;
|
||||
|
||||
// Protect the page now.
|
||||
FEXCore::Allocator::VirtualProtect(Code, FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::Read | FEXCore::Allocator::ProtectOptions::Exec);
|
||||
|
||||
Ptrs = {
|
||||
.VirtualName = Trampoline::VirtualName,
|
||||
.VirtualTHPControl = Trampoline::VirtualTHPControl,
|
||||
};
|
||||
} else {
|
||||
Ptrs = {
|
||||
.VirtualName = Illegal::VirtualName,
|
||||
.VirtualTHPControl = Illegal::VirtualTHPControl,
|
||||
};
|
||||
}
|
||||
|
||||
SYSTEM_INFO system_info {};
|
||||
GetSystemInfo(&system_info);
|
||||
FEXCore::Allocator::SetupHooks(system_info.dwPageSize, Ptrs);
|
||||
}
|
||||
} // namespace FEX::Windows::Allocator
|
||||
@@ -0,0 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <winternl.h>
|
||||
|
||||
namespace FEX::Windows::Allocator {
|
||||
void SetupHooks(HMODULE ntdll);
|
||||
}
|
||||
@@ -7,6 +7,7 @@ target_compile_options(CommonWindowsRuntime PRIVATE -Wno-inconsistent-dllimport)
|
||||
target_include_directories(CommonWindowsRuntime PRIVATE "${CMAKE_SOURCE_DIR}/Source/Windows/include/")
|
||||
|
||||
add_library(CommonWindows STATIC
|
||||
Allocator.cpp
|
||||
CPUFeatures.cpp
|
||||
SHMStats.cpp
|
||||
InvalidationTracker.cpp
|
||||
|
||||
@@ -25,6 +25,11 @@ void InitializeThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
const void* CallRetStackAlloc = ::VirtualAlloc(
|
||||
nullptr, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE + 2 * FEXCore::Utils::FEX_PAGE_SIZE, MEM_RESERVE, PAGE_NOACCESS);
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_CallRetStacks", CallRetStackAlloc,
|
||||
FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE + 2 * FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
FEXCore::Allocator::VirtualTHPControl(CallRetStackAlloc, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE + 2 * FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::THPControl::Disable);
|
||||
|
||||
Thread->CallRetStackBase = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(CallRetStackAlloc) + FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
::VirtualAlloc(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE, MEM_COMMIT, PAGE_READWRITE);
|
||||
|
||||
|
||||
@@ -28,6 +28,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
|
||||
#include "Windows/Common/Allocator.h"
|
||||
#include "Common/CallRetStack.h"
|
||||
#include "Common/JITGuardPage.h"
|
||||
#include "Common/Config.h"
|
||||
@@ -524,6 +525,8 @@ void BTCpuProcessInit() {
|
||||
const bool IsWine = !!GetProcAddress(NtDll, "wine_get_version");
|
||||
OvercommitTracker.emplace(IsWine);
|
||||
|
||||
FEX::Windows::Allocator::SetupHooks(NtDll);
|
||||
|
||||
{
|
||||
auto HostFeatures = FEX::Windows::CPUFeatures::FetchHostFeatures(IsWine);
|
||||
// AVX is unsupported for WOW64
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# FEX-2603
|
||||
# FEX-2604
|
||||
|
||||
## FEXCore
|
||||
See [FEXCore/Readme.md](../FEXCore/Readme.md) for more details
|
||||
|
||||
@@ -6,9 +6,10 @@ set(TESTS
|
||||
FileMappingBaseAddress
|
||||
Filesystem
|
||||
InterruptableConditionVariable
|
||||
StringUtils)
|
||||
StringUtils
|
||||
WildcardMatcher)
|
||||
|
||||
list(APPEND LIBS Common FEXCore JemallocLibs)
|
||||
list(APPEND LIBS Common FEXCore FEXCore_Base JemallocLibs)
|
||||
|
||||
foreach(API_TEST ${TESTS})
|
||||
add_executable(${API_TEST} ${API_TEST}.cpp)
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/WildcardMatcher.h>
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
using namespace FEXCore::Utils::Wildcard;
|
||||
|
||||
TEST_CASE("Singular regex") {
|
||||
CHECK(Matches("a", "a"));
|
||||
CHECK(Matches("a*", "a*"));
|
||||
CHECK(Matches("a*", "aaaaaaa"));
|
||||
}
|
||||
|
||||
TEST_CASE("Concat regex") {
|
||||
CHECK(Matches("aaa", "aaa"));
|
||||
CHECK(Matches("ab", "ab"));
|
||||
CHECK(!Matches("a", "ab"));
|
||||
CHECK(!Matches("ab", "a"));
|
||||
}
|
||||
TEST_CASE("Wildcard beginning end") {
|
||||
CHECK(Matches("a*", "a"));
|
||||
CHECK(Matches("*a", "a"));
|
||||
CHECK(Matches("*a*", "a"));
|
||||
}
|
||||
TEST_CASE("Wildcard middle") {
|
||||
CHECK(Matches("test*pattern", "test__pattern"));
|
||||
}
|
||||
|
||||
TEST_CASE("Wildcard mult") {
|
||||
CHECK(Matches("test*pattern*more", "test__pattern__more"));
|
||||
CHECK(Matches("test**pattern", "test_pattern"));
|
||||
}
|
||||
|
||||
TEST_CASE("Wildcard regex simple") {
|
||||
CHECK(Matches("*", ""));
|
||||
CHECK(Matches("*", "setup.json"));
|
||||
CHECK(Matches("test*pattern", "test__pattern"));
|
||||
CHECK(!Matches("setup.*", "setupjson"));
|
||||
CHECK(Matches("setup*", "setup.json"));
|
||||
CHECK(Matches("setup*", "setup/setup.json"));
|
||||
CHECK(Matches("*setup*", "setup/setup.json"));
|
||||
}
|
||||
|
||||
|
||||
TEST_CASE("FEX regex") {
|
||||
CHECK(Matches("*Config*", "/home/ubuntu/.fex-emu/Config.json"));
|
||||
CHECK(Matches("*Config.json", "/home/ubuntu/.fex-emu/Config.json"));
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX-Emu had a bug where the `prefetch NTA/T0/T1/T2` encoding range wasn't handling nops correctly.
|
||||
; While the first four instructions in the group16 encoding range is considered prefetch,
|
||||
; If the destination is encoded as a register then it is a nop instead.
|
||||
; This is to preserve legacy instruction behaviour where this entire group was considered NOP encoding.
|
||||
; FEX was accidentally declaring these encoded nop instructions to be invalid.
|
||||
|
||||
; nop ebx that overlaps `prefetch t0` instruction.
|
||||
; Just ensure it executes. Seen in `Devil May Cry 4`
|
||||
db 0x0f, 0x18, 0xcb
|
||||
|
||||
hlt
|
||||
@@ -0,0 +1,69 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"R12": "0x55",
|
||||
"R13": "0x890",
|
||||
"R14": "0x55"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov rsi, 0xe0000080
|
||||
mov rsp, 0xe0001000
|
||||
|
||||
; Zero shift amount
|
||||
xor ecx, ecx
|
||||
|
||||
; Zero all flags
|
||||
xor eax, eax
|
||||
push rax
|
||||
popfq
|
||||
|
||||
mov r8b, 255
|
||||
mov r10b, 127
|
||||
mov r11b, 1
|
||||
|
||||
|
||||
add r8b, r11b ; Sets CF, ZF, PF, AF, zeroes OF, SF
|
||||
; Shift by zero, flags should be unaffected
|
||||
; This tests that we didn't optimize away the flag calculations of the add
|
||||
shl rax, cl
|
||||
|
||||
; Ensure we can't predict the next block
|
||||
lea rdi, [rel .next]
|
||||
mov [rsi - 8], rdi
|
||||
jmp [rsi - 8]
|
||||
|
||||
.next:
|
||||
pushfq
|
||||
pop r12
|
||||
|
||||
; Mask with flags we care about
|
||||
and r12, 0x8d5
|
||||
|
||||
add r10b, r11b ; Sets OF, SF, AF, zeroes ZF, CF, PF
|
||||
shr rax, cl
|
||||
|
||||
lea rdi, [rel .next2]
|
||||
mov [rsi - 8], rdi
|
||||
jmp [rsi - 8]
|
||||
|
||||
.next2:
|
||||
pushfq
|
||||
pop r13
|
||||
and r13, 0x8d5
|
||||
|
||||
mov r8b, 255
|
||||
add r8b, r11b ; Sets CF, ZF, PF, AF, zeroes OF, SF
|
||||
sar rax, cl
|
||||
|
||||
lea rdi, [rel .next3]
|
||||
mov [rsi - 8], rdi
|
||||
jmp [rsi - 8]
|
||||
|
||||
.next3:
|
||||
pushfq
|
||||
pop r14
|
||||
and r14, 0x8d5
|
||||
|
||||
hlt
|
||||
@@ -23,8 +23,30 @@
|
||||
setnb al ; stores xmm12 >= xmm13, i.e. abs(REF) * tolerance >= abs(REF - X)
|
||||
%endmacro
|
||||
|
||||
;; Double-precision variant.
|
||||
;; Clobbers xmm12, xmm13, xmm14, xmm15
|
||||
;; Returns result in al.
|
||||
;; Arguments are in memory locations.
|
||||
%macro check_relerr_d 3; %1=REF %2=X %3=TOLERANCE
|
||||
movsd xmm12, qword [ %1 ] ; xmm12 has REF
|
||||
movsd xmm13, qword [ %2 ] ; xmm13 has X
|
||||
movsd xmm15, qword [rel abs_mask_double] ; xmm15 has the abs double mask
|
||||
movapd xmm14, xmm12 ; xmm14 has REF
|
||||
subsd xmm14, xmm13 ; xmm14 = REF - X
|
||||
andpd xmm12, xmm15 ; xmm12 = abs(REF)
|
||||
mulsd xmm12, qword [ %3 ] ; xmm12 = abs(REF) * tolerance
|
||||
movapd xmm13, xmm14 ; xmm13 = REF - X
|
||||
andpd xmm13, xmm15 ; xmm13 = abs(REF - X)
|
||||
|
||||
xor eax, eax ; clears eax
|
||||
comisd xmm12, xmm13 ; compares xmm12 and xmm13
|
||||
setnb al ; stores xmm12 >= xmm13, i.e. abs(REF) * tolerance >= abs(REF - X)
|
||||
%endmacro
|
||||
|
||||
%macro define_check_data_constants 0
|
||||
abs_mask_float:
|
||||
abs_mask_float:
|
||||
dd 0x7fffffff ; Bitmask to get absolute value of a float (single precision)
|
||||
abs_mask_double:
|
||||
dq 0x7fffffffffffffff ; Bitmask to get absolute value of a double
|
||||
%endmacro
|
||||
%endif
|
||||
@@ -0,0 +1,28 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x5152535455565758",
|
||||
"RDX": "0x5858585858585858",
|
||||
"RDI": "0xDFFFFFFF",
|
||||
"RSI": "0xE0000000"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov rdx, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [rdx], rax
|
||||
mov rax, 0x5152535455565758
|
||||
mov [rdx + 8], rax
|
||||
|
||||
; Deliberately overlapping source and destination
|
||||
lea rdi, [rdx + 7]
|
||||
lea rsi, [rdx + 8]
|
||||
|
||||
std
|
||||
mov rcx, 8
|
||||
rep movsb ; rdi <- rsi
|
||||
|
||||
mov rdx, [rdx]
|
||||
hlt
|
||||
@@ -0,0 +1,28 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x5152535455565758",
|
||||
"RDX": "0x5152535455565748",
|
||||
"RDI": "0xE0000009",
|
||||
"RSI": "0xE0000008"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
mov rdx, 0xe0000000
|
||||
|
||||
mov rax, 0x4142434445464748
|
||||
mov [rdx], rax
|
||||
mov rax, 0x5152535455565758
|
||||
mov [rdx + 8], rax
|
||||
|
||||
; Deliberately overlapping source and destination
|
||||
lea rdi, [rdx + 1]
|
||||
lea rsi, [rdx]
|
||||
|
||||
cld
|
||||
mov rcx, 8
|
||||
rep movsb ; rdi <- rsi
|
||||
|
||||
mov rdx, [rdx + 8]
|
||||
hlt
|
||||
Loaded 100 of 157 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user