mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 18:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
512643d3d6 | ||
|
|
b3a69af752 | ||
|
|
64c0dc47a9 | ||
|
|
923c323d6f | ||
|
|
ee47b5bbc9 | ||
|
|
e8cd655c84 | ||
|
|
9af52fb642 | ||
|
|
eaddd44d17 | ||
|
|
854e699589 | ||
|
|
20b00ecc9b | ||
|
|
40662f947f | ||
|
|
6e01934edc | ||
|
|
151fc5e97f | ||
|
|
5f431dc776 | ||
|
|
5ec4d3125a | ||
|
|
8e511e7db4 | ||
|
|
99b8046f03 | ||
|
|
d39dea1ae3 | ||
|
|
e94643d5ca | ||
|
|
1becbab0dc | ||
|
|
d49efb451e | ||
|
|
d2a56ebd8c | ||
|
|
0e11a9b7ac | ||
|
|
68c77d12d4 | ||
|
|
dcfbc2f20e | ||
|
|
0b29c99fed | ||
|
|
18556a9f75 | ||
|
|
02d93782ba | ||
|
|
62cfc26262 | ||
|
|
b01a6b94e7 | ||
|
|
713ebf1476 | ||
|
|
26e50efdb2 | ||
|
|
c8928999bf | ||
|
|
e1f378c6cf | ||
|
|
d10222329c | ||
|
|
176fa7ab1d | ||
|
|
ebb7137839 | ||
|
|
d7092a1231 | ||
|
|
55cbb0b340 | ||
|
|
db7fb56e9d | ||
|
|
2bed7440a8 | ||
|
|
0019bdecef | ||
|
|
51d355da30 | ||
|
|
28170fd723 | ||
|
|
44c65c35c8 | ||
|
|
d8f8daf48a | ||
|
|
b148cc6ca3 | ||
|
|
2a4c169fff | ||
|
|
c2c84e4bd8 | ||
|
|
c2f8b5b1ba | ||
|
|
3a33f554a0 | ||
|
|
11ce97655b | ||
|
|
dc866538d4 | ||
|
|
48ad9e9a87 | ||
|
|
c75778abeb | ||
|
|
fb2a59a67f | ||
|
|
f4c92756fc | ||
|
|
657c27556c | ||
|
|
42e68d8544 | ||
|
|
ae69c4d895 | ||
|
|
5a0db4d812 | ||
|
|
ddd241fe39 | ||
|
|
3dc7b8d90a | ||
|
|
0bccb1ece5 | ||
|
|
bf1e319d90 | ||
|
|
bc6ae7feb4 | ||
|
|
8d6a43d708 | ||
|
|
bd1bca2c3a | ||
|
|
9858ab7388 | ||
|
|
1f6b69573c | ||
|
|
1c8c5b77f1 | ||
|
|
e9bd037cf9 | ||
|
|
6f8353ab28 | ||
|
|
c25720429d | ||
|
|
276e9aded3 | ||
|
|
b88ac3359d | ||
|
|
264f3be8b4 | ||
|
|
4282f96d35 | ||
|
|
9cdd759fc1 | ||
|
|
403e8f8702 | ||
|
|
2e989e4262 | ||
|
|
5666a352d4 | ||
|
|
1aa8c6f996 | ||
|
|
9882f53613 | ||
|
|
8760c593ec | ||
|
|
ad695bdd59 | ||
|
|
adff4bb1d7 | ||
|
|
42c931cf22 | ||
|
|
5d44dea47c | ||
|
|
56c95e3b36 | ||
|
|
840f306a7d | ||
|
|
9def89d5f8 | ||
|
|
84c2f93dab | ||
|
|
b30733e2a7 | ||
|
|
da58e6a597 | ||
|
|
229e7c5b61 | ||
|
|
26685143be | ||
|
|
e54b9237c6 | ||
|
|
ac1b6d9482 | ||
|
|
bb6e98a6fc | ||
|
|
32c75f06b3 | ||
|
|
a5de2d1008 | ||
|
|
f841912c75 | ||
|
|
3b8c36882d | ||
|
|
fca4c7e6bf | ||
|
|
5ffc611d13 | ||
|
|
130f02647b | ||
|
|
981eea6ade | ||
|
|
3f788eb4a8 | ||
|
|
486dc974c4 | ||
|
|
3b1fbbc766 | ||
|
|
d01db8f293 | ||
|
|
fd09ded049 | ||
|
|
8e2b4a306d | ||
|
|
1d58f38aa5 | ||
|
|
5ff9a83b07 | ||
|
|
f5decb5f83 | ||
|
|
11fc49a0f8 | ||
|
|
48c03d747a | ||
|
|
643750817a | ||
|
|
a52dd71e44 | ||
|
|
8c02bd43df | ||
|
|
f635a12129 | ||
|
|
8191c4905b | ||
|
|
58a034b79d | ||
|
|
2d53867668 | ||
|
|
8a57fc5838 | ||
|
|
5481e6d79a | ||
|
|
c852a58ee3 | ||
|
|
8cfc016b3f | ||
|
|
8c94b782c6 | ||
|
|
1dce4919f2 | ||
|
|
cbda688e29 | ||
|
|
159ed07e68 | ||
|
|
a18b2d0e17 | ||
|
|
4c9adab58d | ||
|
|
4c9f1b105d | ||
|
|
2290353295 | ||
|
|
c16bf09310 | ||
|
|
90db9486ce | ||
|
|
b79faa6207 | ||
|
|
2293d3067a | ||
|
|
de431f113e | ||
|
|
34e265a801 | ||
|
|
a668492fb7 |
No files matched your search
@@ -3,10 +3,13 @@
|
||||
# Ignore all files in the External directory
|
||||
External/*
|
||||
|
||||
# SoftFloat-3e code doesn't belong to us
|
||||
# SoftFloat-3e code doesn't belong to us
|
||||
FEXCore/Source/Common/SoftFloat-3e/*
|
||||
Source/Common/cpp-optparse/*
|
||||
|
||||
# Files with human-indented tables for readability - don't mess with these
|
||||
FEXCore/Source/Interface/Core/X86Tables/*
|
||||
|
||||
# Inline headers with list-like content that can't be processed individually
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
|
||||
@@ -13,3 +13,6 @@
|
||||
|
||||
# Second reformat to find fixed point PR#3577
|
||||
905aa935f5ce344a48ef4d5edab3c31efa8d793e
|
||||
|
||||
# Reformat of CodeEmitter inl files
|
||||
8760c593ece92d7e9fa94c40da0368fd367c9cad
|
||||
@@ -250,7 +250,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
@@ -184,7 +184,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
@@ -97,7 +97,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
@@ -128,7 +128,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
@@ -137,7 +137,7 @@ jobs:
|
||||
|
||||
- name: Upload results InstCountCI
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}-instcountci
|
||||
|
||||
@@ -92,7 +92,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
@@ -126,7 +126,7 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
|
||||
+4
-10
@@ -8,7 +8,7 @@ option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" TRUE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
@@ -26,10 +26,9 @@ option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Enable use of VIXL simulator for emulation (only useful for CI testing)" FALSE)
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
|
||||
option(COMPILE_VIXL_DISASSEMBLER "Compiles the vixl disassembler in to vixl" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
@@ -266,12 +265,7 @@ set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-poin
|
||||
|
||||
include_directories(External/robin-map/include/)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
# Enable vixl disassembler if tests are enabled.
|
||||
set(COMPILE_VIXL_DISASSEMBLER TRUE)
|
||||
endif()
|
||||
|
||||
if (COMPILE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
@@ -299,7 +293,7 @@ add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
find_package(Catch2 QUIET)
|
||||
find_package(Catch2 3 QUIET)
|
||||
if (NOT Catch2_FOUND)
|
||||
add_subdirectory(External/Catch2/)
|
||||
|
||||
|
||||
+157
-183
@@ -11,6 +11,14 @@
|
||||
* FEX-Emu ALU operations usually have a 32-bit or 64-bit operating size encoded in the IR operation,
|
||||
* This allows FEX to use a single helper function which decodes to both handlers.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
namespace ARMEmitter {
|
||||
struct EmitterOps : Emitter {
|
||||
#endif
|
||||
|
||||
private:
|
||||
static bool IsADRRange(int64_t Imm) {
|
||||
return Imm >= -1048576 && Imm <= 1048575;
|
||||
@@ -28,26 +36,23 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
void adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void adr(ARMEmitter::Register rd, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADR });
|
||||
void adr(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADR});
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, BiDirectionalLabel *Label) {
|
||||
void adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adr(rd, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
adr(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
@@ -57,39 +62,34 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
void adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void adrp(ARMEmitter::Register rd, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::ADRP });
|
||||
void adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADRP});
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, BiDirectionalLabel *Label) {
|
||||
void adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adrp(rd, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
adrp(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, BackwardLabel const* Label) {
|
||||
void LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
|
||||
if (IsADRRange(Imm)) {
|
||||
// If the range is in ADR range then we can just use ADR.
|
||||
adr(rd, Label);
|
||||
}
|
||||
else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL)
|
||||
- (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
} else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
|
||||
// If the range is in the ADRP range then we can use ADRP.
|
||||
bool NeedsOffset = !IsADRPAligned(reinterpret_cast<uint64_t>(Label->Location));
|
||||
@@ -102,24 +102,22 @@ public:
|
||||
// Now even an add
|
||||
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset too large");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
void LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
Label->Insts.emplace_back(SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN });
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
|
||||
// Emit a register index and a nop. These will be backpatched.
|
||||
dc32(rd.Idx());
|
||||
nop();
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel *Label) {
|
||||
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
LongAddressGen(rd, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
LongAddressGen(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
@@ -176,11 +174,7 @@ public:
|
||||
// Logical immediate
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
and_(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -191,11 +185,7 @@ public:
|
||||
|
||||
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
ands(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -206,22 +196,14 @@ public:
|
||||
|
||||
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
orr(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm,
|
||||
RegSizeInBits(s),
|
||||
&n,
|
||||
&imms,
|
||||
&immr);
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
eor(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -355,8 +337,8 @@ public:
|
||||
const auto lsb_p_width = lsb + width;
|
||||
|
||||
LOGMAN_THROW_A_FMT(width >= 1, "bfxil needs width >= 1");
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "bfxil lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "bfxil lsb + width ({}) must be <= {}. lsb={}, width={}", lsb_p_width, reg_size_bits,
|
||||
lsb, width);
|
||||
|
||||
bfm(s, rd, rn, lsb, lsb_p_width - 1);
|
||||
}
|
||||
@@ -375,188 +357,142 @@ public:
|
||||
|
||||
// Data processing - 2 source
|
||||
void udiv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0000'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void sdiv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0000'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
|
||||
void lslv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0010'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void lsrv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0010'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void asrv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0010'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void rorv(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0010'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0010'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void crc32b(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0100'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32h(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0100'01U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32w(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0100'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32cb(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0101'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32ch(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0101'01U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32cw(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0101'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void smax(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0110'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umax(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0110'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void smin(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'10U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0110'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umin(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0110'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void subp(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0000'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void irg(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0001'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0001'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void gmi(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0001'01U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0001'01U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void pacga(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0011'00U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0011'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32x(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0100'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0100'11U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void crc32cx(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0101'11U << 10);
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) | (0b0101'11U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
void subps(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b011'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
constexpr uint32_t Op = (0b011'1010'110U << 21) | (0b0000'00U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i64Bit, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Data processing - 1 source
|
||||
void rbit(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'00U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void rev16(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'01U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'01U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void rev(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'10U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'10U << 10);
|
||||
DataProcessing_1Source(Op, ARMEmitter::Size::i32Bit, rd, rn);
|
||||
}
|
||||
void rev32(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'10U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'10U << 10);
|
||||
DataProcessing_1Source(Op, ARMEmitter::Size::i64Bit, rd, rn);
|
||||
}
|
||||
void clz(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'00U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0001'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cls(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'01U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0001'01U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void rev(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'11U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'11U << 10);
|
||||
DataProcessing_1Source(Op, ARMEmitter::Size::i64Bit, rd, rn);
|
||||
}
|
||||
void rev(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0000'10U << 10) |
|
||||
(s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0000'10U << 10) | (s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void ctz(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'10U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0001'10U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cnt(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'11U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0001'11U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void abs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0010'00U << 10);
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) | (0b0'0000U << 16) | (0b0010'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
@@ -573,27 +509,33 @@ public:
|
||||
orr(ARMEmitter::Size::i32Bit, rd.R(), ARMEmitter::Reg::zr, rn.R(), ARMEmitter::ShiftType::LSL, 0);
|
||||
}
|
||||
|
||||
void mvn(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void mvn(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
orn(s, rd, ARMEmitter::Reg::zr, rn, Shift, amt);
|
||||
}
|
||||
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b000'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b110'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void bic(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void bic(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b000'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void bics(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void bics(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b110'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b010'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
@@ -601,30 +543,36 @@ public:
|
||||
ands(s, Reg::zr, rn, rm, shift, amt);
|
||||
}
|
||||
|
||||
void orn(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void orn(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b010'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b100'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void eon(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void eon(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b100'1010'001U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
|
||||
// AddSub - shifted register
|
||||
void add(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void add(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
add(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void adds(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void adds(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void sub(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void neg(ARMEmitter::XRegister rd, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
@@ -633,23 +581,27 @@ public:
|
||||
void cmp(ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void subs(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void subs(ARMEmitter::XRegister rd, ARMEmitter::XRegister rn, ARMEmitter::XRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void negs(ARMEmitter::XRegister rd, ARMEmitter::XRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(rd, ARMEmitter::XReg::zr, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void add(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void add(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
add(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void adds(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void adds(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, ARMEmitter::WReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void sub(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void neg(ARMEmitter::WRegister rd, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
@@ -658,65 +610,78 @@ public:
|
||||
void cmp(ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::rsp, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void subs(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void subs(ARMEmitter::WRegister rd, ARMEmitter::WRegister rn, ARMEmitter::WRegister rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void negs(ARMEmitter::WRegister rd, ARMEmitter::WRegister rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
subs(rd, ARMEmitter::WReg::zr, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void add(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
void add(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b000'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void adds(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
void adds(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b010'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void cmn(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void cmn(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
adds(s, ARMEmitter::Reg::zr, rn, rm, Shift, amt);
|
||||
}
|
||||
void sub(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
void sub(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b100'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void neg(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void neg(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
sub(s, rd, ARMEmitter::Reg::zr, rm, Shift, amt);
|
||||
}
|
||||
void cmp(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void cmp(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
subs(s, ARMEmitter::Reg::zr, rn, rm, Shift, amt);
|
||||
}
|
||||
|
||||
void subs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
void subs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift != ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b110'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void negs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
void negs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift = ARMEmitter::ShiftType::LSL,
|
||||
uint32_t amt = 0) {
|
||||
subs(s, rd, ARMEmitter::Reg::zr, rm, Shift, amt);
|
||||
}
|
||||
|
||||
// AddSub - extended register
|
||||
void add(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift <= 4, "Shift amount is too large");
|
||||
void add(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option,
|
||||
uint32_t Shift = 0) {
|
||||
LOGMAN_THROW_A_FMT(Shift <= 4, "Shift amount is too large");
|
||||
constexpr uint32_t Op = 0b000'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void adds(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
void adds(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option,
|
||||
uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b010'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void cmn(ARMEmitter::Size s, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
adds(s, ARMEmitter::Reg::zr, rn, rm, Option, Shift);
|
||||
}
|
||||
void sub(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
void sub(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option,
|
||||
uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b100'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void subs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
void subs(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option,
|
||||
uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b110'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
@@ -751,8 +716,8 @@ public:
|
||||
|
||||
// Rotate right into flags
|
||||
void rmif(XRegister rn, uint32_t shift, uint32_t mask) {
|
||||
LOGMAN_THROW_AA_FMT(shift <= 63, "Shift must be within 0-63. Shift: {}", shift);
|
||||
LOGMAN_THROW_AA_FMT(mask <= 15, "Mask must be within 0-15. Mask: {}", mask);
|
||||
LOGMAN_THROW_A_FMT(shift <= 63, "Shift must be within 0-63. Shift: {}", shift);
|
||||
LOGMAN_THROW_A_FMT(mask <= 15, "Mask must be within 0-15. Mask: {}", mask);
|
||||
|
||||
uint32_t Op = 0b1011'1010'0000'0000'0000'0100'0000'0000;
|
||||
Op |= rn.Idx() << 5;
|
||||
@@ -816,7 +781,8 @@ public:
|
||||
}
|
||||
void cset(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 0, 0b01, s, rd, ARMEmitter::Reg::zr, ARMEmitter::Reg::zr, static_cast<ARMEmitter::Condition>(FEXCore::ToUnderlying(Cond) ^ FEXCore::ToUnderlying(ARMEmitter::Condition::CC_NE)));
|
||||
ConditionalCompare(Op, 0, 0b01, s, rd, ARMEmitter::Reg::zr, ARMEmitter::Reg::zr,
|
||||
static_cast<ARMEmitter::Condition>(FEXCore::ToUnderlying(Cond) ^ FEXCore::ToUnderlying(ARMEmitter::Condition::CC_NE)));
|
||||
}
|
||||
void csinc(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
@@ -898,8 +864,7 @@ public:
|
||||
private:
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_AA_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV,
|
||||
"Cannot invert CC_AL or CC_NV");
|
||||
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
@@ -950,7 +915,7 @@ private:
|
||||
LSL12 = true;
|
||||
Imm >>= 12;
|
||||
}
|
||||
LOGMAN_THROW_AA_FMT(TooLarge == false, "Imm amount too large: 0x{:x}", Imm);
|
||||
LOGMAN_THROW_A_FMT(TooLarge == false, "Imm amount too large: 0x{:x}", Imm);
|
||||
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
@@ -995,7 +960,8 @@ private:
|
||||
}
|
||||
|
||||
// Logical immediate
|
||||
void DataProcessing_Logical_Imm(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
void DataProcessing_Logical_Imm(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n,
|
||||
uint32_t immr, uint32_t imms) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1014,9 +980,8 @@ private:
|
||||
[[maybe_unused]] const auto lsb_p_width = lsb + width;
|
||||
const auto reg_size_bits = RegSizeInBits(s);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
LOGMAN_THROW_AA_FMT(width >= 1, "xbfiz width must be >= 1");
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}", lsb_p_width, reg_size_bits, lsb, width);
|
||||
LOGMAN_THROW_A_FMT(width >= 1, "xbfiz width must be >= 1");
|
||||
|
||||
const auto immr = (reg_size_bits - lsb) & (reg_size_bits - 1);
|
||||
const auto imms = width - 1;
|
||||
@@ -1028,12 +993,13 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
void DataProcessing_Extract(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, uint32_t Imm) {
|
||||
void DataProcessing_Extract(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm,
|
||||
uint32_t Imm) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
// Current ARMv8 spec hardcodes SF == N for this class of instructions.
|
||||
// Anythign else is undefined behaviour.
|
||||
const uint32_t N = s == ARMEmitter::Size::i64Bit ? (1U << 22) : 0;
|
||||
const uint32_t N = s == ARMEmitter::Size::i64Bit ? (1U << 22) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
@@ -1076,10 +1042,11 @@ private:
|
||||
}
|
||||
|
||||
// AddSub - shifted register
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_AA_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn,
|
||||
ARMEmitter::Register rm, ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_A_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
if (s == ARMEmitter::Size::i32Bit) {
|
||||
LOGMAN_THROW_AA_FMT(amt < 32, "Shift amount for 32-bit must be below 32");
|
||||
LOGMAN_THROW_A_FMT(amt < 32, "Shift amount for 32-bit must be below 32");
|
||||
}
|
||||
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
@@ -1097,7 +1064,8 @@ private:
|
||||
}
|
||||
|
||||
// AddSub - extended register
|
||||
void DataProcessing_Extended_Reg(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift) {
|
||||
void DataProcessing_Extended_Reg(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn,
|
||||
ARMEmitter::Register rm, ARMEmitter::ExtendedType Option, uint32_t Shift) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1113,7 +1081,8 @@ private:
|
||||
}
|
||||
// Conditional compare - register
|
||||
template<typename T>
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, uint32_t o3, ARMEmitter::Size s, ARMEmitter::Register rn, T rm, ARMEmitter::StatusFlags flags, ARMEmitter::Condition Cond) {
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, uint32_t o3, ARMEmitter::Size s, ARMEmitter::Register rn, T rm,
|
||||
ARMEmitter::StatusFlags flags, ARMEmitter::Condition Cond) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1131,7 +1100,8 @@ private:
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, T rm, ARMEmitter::Condition Cond) {
|
||||
void ConditionalCompare(uint32_t Op, uint32_t o1, uint32_t o2, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, T rm,
|
||||
ARMEmitter::Condition Cond) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1148,7 +1118,8 @@ private:
|
||||
}
|
||||
|
||||
// Data-processing - 3 source
|
||||
void DataProcessing_3Source(uint32_t Op, uint32_t Op0, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, ARMEmitter::Register rm, ARMEmitter::Register ra) {
|
||||
void DataProcessing_3Source(uint32_t Op, uint32_t Op0, ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn,
|
||||
ARMEmitter::Register rm, ARMEmitter::Register ra) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
@@ -1170,4 +1141,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
}; // struct LoadstoreEmitterOps
|
||||
} // namespace ARMEmitter
|
||||
#endif
|
||||
+881
-1012
File diff suppressed because it is too large.
Load diff
@@ -3,339 +3,325 @@
|
||||
*
|
||||
* Most of these instructions will use `BackwardLabel`, `ForwardLabel`, or `BiDirectionLabel` to determine where a branch targets.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
namespace ARMEmitter {
|
||||
struct EmitterOps : Emitter {
|
||||
#endif
|
||||
|
||||
public:
|
||||
// Branches, Exception Generating and System instructions
|
||||
public:
|
||||
// Conditional branch immediate
|
||||
///< Branch conditional
|
||||
void b(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm);
|
||||
public:
|
||||
// Conditional branch immediate
|
||||
///< Branch conditional
|
||||
void b(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm);
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
}
|
||||
|
||||
void b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(Cond, &Label->Backward);
|
||||
} else {
|
||||
b(Cond, &Label->Forward);
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
}
|
||||
|
||||
///< Branch consistent conditional
|
||||
void bc(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm);
|
||||
}
|
||||
void bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bc(Cond, &Label->Backward);
|
||||
} else {
|
||||
bc(Cond, &Label->Forward);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void b(ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
}
|
||||
|
||||
// Unconditional branch register
|
||||
void br(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 | 0b0'000 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void blr(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 | 0b0'001 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void ret(ARMEmitter::Register rn = ARMEmitter::Reg::r30) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 | 0b0'010 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
|
||||
// Unconditional branch immediate
|
||||
void b(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
void b(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
void b(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void b(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(&Label->Backward);
|
||||
} else {
|
||||
b(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void b(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(Cond, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
b(Cond, &Label->Forward);
|
||||
}
|
||||
void bl(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
|
||||
void bl(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
void bl(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void bl(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bl(&Label->Backward);
|
||||
} else {
|
||||
bl(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
///< Branch consistent conditional
|
||||
void bc(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm);
|
||||
// Compare and branch
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbz(s, rt, &Label->Forward);
|
||||
}
|
||||
void bc(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbnz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbnz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void bc(ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
// Test and branch immediate
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bc(Cond, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
bc(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
// Unconditional branch register
|
||||
void br(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'000 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void blr(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'001 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void ret(ARMEmitter::Register rn = ARMEmitter::Reg::r30) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'010 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
0b0000'00 << 10 | // op3
|
||||
0b0'0000; // op4
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
// Unconditional branch immediate
|
||||
void b(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
void b(BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void b(LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void b(BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(&Label->Backward);
|
||||
}
|
||||
else {
|
||||
b(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void bl(uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
|
||||
void bl(BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void bl(LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
}
|
||||
|
||||
void bl(BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bl(&Label->Backward);
|
||||
}
|
||||
else {
|
||||
bl(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
// Compare and branch
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbz(s, rt, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
cbz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbnz(s, rt, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
cbnz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
// Test and branch immediate
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbz(rt, Bit, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
tbz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
}
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbnz(rt, Bit, &Label->Backward);
|
||||
}
|
||||
else {
|
||||
tbnz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbnz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbnz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Conditional branch immediate
|
||||
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
// Conditional branch immediate
|
||||
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= Op1 << 24;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Op0 << 4;
|
||||
Instr |= FEXCore::ToUnderlying(Cond);
|
||||
Instr |= Op1 << 24;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Op0 << 4;
|
||||
Instr |= FEXCore::ToUnderlying(Cond);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Unconditional branch register
|
||||
void UnconditionalBranch(uint32_t Op, ARMEmitter::Register rn) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Encode_rn(rn);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Unconditional branch register
|
||||
void UnconditionalBranch(uint32_t Op, ARMEmitter::Register rn) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Encode_rn(rn);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Unconditional branch - immediate
|
||||
void UnconditionalBranch(uint32_t Op, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Imm & 0x3FF'FFFF;
|
||||
dc32(Instr);
|
||||
}
|
||||
// Unconditional branch - immediate
|
||||
void UnconditionalBranch(uint32_t Op, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Imm & 0x3FF'FFFF;
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Compare and branch
|
||||
void CompareAndBranch(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
// Compare and branch
|
||||
void CompareAndBranch(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
Instr |= SF;
|
||||
Instr |= (Imm & 0x7'FFFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Test and branch - immediate
|
||||
void TestAndBranch(uint32_t Op, ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
// Test and branch - immediate
|
||||
void TestAndBranch(uint32_t Op, ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= (Bit >> 5) << 31;
|
||||
Instr |= (Bit & 0b1'1111) << 19;
|
||||
Instr |= (Imm & 0x3FFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
Instr |= (Bit >> 5) << 31;
|
||||
Instr |= (Bit & 0b1'1111) << 19;
|
||||
Instr |= (Imm & 0x3FFF) << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
}; // struct LoadstoreEmitterOps
|
||||
} // namespace ARMEmitter
|
||||
#endif
|
||||
@@ -341,94 +341,88 @@ public:
|
||||
};
|
||||
|
||||
template<uint32_t op0, uint32_t op1, uint32_t CRn, uint32_t CRm, uint32_t op2>
|
||||
constexpr uint32_t GenSystemReg() {
|
||||
return op0 << 19 | op1 << 16 | CRn << 12 | CRm << 8 | op2 << 5;
|
||||
};
|
||||
inline constexpr uint32_t GenSystemReg = op0 << 19 | op1 << 16 | CRn << 12 | CRm << 8 | op2 << 5;
|
||||
|
||||
// This `SystemRegister` enum is used for the mrs/msr instructions.
|
||||
enum class SystemRegister : uint32_t {
|
||||
CTR_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b001>(),
|
||||
DCZID_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b111>(),
|
||||
TPIDR_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b010>(),
|
||||
RNDR = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b000>(),
|
||||
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>(),
|
||||
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>(),
|
||||
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>(),
|
||||
TPIDRRO_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b011>(),
|
||||
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>(),
|
||||
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>(),
|
||||
CTR_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b001>,
|
||||
DCZID_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b111>,
|
||||
TPIDR_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b010>,
|
||||
RNDR = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b000>,
|
||||
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>,
|
||||
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>,
|
||||
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>,
|
||||
TPIDRRO_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b011>,
|
||||
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>,
|
||||
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>,
|
||||
};
|
||||
|
||||
template<uint32_t op1, uint32_t CRm, uint32_t op2>
|
||||
constexpr uint32_t GenDCReg() {
|
||||
return op1 << 16 | CRm << 8 | op2 << 5;
|
||||
};
|
||||
inline constexpr uint32_t GenDCReg = op1 << 16 | CRm << 8 | op2 << 5;
|
||||
|
||||
// This `DataCacheOperation` enum is used for the dc instruction.
|
||||
enum class DataCacheOperation : uint32_t {
|
||||
IVAC = GenDCReg<0b000, 0b0110, 0b001>(),
|
||||
ISW = GenDCReg<0b000, 0b0110, 0b010>(),
|
||||
CSW = GenDCReg<0b000, 0b1010, 0b010>(),
|
||||
CISW = GenDCReg<0b000, 0b1110, 0b010>(),
|
||||
ZVA = GenDCReg<0b011, 0b0100, 0b001>(),
|
||||
CVAC = GenDCReg<0b011, 0b1010, 0b001>(),
|
||||
CVAU = GenDCReg<0b011, 0b1011, 0b001>(),
|
||||
CIVAC = GenDCReg<0b011, 0b1110, 0b001>(),
|
||||
IVAC = GenDCReg<0b000, 0b0110, 0b001>,
|
||||
ISW = GenDCReg<0b000, 0b0110, 0b010>,
|
||||
CSW = GenDCReg<0b000, 0b1010, 0b010>,
|
||||
CISW = GenDCReg<0b000, 0b1110, 0b010>,
|
||||
ZVA = GenDCReg<0b011, 0b0100, 0b001>,
|
||||
CVAC = GenDCReg<0b011, 0b1010, 0b001>,
|
||||
CVAU = GenDCReg<0b011, 0b1011, 0b001>,
|
||||
CIVAC = GenDCReg<0b011, 0b1110, 0b001>,
|
||||
|
||||
// MTE2
|
||||
IGVAC = GenDCReg<0b000, 0b0110, 0b011>(),
|
||||
IGSW = GenDCReg<0b000, 0b0110, 0b100>(),
|
||||
IGDVAC = GenDCReg<0b000, 0b0110, 0b101>(),
|
||||
IGDSW = GenDCReg<0b000, 0b0110, 0b110>(),
|
||||
CGSW = GenDCReg<0b000, 0b1010, 0b100>(),
|
||||
CGDSW = GenDCReg<0b000, 0b1010, 0b110>(),
|
||||
CIGSW = GenDCReg<0b000, 0b1110, 0b100>(),
|
||||
CIGDSW = GenDCReg<0b000, 0b1110, 0b110>(),
|
||||
IGVAC = GenDCReg<0b000, 0b0110, 0b011>,
|
||||
IGSW = GenDCReg<0b000, 0b0110, 0b100>,
|
||||
IGDVAC = GenDCReg<0b000, 0b0110, 0b101>,
|
||||
IGDSW = GenDCReg<0b000, 0b0110, 0b110>,
|
||||
CGSW = GenDCReg<0b000, 0b1010, 0b100>,
|
||||
CGDSW = GenDCReg<0b000, 0b1010, 0b110>,
|
||||
CIGSW = GenDCReg<0b000, 0b1110, 0b100>,
|
||||
CIGDSW = GenDCReg<0b000, 0b1110, 0b110>,
|
||||
|
||||
// MTE
|
||||
GVA = GenDCReg<0b011, 0b0100, 0b011>(),
|
||||
GZVA = GenDCReg<0b011, 0b0100, 0b100>(),
|
||||
CGVAC = GenDCReg<0b011, 0b1010, 0b011>(),
|
||||
CGDVAC = GenDCReg<0b011, 0b1010, 0b101>(),
|
||||
CGVAP = GenDCReg<0b011, 0b1100, 0b011>(),
|
||||
CGDVAP = GenDCReg<0b011, 0b1100, 0b101>(),
|
||||
CGVADP = GenDCReg<0b011, 0b1101, 0b011>(),
|
||||
CGDVADP = GenDCReg<0b011, 0b1101, 0b101>(),
|
||||
CIGVAC = GenDCReg<0b011, 0b1110, 0b011>(),
|
||||
CIGDVAC = GenDCReg<0b011, 0b1110, 0b101>(),
|
||||
GVA = GenDCReg<0b011, 0b0100, 0b011>,
|
||||
GZVA = GenDCReg<0b011, 0b0100, 0b100>,
|
||||
CGVAC = GenDCReg<0b011, 0b1010, 0b011>,
|
||||
CGDVAC = GenDCReg<0b011, 0b1010, 0b101>,
|
||||
CGVAP = GenDCReg<0b011, 0b1100, 0b011>,
|
||||
CGDVAP = GenDCReg<0b011, 0b1100, 0b101>,
|
||||
CGVADP = GenDCReg<0b011, 0b1101, 0b011>,
|
||||
CGDVADP = GenDCReg<0b011, 0b1101, 0b101>,
|
||||
CIGVAC = GenDCReg<0b011, 0b1110, 0b011>,
|
||||
CIGDVAC = GenDCReg<0b011, 0b1110, 0b101>,
|
||||
|
||||
// DPB
|
||||
CVAP = GenDCReg<0b011, 0b1100, 0b001>(),
|
||||
CVAP = GenDCReg<0b011, 0b1100, 0b001>,
|
||||
|
||||
// DPB2
|
||||
CVADP = GenDCReg<0b011, 0b1101, 0b001>(),
|
||||
CVADP = GenDCReg<0b011, 0b1101, 0b001>,
|
||||
};
|
||||
|
||||
template<uint32_t CRm, uint32_t op2>
|
||||
constexpr uint32_t GenHintBarrierReg() {
|
||||
return CRm << 8 | op2 << 5;
|
||||
}
|
||||
inline constexpr uint32_t GenHintBarrierReg = CRm << 8 | op2 << 5;
|
||||
|
||||
// This `HintRegister` enum is used for the hint instruction.
|
||||
enum class HintRegister : uint32_t {
|
||||
NOP = GenHintBarrierReg<0b0000, 0b000>(),
|
||||
YIELD = GenHintBarrierReg<0b0000, 0b001>(),
|
||||
WFE = GenHintBarrierReg<0b0000, 0b010>(),
|
||||
WFI = GenHintBarrierReg<0b0000, 0b011>(),
|
||||
SEV = GenHintBarrierReg<0b0000, 0b100>(),
|
||||
SEVL = GenHintBarrierReg<0b0000, 0b101>(),
|
||||
DGH = GenHintBarrierReg<0b0000, 0b110>(),
|
||||
CSDB = GenHintBarrierReg<0b0010, 0b100>(),
|
||||
NOP = GenHintBarrierReg<0b0000, 0b000>,
|
||||
YIELD = GenHintBarrierReg<0b0000, 0b001>,
|
||||
WFE = GenHintBarrierReg<0b0000, 0b010>,
|
||||
WFI = GenHintBarrierReg<0b0000, 0b011>,
|
||||
SEV = GenHintBarrierReg<0b0000, 0b100>,
|
||||
SEVL = GenHintBarrierReg<0b0000, 0b101>,
|
||||
DGH = GenHintBarrierReg<0b0000, 0b110>,
|
||||
CSDB = GenHintBarrierReg<0b0010, 0b100>,
|
||||
};
|
||||
|
||||
// This `BarrierRegister` enum is used for the various barrier instructions.
|
||||
enum class BarrierRegister : uint32_t {
|
||||
CLREX = GenHintBarrierReg<0b0000, 0b010>(),
|
||||
TCOMMIT = GenHintBarrierReg<0b0000, 0b011>(),
|
||||
DSB = GenHintBarrierReg<0b0000, 0b100>(),
|
||||
DMB = GenHintBarrierReg<0b0000, 0b101>(),
|
||||
ISB = GenHintBarrierReg<0b0000, 0b110>(),
|
||||
SB = GenHintBarrierReg<0b0000, 0b111>(),
|
||||
CLREX = GenHintBarrierReg<0b0000, 0b010>,
|
||||
TCOMMIT = GenHintBarrierReg<0b0000, 0b011>,
|
||||
DSB = GenHintBarrierReg<0b0000, 0b100>,
|
||||
DMB = GenHintBarrierReg<0b0000, 0b101>,
|
||||
ISB = GenHintBarrierReg<0b0000, 0b110>,
|
||||
SB = GenHintBarrierReg<0b0000, 0b111>,
|
||||
};
|
||||
|
||||
// This `BarrierScope` enum is used for the dsb/dmb instructions.
|
||||
@@ -513,7 +507,7 @@ enum class SVEFMaxMinImm : uint32_t {
|
||||
_1_0,
|
||||
};
|
||||
|
||||
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
/* This `BackwardLabel` struct is used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `below` an instruction that uses it.
|
||||
* Which means that a branch would jump backwards.
|
||||
*/
|
||||
@@ -521,13 +515,11 @@ struct BackwardLabel {
|
||||
uint8_t* Location {};
|
||||
};
|
||||
|
||||
/* This `SingleUseForwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
/* This `ForwardLabel` struct is used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `above` an instruction that uses it.
|
||||
* Which means that a branch would jump forwards.
|
||||
*
|
||||
* The `ForwardLabel` struct can be bound to multiple instructions, so it needs a vector for each bind instruction type.
|
||||
*/
|
||||
struct SingleUseForwardLabel {
|
||||
struct ForwardLabel {
|
||||
enum class InstType {
|
||||
UNKNOWN,
|
||||
ADR,
|
||||
@@ -538,12 +530,16 @@ struct SingleUseForwardLabel {
|
||||
RELATIVE_LOAD,
|
||||
LONG_ADDRESS_GEN,
|
||||
};
|
||||
uint8_t* Location {};
|
||||
InstType Type = InstType::UNKNOWN;
|
||||
};
|
||||
|
||||
struct ForwardLabel {
|
||||
fextl::vector<SingleUseForwardLabel> Insts {};
|
||||
struct Reference {
|
||||
uint8_t* Location {};
|
||||
InstType Type = InstType::UNKNOWN;
|
||||
};
|
||||
|
||||
// The first element is stored separately to avoid allocations for simple cases
|
||||
Reference FirstInst;
|
||||
|
||||
fextl::vector<Reference> Insts;
|
||||
};
|
||||
|
||||
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
@@ -555,14 +551,12 @@ struct BiDirectionalLabel {
|
||||
ForwardLabel Forward;
|
||||
};
|
||||
|
||||
static inline void AddLocationToLabel(SingleUseForwardLabel* Label, SingleUseForwardLabel&& Location) {
|
||||
LOGMAN_THROW_A_FMT(Label->Type == SingleUseForwardLabel::InstType::UNKNOWN, "Trying to bind a SingleUseForwardLabel to multiple "
|
||||
"locations. Use ForwardLabel instead.");
|
||||
*Label = std::move(Location);
|
||||
}
|
||||
|
||||
static inline void AddLocationToLabel(ForwardLabel* Label, SingleUseForwardLabel&& Location) {
|
||||
Label->Insts.emplace_back(std::move(Location));
|
||||
static inline void AddLocationToLabel(ForwardLabel* Label, ForwardLabel::Reference&& Location) {
|
||||
if (Label->FirstInst.Location == nullptr) {
|
||||
Label->FirstInst = Location;
|
||||
} else {
|
||||
Label->Insts.push_back(Location);
|
||||
}
|
||||
}
|
||||
|
||||
// Some FCMA ASIMD instructions support a rotation argument.
|
||||
@@ -631,15 +625,15 @@ public:
|
||||
// Bind a backward label to an address.
|
||||
// Address that is bound is the current emitter location.
|
||||
void Bind(BackwardLabel* Label) {
|
||||
LOGMAN_THROW_AA_FMT(Label->Location == nullptr, "Trying to bind a label twice");
|
||||
LOGMAN_THROW_A_FMT(Label->Location == nullptr, "Trying to bind a label twice");
|
||||
Label->Location = GetCursorAddress<uint8_t*>();
|
||||
}
|
||||
|
||||
void Bind(const SingleUseForwardLabel* Label) {
|
||||
void Bind(const ForwardLabel::Reference* Label) {
|
||||
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
|
||||
// Patch up the instructions
|
||||
switch (Label->Type) {
|
||||
case SingleUseForwardLabel::InstType::ADR: {
|
||||
case ForwardLabel::InstType::ADR: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
@@ -651,7 +645,7 @@ public:
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case SingleUseForwardLabel::InstType::ADRP: {
|
||||
case ForwardLabel::InstType::ADRP: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
@@ -665,7 +659,7 @@ public:
|
||||
break;
|
||||
}
|
||||
|
||||
case SingleUseForwardLabel::InstType::B: {
|
||||
case ForwardLabel::InstType::B: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
@@ -679,7 +673,7 @@ public:
|
||||
break;
|
||||
}
|
||||
|
||||
case SingleUseForwardLabel::InstType::TEST_BRANCH: {
|
||||
case ForwardLabel::InstType::TEST_BRANCH: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
@@ -692,8 +686,8 @@ public:
|
||||
|
||||
break;
|
||||
}
|
||||
case SingleUseForwardLabel::InstType::BC:
|
||||
case SingleUseForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
case ForwardLabel::InstType::BC:
|
||||
case ForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
@@ -705,7 +699,7 @@ public:
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
case SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN: {
|
||||
case ForwardLabel::InstType::LONG_ADDRESS_GEN: {
|
||||
uint32_t* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
@@ -750,10 +744,9 @@ public:
|
||||
// Bind a forward label to a location.
|
||||
// This walks all the instructions in the label's vector.
|
||||
// Then backpatching all instructions that have used the label.
|
||||
template<bool WarnAboutEmpty = false>
|
||||
void Bind(ForwardLabel* Label) {
|
||||
if constexpr (WarnAboutEmpty) {
|
||||
LOGMAN_THROW_A_FMT(Label->Insts.empty() == false, "Binding forward label that didn't have any instructions using it");
|
||||
if (Label->FirstInst.Location) {
|
||||
Bind(&Label->FirstInst);
|
||||
}
|
||||
for (auto& Inst : Label->Insts) {
|
||||
Bind(&Inst);
|
||||
@@ -766,12 +759,18 @@ public:
|
||||
if (!Label->Backward.Location) {
|
||||
Bind(&Label->Backward);
|
||||
}
|
||||
Bind<false>(&Label->Forward);
|
||||
Bind(&Label->Forward);
|
||||
}
|
||||
|
||||
#include <CodeEmitter/VixlUtils.inl>
|
||||
|
||||
public:
|
||||
|
||||
// This symbol is used to allow external tooling (IDEs, clang-format, ...) to process the included files individually:
|
||||
// If defined, the files will inject member functions into this class.
|
||||
// If not, the files will wrap the member functions in a class so that tooling will process them properly.
|
||||
#define INCLUDED_BY_EMITTER
|
||||
|
||||
// TODO: Implement SME when it matters.
|
||||
#include <CodeEmitter/ALUOps.inl>
|
||||
#include <CodeEmitter/BranchOps.inl>
|
||||
@@ -781,7 +780,9 @@ public:
|
||||
#include <CodeEmitter/ASIMDOps.inl>
|
||||
#include <CodeEmitter/SVEOps.inl>
|
||||
|
||||
private:
|
||||
#undef INCLUDED_BY_EMITTER
|
||||
|
||||
protected:
|
||||
template<typename T>
|
||||
uint32_t Encode_ra(T Reg) const {
|
||||
return Reg.Idx() << 10;
|
||||
@@ -793,7 +794,6 @@ private:
|
||||
uint32_t Encode_rt2(T Reg) const {
|
||||
return Reg.Idx() << 10;
|
||||
}
|
||||
template<>
|
||||
uint32_t Encode_rt2(uint32_t Reg) const {
|
||||
return Reg << 10;
|
||||
}
|
||||
@@ -829,7 +829,6 @@ private:
|
||||
uint32_t Encode_rt(T Reg) const {
|
||||
return Reg.Idx();
|
||||
}
|
||||
template<>
|
||||
uint32_t Encode_rt(Prefetch Reg) const {
|
||||
return FEXCore::ToUnderlying(Reg);
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
+449
-650
File diff suppressed because it is too large.
Load diff
@@ -16,17 +16,25 @@
|
||||
* Exceptions to this rule will have asserts in the emitter implementation when misused.
|
||||
*
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
namespace ARMEmitter {
|
||||
struct EmitterOps : Emitter {
|
||||
#endif
|
||||
|
||||
public:
|
||||
// Advanced SIMD scalar copy
|
||||
// Advanced SIMD scalar copy
|
||||
void dup(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Index) {
|
||||
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'01 << 10;
|
||||
|
||||
const uint32_t SizeImm = FEXCore::ToUnderlying(size);
|
||||
const uint32_t IndexShift = SizeImm + 1;
|
||||
const uint32_t ElementSize = 1U << SizeImm;
|
||||
const uint32_t MaxIndex = 128U / (ElementSize * 8);
|
||||
[[maybe_unused]] const uint32_t MaxIndex = 128U / (ElementSize * 8);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Index < MaxIndex, "Index too large. Index={}, Max Index: {}", Index, MaxIndex);
|
||||
LOGMAN_THROW_A_FMT(Index < MaxIndex, "Index too large. Index={}, Max Index: {}", Index, MaxIndex);
|
||||
|
||||
const uint32_t imm5 = (Index << IndexShift) | ElementSize;
|
||||
|
||||
@@ -37,7 +45,7 @@ public:
|
||||
dup(size, rd, rn, Index);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar three same FP16
|
||||
// Advanced SIMD scalar three same FP16
|
||||
void fmulx(HRegister rd, HRegister rn, HRegister rm) {
|
||||
ASIMDScalarThreeSameFP16(0, 0, 0b011, rm, rn, rd);
|
||||
}
|
||||
@@ -66,7 +74,7 @@ public:
|
||||
ASIMDScalarThreeSameFP16(1, 1, 0b101, rm, rn, rd);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar two-register miscellaneous FP16
|
||||
// Advanced SIMD scalar two-register miscellaneous FP16
|
||||
void fcvtns(HRegister rd, HRegister rn) {
|
||||
ASIMDScalarTwoRegMiscFP16(0, 0, 0b11010, rn, rd);
|
||||
}
|
||||
@@ -128,9 +136,9 @@ public:
|
||||
ASIMDScalarTwoRegMiscFP16(1, 1, 0b11101, rn, rd);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar three same extra
|
||||
// XXX:
|
||||
// Advanced SIMD scalar two-register miscellaneous
|
||||
// Advanced SIMD scalar three same extra
|
||||
// XXX:
|
||||
// Advanced SIMD scalar two-register miscellaneous
|
||||
void suqadd(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b00011, rd, rn);
|
||||
}
|
||||
@@ -140,67 +148,55 @@ public:
|
||||
|
||||
///< Comparison against 0.0
|
||||
void cmgt(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b01000, rd, rn);
|
||||
}
|
||||
///< Comparison against 0.0
|
||||
void cmeq(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b01001, rd, rn);
|
||||
}
|
||||
|
||||
///< Comparison against 0.0
|
||||
void cmlt(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b01010, rd, rn);
|
||||
}
|
||||
void abs(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b01011, rd, rn);
|
||||
}
|
||||
///< size is destination size.
|
||||
void sqxtn(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
ASIMDScalar2RegMisc(0, 0, size, 0b10100, rd, rn);
|
||||
}
|
||||
|
||||
void fcvtns(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 0, ConvertedSize, 0b11010, rd, rn);
|
||||
}
|
||||
void fcvtms(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 0, ConvertedSize, 0b11011, rd, rn);
|
||||
}
|
||||
void fcvtas(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 0, ConvertedSize, 0b11100, rd, rn);
|
||||
}
|
||||
void scvtf(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 0, ConvertedSize, 0b11101, rd, rn);
|
||||
}
|
||||
@@ -249,70 +245,58 @@ public:
|
||||
}
|
||||
///< Comparison against 0.0
|
||||
void cmge(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b01000, rd, rn);
|
||||
}
|
||||
///< Comparison against 0.0
|
||||
void cmle(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b01001, rd, rn);
|
||||
}
|
||||
void neg(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b01011, rd, rn);
|
||||
}
|
||||
///< size is destination.
|
||||
void sqxtun(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b10010, rd, rn);
|
||||
}
|
||||
///< size is destination.
|
||||
void uqxtn(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i64Bit, "64-bit destination not supported");
|
||||
ASIMDScalar2RegMisc(0, 1, size, 0b10100, rd, rn);
|
||||
}
|
||||
///< size is destination.
|
||||
void fcvtxn(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMDScalar2RegMisc(0, 1, ScalarRegSize::i16Bit, 0b10110, rd, rn);
|
||||
}
|
||||
void fcvtnu(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 1, ConvertedSize, 0b11010, rd, rn);
|
||||
}
|
||||
void fcvtmu(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 1, ConvertedSize, 0b11011, rd, rn);
|
||||
}
|
||||
void fcvtau(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 1, ConvertedSize, 0b11100, rd, rn);
|
||||
}
|
||||
void ucvtf(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(0, 1, ConvertedSize, 0b11101, rd, rn);
|
||||
}
|
||||
@@ -366,73 +350,55 @@ public:
|
||||
}
|
||||
|
||||
void fmaxnmp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(1, 1, ConvertedSize, 0b01100, rd, rn);
|
||||
}
|
||||
void faddp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(1, 1, ConvertedSize, 0b01101, rd, rn);
|
||||
}
|
||||
void fmaxp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMDScalar2RegMisc(1, 1, ConvertedSize, 0b01111, rd, rn);
|
||||
}
|
||||
void fminnmp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMDScalar2RegMisc(1, 1, size, 0b01100, rd, rn);
|
||||
}
|
||||
void fminp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMDScalar2RegMisc(1, 1, size, 0b01111, rd, rn);
|
||||
}
|
||||
// Advanced SIMD scalar three different
|
||||
// Advanced SIMD scalar three different
|
||||
///< size is destination.
|
||||
void sqdmlal(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i32Bit :
|
||||
ScalarRegSize::i16Bit;
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i32Bit : ScalarRegSize::i16Bit;
|
||||
ASIMD3RegDifferent(0, ConvertedSize, 0b1001, rd, rn, rm);
|
||||
}
|
||||
///< size is destination.
|
||||
void sqdmlsl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i32Bit :
|
||||
ScalarRegSize::i16Bit;
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i32Bit : ScalarRegSize::i16Bit;
|
||||
ASIMD3RegDifferent(0, ConvertedSize, 0b1011, rd, rn, rm);
|
||||
}
|
||||
|
||||
///< size is destination.
|
||||
void sqdmull(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i32Bit :
|
||||
ScalarRegSize::i16Bit;
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i32Bit : ScalarRegSize::i16Bit;
|
||||
ASIMD3RegDifferent(0, ConvertedSize, 0b1101, rd, rn, rm);
|
||||
}
|
||||
// Advanced SIMD scalar three same
|
||||
// Advanced SIMD scalar three same
|
||||
void sqadd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(0, size, 0b00001, rd, rn, rm);
|
||||
}
|
||||
@@ -440,71 +406,62 @@ public:
|
||||
ASIMD3RegSame(0, size, 0b00101, rd, rn, rm);
|
||||
}
|
||||
void cmgt(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b00110, rd, rn, rm);
|
||||
}
|
||||
void cmge(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b00111, rd, rn, rm);
|
||||
}
|
||||
void sshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b01000, rd, rn, rm);
|
||||
}
|
||||
void sqshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(0, size, 0b01001, rd, rn, rm);
|
||||
}
|
||||
void srshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b01010, rd, rn, rm);
|
||||
}
|
||||
void sqrshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(0, size, 0b01011, rd, rn, rm);
|
||||
}
|
||||
void add(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b10000, rd, rn, rm);
|
||||
}
|
||||
void cmtst(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(0, size, 0b10001, rd, rn, rm);
|
||||
}
|
||||
void sqdmulh(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i32Bit || size == ScalarRegSize::i16Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i32Bit || size == ScalarRegSize::i16Bit, "Invalid size");
|
||||
ASIMD3RegSame(0, size, 0b10110, rd, rn, rm);
|
||||
}
|
||||
void fmulx(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(0, ConvertedSize, 0b11011, rd, rn, rm);
|
||||
}
|
||||
void fcmeq(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(0, ConvertedSize, 0b11100, rd, rn, rm);
|
||||
}
|
||||
void frecps(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(0, ConvertedSize, 0b11111, rd, rn, rm);
|
||||
}
|
||||
void frsqrts(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMD3RegSame(0, size, 0b11111, rd, rn, rm);
|
||||
}
|
||||
void uqadd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
@@ -514,75 +471,69 @@ public:
|
||||
ASIMD3RegSame(1, size, 0b00101, rd, rn, rm);
|
||||
}
|
||||
void cmhi(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b00110, rd, rn, rm);
|
||||
}
|
||||
void cmhs(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b00111, rd, rn, rm);
|
||||
}
|
||||
void ushl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b01000, rd, rn, rm);
|
||||
}
|
||||
void uqshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(1, size, 0b01001, rd, rn, rm);
|
||||
}
|
||||
void urshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b01010, rd, rn, rm);
|
||||
}
|
||||
void uqrshl(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
ASIMD3RegSame(1, size, 0b01011, rd, rn, rm);
|
||||
}
|
||||
void sub(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b10000, rd, rn, rm);
|
||||
}
|
||||
void cmeq(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit, "Only supports 64-bit");
|
||||
ASIMD3RegSame(1, size, 0b10001, rd, rn, rm);
|
||||
}
|
||||
void sqrdmulh(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i32Bit || size == ScalarRegSize::i16Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i32Bit || size == ScalarRegSize::i16Bit, "Invalid size");
|
||||
ASIMD3RegSame(1, size, 0b10110, rd, rn, rm);
|
||||
}
|
||||
void fcmge(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(1, ConvertedSize, 0b11100, rd, rn, rm);
|
||||
}
|
||||
void facge(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
|
||||
const ScalarRegSize ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ?
|
||||
ScalarRegSize::i16Bit :
|
||||
ScalarRegSize::i8Bit;
|
||||
const ScalarRegSize ConvertedSize = size == ScalarRegSize::i64Bit ? ScalarRegSize::i16Bit : ScalarRegSize::i8Bit;
|
||||
|
||||
ASIMD3RegSame(1, ConvertedSize, 0b11101, rd, rn, rm);
|
||||
}
|
||||
void fabd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMD3RegSame(1, size, 0b11010, rd, rn, rm);
|
||||
}
|
||||
void fcmgt(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMD3RegSame(1, size, 0b11100, rd, rn, rm);
|
||||
}
|
||||
void facgt(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for float convert");
|
||||
ASIMD3RegSame(1, size, 0b11101, rd, rn, rm);
|
||||
}
|
||||
// Advanced SIMD scalar shift by immediate
|
||||
// Advanced SIMD scalar shift by immediate
|
||||
void sshr(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -592,8 +543,8 @@ public:
|
||||
ASIMDScalarShiftByImm(0, immh, immb, 0b00000, rd, rn);
|
||||
}
|
||||
void ssra(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -603,8 +554,8 @@ public:
|
||||
ASIMDScalarShiftByImm(0, immh, immb, 0b00010, rd, rn);
|
||||
}
|
||||
void srshr(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -614,8 +565,8 @@ public:
|
||||
ASIMDScalarShiftByImm(0, immh, immb, 0b00100, rd, rn);
|
||||
}
|
||||
void srsra(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -625,8 +576,8 @@ public:
|
||||
ASIMDScalarShiftByImm(0, immh, immb, 0b00110, rd, rn);
|
||||
}
|
||||
void shl(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
// Shift encoded a bit weirdly.
|
||||
// shift = immh:immb - elementsize but immh is /also/ used for element size.
|
||||
const uint32_t immh = 1 << FEXCore::ToUnderlying(size) | (Shift >> 3);
|
||||
@@ -644,7 +595,7 @@ public:
|
||||
///< size is destination
|
||||
void sqshrn(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrn");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrn");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -655,7 +606,7 @@ public:
|
||||
}
|
||||
void sqrshrn(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrn");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrn");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -666,8 +617,8 @@ public:
|
||||
}
|
||||
// TODO: SCVTF, FCVTZS
|
||||
void ushr(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -677,8 +628,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b00000, rd, rn);
|
||||
}
|
||||
void usra(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -688,8 +639,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b00010, rd, rn);
|
||||
}
|
||||
void urshr(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -699,8 +650,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b00100, rd, rn);
|
||||
}
|
||||
void ursra(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -710,8 +661,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b00110, rd, rn);
|
||||
}
|
||||
void sri(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -721,8 +672,8 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b01000, rd, rn);
|
||||
}
|
||||
void sli(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_AA_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < 64, "Invalid shift for sshr");
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sshr");
|
||||
// Shift encoded a bit weirdly.
|
||||
// shift = immh:immb - elementsize but immh is /also/ used for element size.
|
||||
const uint32_t immh = 1 << FEXCore::ToUnderlying(size) | (Shift >> 3);
|
||||
@@ -748,7 +699,7 @@ public:
|
||||
///< size is destination.
|
||||
void sqshrun(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrun");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqshrun");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -760,7 +711,7 @@ public:
|
||||
///< size is destination.
|
||||
void sqrshrun(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -772,7 +723,7 @@ public:
|
||||
///< size is destination.
|
||||
void uqshrn(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -784,7 +735,7 @@ public:
|
||||
///< size is destination.
|
||||
void uqrshrn(ScalarRegSize size, VRegister rd, VRegister rn, uint32_t Shift) {
|
||||
LOGMAN_THROW_A_FMT(Shift > 0 && Shift < ScalarRegSizeInBits(size), "Invalid shift for sshr");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::ScalarRegSize::i64Bit, "Invalid size selected for sqrshrun");
|
||||
const size_t SubregSizeInBits = ScalarRegSizeInBits(size);
|
||||
// Shift encoded in immh:immb, but inverted with 128-bit source
|
||||
// shift = (esize * 2) - immh:immb
|
||||
@@ -794,10 +745,10 @@ public:
|
||||
ASIMDScalarShiftByImm(1, immh, immb, 0b10011, rd, rn);
|
||||
}
|
||||
// TODO: UCVTF, FCVTZU
|
||||
// Advanced SIMD scalar x indexed element
|
||||
// XXX:
|
||||
//
|
||||
// Floating-point data-processing (1 source)
|
||||
// Advanced SIMD scalar x indexed element
|
||||
// XXX:
|
||||
//
|
||||
// Floating-point data-processing (1 source)
|
||||
void fmov(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000000, rd, rn);
|
||||
}
|
||||
@@ -991,14 +942,14 @@ public:
|
||||
Float1Source(0, 0, 0b11, 0b001111, rd.V(), rn.V());
|
||||
}
|
||||
|
||||
// Floating-point compare
|
||||
// Floating-point compare
|
||||
void fcmp(ScalarRegSize Size, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(Size != ScalarRegSize::i8Bit, "8-bit destination not supported");
|
||||
LOGMAN_THROW_A_FMT(Size != ScalarRegSize::i8Bit, "8-bit destination not supported");
|
||||
|
||||
const auto ConvertedSize =
|
||||
Size == ARMEmitter::ScalarRegSize::i64Bit ? 0b01 :
|
||||
Size == ARMEmitter::ScalarRegSize::i32Bit ? 0b00 :
|
||||
Size == ARMEmitter::ScalarRegSize::i16Bit ? 0b11 : 0;
|
||||
const auto ConvertedSize = Size == ARMEmitter::ScalarRegSize::i64Bit ? 0b01 :
|
||||
Size == ARMEmitter::ScalarRegSize::i32Bit ? 0b00 :
|
||||
Size == ARMEmitter::ScalarRegSize::i16Bit ? 0b11 :
|
||||
0;
|
||||
|
||||
FloatCompare(0, 0, ConvertedSize, 0b00, 0b00000, rn, rm);
|
||||
}
|
||||
@@ -1051,7 +1002,7 @@ public:
|
||||
FloatCompare(0, 0, 0b11, 0b00, 0b11000, rn.V(), VReg::v0);
|
||||
}
|
||||
|
||||
// Floating-point immediate
|
||||
// Floating-point immediate
|
||||
void fmov(ARMEmitter::ScalarRegSize size, ARMEmitter::VRegister rd, float Value) {
|
||||
uint32_t M = 0;
|
||||
uint32_t S = 0;
|
||||
@@ -1061,16 +1012,13 @@ public:
|
||||
if (size == ARMEmitter::ScalarRegSize::i16Bit) {
|
||||
LOGMAN_MSG_A_FMT("Unsupported");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
else if (size == ARMEmitter::ScalarRegSize::i32Bit) {
|
||||
} else if (size == ARMEmitter::ScalarRegSize::i32Bit) {
|
||||
ptype = 0b00;
|
||||
imm8 = FP32ToImm8(Value);
|
||||
}
|
||||
else if (size == ARMEmitter::ScalarRegSize::i64Bit) {
|
||||
} else if (size == ARMEmitter::ScalarRegSize::i64Bit) {
|
||||
ptype = 0b01;
|
||||
imm8 = FP64ToImm8(Value);
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
@@ -1090,7 +1038,7 @@ public:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Floating-point conditional compare
|
||||
// Floating-point conditional compare
|
||||
void fccmp(SRegister rn, SRegister rm, StatusFlags flags, Condition Cond) {
|
||||
FloatConditionalCompare(0, 0, 0b00, 0b0, rn.V(), rm.V(), flags, Cond);
|
||||
}
|
||||
@@ -1110,7 +1058,7 @@ public:
|
||||
FloatConditionalCompare(0, 0, 0b11, 0b1, rn.V(), rm.V(), flags, Cond);
|
||||
}
|
||||
|
||||
// Floating-point data-processing (2 source)
|
||||
// Floating-point data-processing (2 source)
|
||||
void fmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0000, rd, rn, rm);
|
||||
}
|
||||
@@ -1225,11 +1173,10 @@ public:
|
||||
|
||||
// Floating-point conditional select
|
||||
void fcsel(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit,
|
||||
"Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
const uint32_t ConvertedSize = size == ScalarRegSize::i64Bit ? 0b01 : size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
FloatConditionalSelect(0, 0, ConvertedSize, rd, rn, rm, Cond);
|
||||
}
|
||||
@@ -1244,7 +1191,7 @@ public:
|
||||
FloatConditionalSelect(0, 0, 0b11, rd.V(), rn.V(), rm.V(), Cond);
|
||||
}
|
||||
|
||||
// Floating-point data-processing (3 source)
|
||||
// Floating-point data-processing (3 source)
|
||||
void fmadd(SRegister rd, SRegister rn, SRegister rm, SRegister ra) {
|
||||
Float3Source(0, 0, 0b00, 0, 0, rd.V(), rn.V(), rm.V(), ra.V());
|
||||
}
|
||||
@@ -1285,7 +1232,7 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
// Advanced SIMD scalar copy
|
||||
// Advanced SIMD scalar copy
|
||||
void ASIMDScalarCopy(uint32_t Op, uint32_t Q, uint32_t imm5, uint32_t imm4, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
@@ -1297,7 +1244,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar three same FP16
|
||||
// Advanced SIMD scalar three same FP16
|
||||
void ASIMDScalarThreeSameFP16(uint32_t U, uint32_t a, uint32_t opcode, HRegister rm, HRegister rn, HRegister rd) {
|
||||
uint32_t Instr = 0b0101'1110'0100'0000'0000'0100'0000'0000;
|
||||
|
||||
@@ -1309,7 +1256,7 @@ private:
|
||||
Instr |= rd.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
// Advanced SIMD scalar two-register miscellaneous FP16
|
||||
// Advanced SIMD scalar two-register miscellaneous FP16
|
||||
void ASIMDScalarTwoRegMiscFP16(uint32_t U, uint32_t a, uint32_t opcode, HRegister rn, HRegister rd) {
|
||||
uint32_t Instr = 0b0101'1110'0111'1000'0000'1000'0000'0000;
|
||||
|
||||
@@ -1321,9 +1268,9 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar three same extra
|
||||
// XXX:
|
||||
// Advanced SIMD scalar two-register miscellaneous
|
||||
// Advanced SIMD scalar three same extra
|
||||
// XXX:
|
||||
// Advanced SIMD scalar two-register miscellaneous
|
||||
void ASIMDScalar2RegMisc(uint32_t b20, uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
uint32_t Instr = 0b0101'1110'0010'0000'0000'1000'0000'0000;
|
||||
|
||||
@@ -1336,9 +1283,9 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Advanced SIMD scalar pairwise
|
||||
// XXX:
|
||||
// Advanced SIMD scalar three different
|
||||
// Advanced SIMD scalar pairwise
|
||||
// XXX:
|
||||
// Advanced SIMD scalar three different
|
||||
void ASIMD3RegDifferent(uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0101'1110'0010'0000'0000'0000'0000'0000;
|
||||
|
||||
@@ -1350,7 +1297,7 @@ private:
|
||||
Instr |= Encode_rd(rd);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Advanced SIMD scalar three same
|
||||
// Advanced SIMD scalar three same
|
||||
void ASIMD3RegSame(uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0101'1110'0010'0000'0000'0100'0000'0000;
|
||||
|
||||
@@ -1362,7 +1309,7 @@ private:
|
||||
Instr |= Encode_rd(rd);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Advanced SIMD scalar shift by immediate
|
||||
// Advanced SIMD scalar shift by immediate
|
||||
void ASIMDScalarShiftByImm(uint32_t U, uint32_t immh, uint32_t immb, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
uint32_t Instr = 0b0101'1111'0000'0000'0000'0100'0000'0000;
|
||||
|
||||
@@ -1374,9 +1321,9 @@ private:
|
||||
Instr |= Encode_rd(rd);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Advanced SIMD scalar x indexed element
|
||||
// XXX:
|
||||
// Floating-point data-processing (1 source)
|
||||
// Advanced SIMD scalar x indexed element
|
||||
// XXX:
|
||||
// Floating-point data-processing (1 source)
|
||||
void Float1Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0100'0000'0000'0000;
|
||||
|
||||
@@ -1390,16 +1337,15 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
void Float1Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit,
|
||||
"Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
const uint32_t ConvertedSize = size == ScalarRegSize::i64Bit ? 0b01 : size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float1Source(M, S, ConvertedSize, opcode, rd, rn);
|
||||
}
|
||||
|
||||
// Floating-point compare
|
||||
// Floating-point compare
|
||||
void FloatCompare(uint32_t M, uint32_t S, uint32_t ftype, uint32_t op, uint32_t opcode2, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0010'0000'0000'0000;
|
||||
|
||||
@@ -1413,9 +1359,9 @@ private:
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
// Floating-point immediate
|
||||
// XXX:
|
||||
// Floating-point conditional compare
|
||||
// Floating-point immediate
|
||||
// XXX:
|
||||
// Floating-point conditional compare
|
||||
void FloatConditionalCompare(uint32_t M, uint32_t S, uint32_t ptype, uint32_t op, VRegister rn, VRegister rm, StatusFlags flags, Condition Cond) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'0100'0000'0000;
|
||||
|
||||
@@ -1430,7 +1376,7 @@ private:
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
// Floating-point data-processing (2 source)
|
||||
// Floating-point data-processing (2 source)
|
||||
|
||||
void Float2Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1000'0000'0000;
|
||||
@@ -1447,16 +1393,15 @@ private:
|
||||
}
|
||||
|
||||
void Float2Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
LOGMAN_THROW_A_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit,
|
||||
"Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
const uint32_t ConvertedSize = size == ScalarRegSize::i64Bit ? 0b01 : size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float2Source(M, S, ConvertedSize, opcode, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Floating-point conditional select
|
||||
// Floating-point conditional select
|
||||
void FloatConditionalSelect(uint32_t M, uint32_t S, uint32_t ptype, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1100'0000'0000;
|
||||
|
||||
@@ -1470,7 +1415,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Floating-point data-processing (3 source)
|
||||
// Floating-point data-processing (3 source)
|
||||
void Float3Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t o1, uint32_t o0, VRegister rd, VRegister rn, VRegister rm, VRegister ra) {
|
||||
uint32_t Instr = 0b0001'1111'0000'0000'0000'0000'0000'0000;
|
||||
|
||||
@@ -1485,3 +1430,8 @@ private:
|
||||
Instr |= Encode_rd(rd);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
}; // struct LoadstoreEmitterOps
|
||||
} // namespace ARMEmitter
|
||||
#endif
|
||||
@@ -4,173 +4,185 @@
|
||||
* This is mostly a mashup of various instruction types.
|
||||
* Nothing follows an explicit pattern since they are mostly different.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
namespace ARMEmitter {
|
||||
struct EmitterOps : Emitter {
|
||||
#endif
|
||||
|
||||
public:
|
||||
// System with result
|
||||
// TODO: SYSL
|
||||
// System Instruction
|
||||
// TODO: AT
|
||||
// TODO: CFP
|
||||
// TODO: CPP
|
||||
void dc(ARMEmitter::DataCacheOperation DCOp, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'1000'0111 << 12;
|
||||
SystemInstruction(Op, 0, FEXCore::ToUnderlying(DCOp), rt);
|
||||
}
|
||||
// TODO: DVP
|
||||
// TODO: IC
|
||||
// TODO: TLBI
|
||||
// System with result
|
||||
// TODO: SYSL
|
||||
// System Instruction
|
||||
// TODO: AT
|
||||
// TODO: CFP
|
||||
// TODO: CPP
|
||||
void dc(ARMEmitter::DataCacheOperation DCOp, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'1000'0111 << 12;
|
||||
SystemInstruction(Op, 0, FEXCore::ToUnderlying(DCOp), rt);
|
||||
}
|
||||
// TODO: DVP
|
||||
// TODO: IC
|
||||
// TODO: TLBI
|
||||
|
||||
// Exception generation
|
||||
void svc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b01, Imm);
|
||||
}
|
||||
void hvc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b10, Imm);
|
||||
}
|
||||
void smc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b11, Imm);
|
||||
}
|
||||
void brk(uint32_t Imm) {
|
||||
ExceptionGeneration(0b001, 0b000, 0b00, Imm);
|
||||
}
|
||||
void hlt(uint32_t Imm) {
|
||||
ExceptionGeneration(0b010, 0b000, 0b00, Imm);
|
||||
}
|
||||
void tcancel(uint32_t Imm) {
|
||||
ExceptionGeneration(0b011, 0b000, 0b00, Imm);
|
||||
}
|
||||
void dcps1(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b01, Imm);
|
||||
}
|
||||
void dcps2(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b10, Imm);
|
||||
}
|
||||
void dcps3(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b11, Imm);
|
||||
}
|
||||
// System instructions with register argument
|
||||
void wfet(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b000, rt);
|
||||
}
|
||||
void wfit(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b001, rt);
|
||||
}
|
||||
// Exception generation
|
||||
void svc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b01, Imm);
|
||||
}
|
||||
void hvc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b10, Imm);
|
||||
}
|
||||
void smc(uint32_t Imm) {
|
||||
ExceptionGeneration(0b000, 0b000, 0b11, Imm);
|
||||
}
|
||||
void brk(uint32_t Imm) {
|
||||
ExceptionGeneration(0b001, 0b000, 0b00, Imm);
|
||||
}
|
||||
void hlt(uint32_t Imm) {
|
||||
ExceptionGeneration(0b010, 0b000, 0b00, Imm);
|
||||
}
|
||||
void tcancel(uint32_t Imm) {
|
||||
ExceptionGeneration(0b011, 0b000, 0b00, Imm);
|
||||
}
|
||||
void dcps1(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b01, Imm);
|
||||
}
|
||||
void dcps2(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b10, Imm);
|
||||
}
|
||||
void dcps3(uint32_t Imm) {
|
||||
ExceptionGeneration(0b101, 0b000, 0b11, Imm);
|
||||
}
|
||||
// System instructions with register argument
|
||||
void wfet(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b000, rt);
|
||||
}
|
||||
void wfit(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b001, rt);
|
||||
}
|
||||
|
||||
// Hints
|
||||
void nop() {
|
||||
Hint(ARMEmitter::HintRegister::NOP);
|
||||
}
|
||||
void yield() {
|
||||
Hint(ARMEmitter::HintRegister::YIELD);
|
||||
}
|
||||
void wfe() {
|
||||
Hint(ARMEmitter::HintRegister::WFE);
|
||||
}
|
||||
void wfi() {
|
||||
Hint(ARMEmitter::HintRegister::WFI);
|
||||
}
|
||||
void sev() {
|
||||
Hint(ARMEmitter::HintRegister::SEV);
|
||||
}
|
||||
void sevl() {
|
||||
Hint(ARMEmitter::HintRegister::SEVL);
|
||||
}
|
||||
void dgh() {
|
||||
Hint(ARMEmitter::HintRegister::DGH);
|
||||
}
|
||||
void csdb() {
|
||||
Hint(ARMEmitter::HintRegister::CSDB);
|
||||
}
|
||||
// Hints
|
||||
void nop() {
|
||||
Hint(ARMEmitter::HintRegister::NOP);
|
||||
}
|
||||
void yield() {
|
||||
Hint(ARMEmitter::HintRegister::YIELD);
|
||||
}
|
||||
void wfe() {
|
||||
Hint(ARMEmitter::HintRegister::WFE);
|
||||
}
|
||||
void wfi() {
|
||||
Hint(ARMEmitter::HintRegister::WFI);
|
||||
}
|
||||
void sev() {
|
||||
Hint(ARMEmitter::HintRegister::SEV);
|
||||
}
|
||||
void sevl() {
|
||||
Hint(ARMEmitter::HintRegister::SEVL);
|
||||
}
|
||||
void dgh() {
|
||||
Hint(ARMEmitter::HintRegister::DGH);
|
||||
}
|
||||
void csdb() {
|
||||
Hint(ARMEmitter::HintRegister::CSDB);
|
||||
}
|
||||
|
||||
// Barriers
|
||||
void clrex(uint32_t imm = 15) {
|
||||
LOGMAN_THROW_AA_FMT(imm < 16, "Immediate out of range");
|
||||
Barrier(ARMEmitter::BarrierRegister::CLREX, imm);
|
||||
}
|
||||
void dsb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void dmb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void isb() {
|
||||
Barrier(ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(ARMEmitter::BarrierScope::SY));
|
||||
}
|
||||
void sb() {
|
||||
Barrier(ARMEmitter::BarrierRegister::SB, 0);
|
||||
}
|
||||
void tcommit() {
|
||||
Barrier(ARMEmitter::BarrierRegister::TCOMMIT, 0);
|
||||
}
|
||||
// Barriers
|
||||
void clrex(uint32_t imm = 15) {
|
||||
LOGMAN_THROW_A_FMT(imm < 16, "Immediate out of range");
|
||||
Barrier(ARMEmitter::BarrierRegister::CLREX, imm);
|
||||
}
|
||||
void dsb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void dmb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void isb() {
|
||||
Barrier(ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(ARMEmitter::BarrierScope::SY));
|
||||
}
|
||||
void sb() {
|
||||
Barrier(ARMEmitter::BarrierRegister::SB, 0);
|
||||
}
|
||||
void tcommit() {
|
||||
Barrier(ARMEmitter::BarrierRegister::TCOMMIT, 0);
|
||||
}
|
||||
|
||||
// System register move
|
||||
void msr(ARMEmitter::SystemRegister reg, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0001 << 20;
|
||||
SystemRegisterMove(Op, rt, reg);
|
||||
}
|
||||
// System register move
|
||||
void msr(ARMEmitter::SystemRegister reg, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0001 << 20;
|
||||
SystemRegisterMove(Op, rt, reg);
|
||||
}
|
||||
|
||||
void mrs(ARMEmitter::Register rd, ARMEmitter::SystemRegister reg) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0011 << 20;
|
||||
SystemRegisterMove(Op, rd, reg);
|
||||
}
|
||||
void mrs(ARMEmitter::Register rd, ARMEmitter::SystemRegister reg) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0011 << 20;
|
||||
SystemRegisterMove(Op, rd, reg);
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
// Exception Generation
|
||||
void ExceptionGeneration(uint32_t opc, uint32_t op2, uint32_t LL, uint32_t Imm) {
|
||||
LOGMAN_THROW_AA_FMT((Imm & 0xFFFF'0000) == 0, "Imm amount too large");
|
||||
// Exception Generation
|
||||
void ExceptionGeneration(uint32_t opc, uint32_t op2, uint32_t LL, uint32_t Imm) {
|
||||
LOGMAN_THROW_A_FMT((Imm & 0xFFFF'0000) == 0, "Imm amount too large");
|
||||
|
||||
uint32_t Instr = 0b1101'0100 << 24;
|
||||
uint32_t Instr = 0b1101'0100 << 24;
|
||||
|
||||
Instr |= opc << 21;
|
||||
Instr |= Imm << 5;
|
||||
Instr |= op2 << 2;
|
||||
Instr |= LL;
|
||||
Instr |= opc << 21;
|
||||
Instr |= Imm << 5;
|
||||
Instr |= op2 << 2;
|
||||
Instr |= LL;
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// System instructions with register argument
|
||||
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0001 << 12;
|
||||
// System instructions with register argument
|
||||
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0001 << 12;
|
||||
|
||||
Instr |= CRm << 8;
|
||||
Instr |= op2 << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
Instr |= CRm << 8;
|
||||
Instr |= op2 << 5;
|
||||
Instr |= Encode_rt(rt);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Hints
|
||||
void Hint(ARMEmitter::HintRegister Reg) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0010'0000'0001'1111U;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Barriers
|
||||
void Barrier(ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0011'0000'0001'1111U;
|
||||
Instr |= CRm << 8;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Hints
|
||||
void Hint(ARMEmitter::HintRegister Reg) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0010'0000'0001'1111U;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Barriers
|
||||
void Barrier(ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0011'0000'0001'1111U;
|
||||
Instr |= CRm << 8;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// System Instruction
|
||||
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = Op;
|
||||
// System Instruction
|
||||
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= L << 21;
|
||||
Instr |= SubOp;
|
||||
Instr |= Encode_rt(rt);
|
||||
Instr |= L << 21;
|
||||
Instr |= SubOp;
|
||||
Instr |= Encode_rt(rt);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// System register move
|
||||
void SystemRegisterMove(uint32_t Op, ARMEmitter::Register rt, ARMEmitter::SystemRegister reg) {
|
||||
uint32_t Instr = Op;
|
||||
// System register move
|
||||
void SystemRegisterMove(uint32_t Op, ARMEmitter::Register rt, ARMEmitter::SystemRegister reg) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= FEXCore::ToUnderlying(reg);
|
||||
Instr |= Encode_rt(rt);
|
||||
Instr |= FEXCore::ToUnderlying(reg);
|
||||
Instr |= Encode_rt(rt);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
#ifndef INCLUDED_BY_EMITTER
|
||||
}; // struct LoadstoreEmitterOps
|
||||
} // namespace ARMEmitter
|
||||
#endif
|
||||
@@ -34,11 +34,7 @@
|
||||
// by the corresponding fields in the logical instruction.
|
||||
// If it can not be encoded, the function returns false, and the values pointed
|
||||
// to by n, imm_s and imm_r are undefined.
|
||||
static bool IsImmLogical(uint64_t value,
|
||||
unsigned width,
|
||||
unsigned* n = nullptr,
|
||||
unsigned* imm_s = nullptr,
|
||||
unsigned* imm_r = nullptr) {
|
||||
static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr, unsigned* imm_s = nullptr, unsigned* imm_r = nullptr) {
|
||||
[[maybe_unused]] constexpr auto kBRegSize = 8;
|
||||
[[maybe_unused]] constexpr auto kHRegSize = 16;
|
||||
[[maybe_unused]] constexpr auto kSRegSize = 32;
|
||||
@@ -47,8 +43,7 @@ static bool IsImmLogical(uint64_t value,
|
||||
constexpr auto kWRegSize = 32;
|
||||
constexpr auto kXRegSize = 64;
|
||||
|
||||
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) ||
|
||||
(width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
|
||||
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) || (width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
|
||||
|
||||
bool negate = false;
|
||||
|
||||
@@ -182,12 +177,7 @@ static bool IsImmLogical(uint64_t value,
|
||||
// (1 + 2^d + 2^(2d) + ...), i.e. 0x0001000100010001 or similar. These can
|
||||
// be derived using a table lookup on CLZ(d).
|
||||
static const uint64_t multipliers[] = {
|
||||
0x0000000000000001UL,
|
||||
0x0000000100000001UL,
|
||||
0x0001000100010001UL,
|
||||
0x0101010101010101UL,
|
||||
0x1111111111111111UL,
|
||||
0x5555555555555555UL,
|
||||
0x0000000000000001UL, 0x0000000100000001UL, 0x0001000100010001UL, 0x0101010101010101UL, 0x1111111111111111UL, 0x5555555555555555UL,
|
||||
};
|
||||
uint64_t multiplier = multipliers[CountLeadingZeros(d, kXRegSize) - 57];
|
||||
uint64_t candidate = (b - a) * multiplier;
|
||||
@@ -244,7 +234,9 @@ static bool IsImmLogical(uint64_t value,
|
||||
}
|
||||
|
||||
static inline bool IsIntN(unsigned n, int64_t x) {
|
||||
if (n == 64) return true;
|
||||
if (n == 64) {
|
||||
return true;
|
||||
}
|
||||
int64_t limit = INT64_C(1) << (n - 1);
|
||||
return (-limit <= x) && (x < limit);
|
||||
}
|
||||
@@ -271,11 +263,15 @@ V(57) V(58) V(59) V(60) V(61) V(62) V(63)
|
||||
|
||||
// clang-format on
|
||||
|
||||
#define DECLARE_IS_INT_N(N) \
|
||||
static inline bool IsInt##N(int64_t x) { return IsIntN(N, x); }
|
||||
#define DECLARE_IS_INT_N(N) \
|
||||
static inline bool IsInt##N(int64_t x) { \
|
||||
return IsIntN(N, x); \
|
||||
}
|
||||
|
||||
#define DECLARE_IS_UINT_N(N) \
|
||||
static inline bool IsUint##N(int64_t x) { return IsUintN(N, x); }
|
||||
#define DECLARE_IS_UINT_N(N) \
|
||||
static inline bool IsUint##N(int64_t x) { \
|
||||
return IsUintN(N, x); \
|
||||
}
|
||||
|
||||
INT_1_TO_63_LIST(DECLARE_IS_INT_N)
|
||||
INT_1_TO_63_LIST(DECLARE_IS_UINT_N)
|
||||
@@ -285,14 +281,14 @@ INT_1_TO_63_LIST(DECLARE_IS_UINT_N)
|
||||
|
||||
private:
|
||||
|
||||
template <typename V>
|
||||
template<typename V>
|
||||
static inline bool IsPowerOf2(V value) {
|
||||
return (value != 0) && ((value & (value - 1)) == 0);
|
||||
}
|
||||
|
||||
// Some compilers dislike negating unsigned integers,
|
||||
// so we provide an equivalent.
|
||||
template <typename T>
|
||||
template<typename T>
|
||||
static inline T UnsignedNegate(T value) {
|
||||
static_assert(std::is_unsigned<T>::value);
|
||||
return ~value + 1;
|
||||
@@ -302,7 +298,7 @@ static inline uint64_t LowestSetBit(uint64_t value) {
|
||||
return value & UnsignedNegate(value);
|
||||
}
|
||||
|
||||
template <typename V>
|
||||
template<typename V>
|
||||
static inline int CountLeadingZeros(V value, int width = (sizeof(V) * 8)) {
|
||||
#if COMPILER_HAS_BUILTIN_CLZ
|
||||
if (width == 32) {
|
||||
|
||||
Vendored
+1
-1
Submodule External/drm-headers updated: 8efb6dc03f...0675d2f291.
Vendored
+1
-1
@@ -1,3 +1,3 @@
|
||||
set(NAME tiny-json)
|
||||
set(SRCS tiny-json.c)
|
||||
add_library(${NAME} ${SRCS})
|
||||
add_library(${NAME} STATIC ${SRCS})
|
||||
Vendored
+1
-1
Submodule External/vixl updated: a90f5d5020...3180ab603b.
@@ -220,7 +220,11 @@ def print_man_environment_tail():
|
||||
print_man_env_option(
|
||||
"FEX_PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored. These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
|
||||
"For FEXInterpreter on Linux:",
|
||||
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"For Arm64ec/Wow64 WINE builds:",
|
||||
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
|
||||
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
|
||||
],
|
||||
"''", True)
|
||||
|
||||
@@ -101,8 +101,7 @@ def is_ssa_type(type):
|
||||
if (type == "SSA" or
|
||||
type == "GPR" or
|
||||
type == "GPRPair" or
|
||||
type == "FPR" or
|
||||
type == "PRED"):
|
||||
type == "FPR"):
|
||||
return True
|
||||
return False
|
||||
|
||||
@@ -151,8 +150,8 @@ def parse_ops(ops):
|
||||
RHS += f", {DType}:$Out{Name}"
|
||||
else:
|
||||
# Single anonymous destination
|
||||
if LHS not in ["SSA", "GPR", "GPRPair", "FPR", "PRED"]:
|
||||
ExitError(f"Unknown destination class type {LHS}. Needs to be one of SSA, GPR, GPRPair, FPR, PRED")
|
||||
if LHS not in ["SSA", "GPR", "GPRPair", "FPR"]:
|
||||
ExitError(f"Unknown destination class type {LHS}. Needs to be one of SSA, GPR, GPRPair, FPR")
|
||||
|
||||
OpDef.HasDest = True
|
||||
OpDef.DestType = LHS
|
||||
@@ -222,8 +221,7 @@ def parse_ops(ops):
|
||||
if (OpArg.IsSSA and
|
||||
(OpArg.Type == "GPR" or
|
||||
OpArg.Type == "GPRPair" or
|
||||
OpArg.Type == "FPR" or
|
||||
OpArg.Type == "PRED")):
|
||||
OpArg.Type == "FPR")):
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
|
||||
|
||||
OpArg.Name = ArgName
|
||||
|
||||
@@ -21,12 +21,12 @@ struct BitSet final {
|
||||
ElementType* Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = ToBytes(Elements);
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
|
||||
}
|
||||
void Realloc(size_t Elements) {
|
||||
size_t AllocateSize = ToBytes(Elements);
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
|
||||
}
|
||||
void Free() {
|
||||
@@ -68,7 +68,7 @@ struct BitSetView final {
|
||||
ElementType* Memory;
|
||||
|
||||
void GetView(BitSet<T>& Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
|
||||
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
|
||||
@@ -334,9 +334,14 @@ void ReloadMetaLayer() {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedRootFS = GetDataDirectory(false) + "RootFS/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
const auto PathNameCopy = *PathName;
|
||||
for (auto Global : {true, false}) {
|
||||
for (auto DirectoryFetchers : {GetDataDirectory, GetConfigDirectory}) {
|
||||
fextl::string NamedRootFS = DirectoryFetchers(Global) + "RootFS/" + PathNameCopy;
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -356,9 +361,14 @@ void ReloadMetaLayer() {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
fextl::string NamedConfig = GetDataDirectory(false) + "ThunkConfigs/" + *PathName;
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
const auto PathNameCopy = *PathName;
|
||||
for (auto Global : {true, false}) {
|
||||
for (auto DirectoryFetchers : {GetDataDirectory, GetConfigDirectory}) {
|
||||
fextl::string NamedConfig = DirectoryFetchers(Global) + "ThunkConfigs/" + PathNameCopy;
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -472,6 +472,13 @@
|
||||
"Sleeps the process at startup for a duration of seconds.",
|
||||
"Useful if an application crashes too quickly to attach a debugger."
|
||||
]
|
||||
},
|
||||
"StartupSleepProcName": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Contrains the startup sleep to only apply to processes that match this name."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
|
||||
@@ -271,7 +271,8 @@ public:
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
struct GenerateIRResult {
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
std::optional<IR::IRListView> IRView;
|
||||
IR::RegisterAllocationData* RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
@@ -282,9 +283,7 @@ public:
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
bool GeneratedIR;
|
||||
fextl::unique_ptr<FEXCore::Core::DebugData> DebugData;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
@@ -337,6 +336,10 @@ protected:
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
VectorAtomicTSOEmulationEnabled = false;
|
||||
MemcpyAtomicTSOEmulationEnabled = false;
|
||||
} else if (Config.ParanoidTSO) {
|
||||
AtomicTSOEmulationEnabled = true;
|
||||
VectorAtomicTSOEmulationEnabled = true;
|
||||
MemcpyAtomicTSOEmulationEnabled = true;
|
||||
} else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
@@ -57,12 +56,6 @@ namespace x64 {
|
||||
ARMEmitter::Reg::r24, ARMEmitter::Reg::r25, ARMEmitter::Reg::r30, ARMEmitter::Reg::r18,
|
||||
};
|
||||
|
||||
// p6 and p7 registers are used as temporaries no not added here for RA
|
||||
// See PREF_TMP_16B and PREF_TMP_32B
|
||||
// p0-p1 are also used in the jit as temps.
|
||||
// Also p8-p15 cannot be used can only encode p0-p7, so we're left with p2-p5.
|
||||
constexpr std::array<ARMEmitter::PRegister, 4> PR = {ARMEmitter::PReg::p2, ARMEmitter::PReg::p3, ARMEmitter::PReg::p4, ARMEmitter::PReg::p5};
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
// All are caller saved
|
||||
@@ -100,6 +93,7 @@ namespace x64 {
|
||||
ARMEmitter::Reg::r20,
|
||||
ARMEmitter::Reg::r21,
|
||||
ARMEmitter::Reg::r22,
|
||||
// PF/AF must be last.
|
||||
REG_PF,
|
||||
REG_AF,
|
||||
};
|
||||
@@ -109,12 +103,6 @@ namespace x64 {
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
// p6 and p7 registers are used as temporaries no not added here for RA
|
||||
// See PREF_TMP_16B and PREF_TMP_32B
|
||||
// p0-p1 are also used in the jit as temps.
|
||||
// Also p8-p15 cannot be used can only encode p0-p7, so we're left with p2-p5.
|
||||
constexpr std::array<ARMEmitter::PRegister, 4> PR = {ARMEmitter::PReg::p2, ARMEmitter::PReg::p3, ARMEmitter::PReg::p4, ARMEmitter::PReg::p5};
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
@@ -246,12 +234,6 @@ namespace x32 {
|
||||
|
||||
constexpr unsigned RAPairs = 12;
|
||||
|
||||
// p6 and p7 registers are used as temporaries no not added here for RA
|
||||
// See PREF_TMP_16B and PREF_TMP_32B
|
||||
// p0-p1 are also used in the jit as temps.
|
||||
// Also p8-p15 cannot be used can only encode p0-p7, so we're left with p2-p5.
|
||||
constexpr std::array<ARMEmitter::PRegister, 4> PR = {ARMEmitter::PReg::p2, ARMEmitter::PReg::p3, ARMEmitter::PReg::p4, ARMEmitter::PReg::p5};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
ARMEmitter::VReg::v16, ARMEmitter::VReg::v17, ARMEmitter::VReg::v18, ARMEmitter::VReg::v19,
|
||||
@@ -375,7 +357,6 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
GeneralRegisters = x64::RA;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
PredicateRegisters = x64::PR;
|
||||
PairRegisters = x64::RAPairs;
|
||||
#ifdef _M_ARM_64EC
|
||||
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
|
||||
@@ -389,8 +370,6 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
|
||||
PredicateRegisters = x32::PR;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -631,7 +610,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
}
|
||||
#endif
|
||||
|
||||
if (SetPredRegs) {
|
||||
if (SetPredRegs && (EmitterCTX->HostFeatures.SupportsSVE256 || EmitterCTX->HostFeatures.SupportsSVE128)) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
@@ -643,6 +622,9 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE128) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
|
||||
// Fill in the predicate register for the x87 ldst SVE optimization.
|
||||
ptrue(ARMEmitter::SubRegSize::i16Bit, PRED_X87_SVEOPT, ARMEmitter::PredicatePattern::SVE_VL5);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1067,7 +1049,7 @@ void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
}
|
||||
|
||||
// Fill the static registers.
|
||||
FillStaticRegs(true, PreserveSRAMask, PreserveSRAFPRMask);
|
||||
FillStaticRegs(FPRs, PreserveSRAMask, PreserveSRAFPRMask);
|
||||
|
||||
// Pop the vector registers.
|
||||
PopVectorRegisters(CanUseSVE256, DynamicFPRs);
|
||||
|
||||
@@ -18,10 +18,8 @@
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
@@ -48,6 +46,10 @@ constexpr auto REG_AF = ARMEmitter::Reg::r27;
|
||||
// Vector temporaries
|
||||
constexpr auto VTMP1 = ARMEmitter::VReg::v0;
|
||||
constexpr auto VTMP2 = ARMEmitter::VReg::v1;
|
||||
|
||||
// Predicate register for X87 SVE Optimization
|
||||
constexpr auto SVE_OPT_PRED = ARMEmitter::PReg::p2;
|
||||
|
||||
#else
|
||||
constexpr auto TMP1 = ARMEmitter::XReg::x10;
|
||||
constexpr auto TMP2 = ARMEmitter::XReg::x11;
|
||||
@@ -67,6 +69,9 @@ constexpr auto VTMP2 = ARMEmitter::VReg::v17;
|
||||
constexpr auto EC_CALL_CHECKER_PC_REG = ARMEmitter::XReg::x9;
|
||||
constexpr auto EC_ENTRY_CPUAREA_REG = ARMEmitter::XReg::x17;
|
||||
|
||||
// Predicate register for X87 SVE Optimization
|
||||
constexpr auto SVE_OPT_PRED = ARMEmitter::PReg::p2;
|
||||
|
||||
// These structures are not included in the standard Windows headers, define the offsets of members we care about for EC here.
|
||||
constexpr size_t TEB_CPU_AREA_OFFSET = 0x1788;
|
||||
constexpr size_t TEB_PEB_OFFSET = 0x60;
|
||||
@@ -74,11 +79,16 @@ constexpr size_t PEB_EC_CODE_BITMAP_OFFSET = 0x368;
|
||||
constexpr size_t CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET = 0x1;
|
||||
constexpr size_t CPU_AREA_EMULATOR_STACK_BASE_OFFSET = 0x8;
|
||||
constexpr size_t CPU_AREA_EMULATOR_DATA_OFFSET = 0x30;
|
||||
|
||||
constexpr uint64_t EC_CODE_BITMAP_MAX_ADDRESS = 1ULL << 47;
|
||||
#endif
|
||||
|
||||
// Will force one single instruction block to be generated first if set when entering the JIT filling SRA.
|
||||
constexpr auto ENTRY_FILL_SRA_SINGLE_INST_REG = TMP1;
|
||||
|
||||
// Predicate to use in the X87 SVE optimization
|
||||
constexpr ARMEmitter::PRegister PRED_X87_SVEOPT = ARMEmitter::PReg::p2;
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
|
||||
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
|
||||
@@ -97,7 +107,6 @@ protected:
|
||||
std::span<const ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegisters {};
|
||||
std::span<const ARMEmitter::PRegister> PredicateRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> StaticFPRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
|
||||
@@ -373,7 +373,7 @@ namespace CPU {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
|
||||
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
|
||||
@@ -102,22 +102,6 @@ namespace CPU {
|
||||
uint32_t _Pad;
|
||||
};
|
||||
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
// Using these entries FEX is able to reconstruct the guest RIP accurately when an instruction cause a signal fault.
|
||||
// Packed using 16-bit entries to ensure the size isn't too large.
|
||||
// These smaller sizes means that each entry is relative to each other instead of absolute offset from the start of the JIT block.
|
||||
// When reconstructing the RIP, each entry must be walked linearly and accumulated with the previous entries.
|
||||
// This is a trade-off between compression inside the JIT code space and execution time when reconstruction the RIP.
|
||||
// RIP reconstruction when faulting is less likely so we are requiring the accumulation.
|
||||
struct JITRIPReconstructEntries {
|
||||
// The Host PC offset from the previous entry.
|
||||
uint16_t HostPCOffset;
|
||||
|
||||
// How much to offset the RIP from the previous entry.
|
||||
uint16_t GuestRIPOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Tells this CPUBackend to compile code for the provided IR and DebugData
|
||||
*
|
||||
|
||||
@@ -192,7 +192,6 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
|
||||
@@ -28,6 +28,7 @@ $end_info$
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include "Utils/variable_length_integer.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -144,24 +145,27 @@ uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* T
|
||||
auto [InlineHeader, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
|
||||
if (InlineHeader) {
|
||||
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries*>(
|
||||
Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
// Check if the host PC is currently within a code block.
|
||||
// If it is then RIP can be reconstructed from the beginning of the code block.
|
||||
// This is currently as close as FEX can get RIP reconstructions.
|
||||
if (HostPC >= reinterpret_cast<uint64_t>(BlockBegin) && HostPC < reinterpret_cast<uint64_t>(BlockBegin + InlineTail->Size)) {
|
||||
|
||||
auto RIPEntry =
|
||||
reinterpret_cast<const uint8_t*>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
// Reconstruct RIP from JIT entries for this block.
|
||||
uint64_t StartingHostPC = BlockBegin;
|
||||
uint64_t StartingGuestRIP = InlineTail->RIP;
|
||||
|
||||
for (uint32_t i = 0; i < InlineTail->NumberOfRIPEntries; ++i) {
|
||||
const auto& RIPEntry = RIPEntries[i];
|
||||
if (HostPC >= (StartingHostPC + RIPEntry.HostPCOffset)) {
|
||||
auto HostPCOffset = FEXCore::Utils::vl64::Decode(RIPEntry);
|
||||
RIPEntry += HostPCOffset.Size;
|
||||
auto GuestRIPOffset = FEXCore::Utils::vl64::Decode(RIPEntry);
|
||||
RIPEntry += GuestRIPOffset.Size;
|
||||
if (HostPC >= (StartingHostPC + HostPCOffset.Integer)) {
|
||||
// We are beyond this entry, keep going forward.
|
||||
StartingHostPC += RIPEntry.HostPCOffset;
|
||||
StartingGuestRIP += RIPEntry.GuestRIPOffset;
|
||||
StartingHostPC += HostPCOffset.Integer;
|
||||
StartingGuestRIP += GuestRIPOffset.Integer;
|
||||
} else {
|
||||
// Passed where the Host PC is at. Break now.
|
||||
break;
|
||||
@@ -518,40 +522,6 @@ static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter*
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
};
|
||||
|
||||
// IRStorageBase with fully owned memory
|
||||
struct IRListCopy : public IR::IRStorageBase {
|
||||
std::span<std::byte> IRData;
|
||||
std::span<std::byte> ListData;
|
||||
|
||||
// TODO: Consider defaulting to empty RAData instead?
|
||||
IR::RegisterAllocationData::UniquePtr RADataInternal;
|
||||
|
||||
IRListCopy(const IR::IRListView& view, IR::RegisterAllocationData::UniquePtr RAData)
|
||||
: RADataInternal(std::move(RAData)) {
|
||||
std::byte* Storage = reinterpret_cast<std::byte*>(FEXCore::Allocator::malloc(view.GetDataSize() + view.GetListSize()));
|
||||
|
||||
IRData = {Storage, Storage + view.GetDataSize()};
|
||||
ListData = {Storage + view.GetDataSize(), Storage + view.GetDataSize() + view.GetListSize()};
|
||||
memcpy(IRData.data(), (char*)view.GetData(), IRData.size());
|
||||
memcpy(ListData.data(), (char*)view.GetListData(), ListData.size());
|
||||
}
|
||||
|
||||
IRListCopy(const IRListCopy& other) = delete;
|
||||
IRListCopy(IRListCopy&& other) = delete;
|
||||
|
||||
~IRListCopy() {
|
||||
FEXCore::Allocator::free(IRData.data());
|
||||
}
|
||||
|
||||
const IR::RegisterAllocationData* RAData() override {
|
||||
return RADataInternal.get();
|
||||
}
|
||||
IR::IRListView GetIRView() override {
|
||||
return IR::IRListView {IRData.data(), ListData.data(), IRData.size(), ListData.size()};
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
ContextImpl::GenerateIRResult
|
||||
ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
@@ -700,7 +670,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return {nullptr, 0, 0, 0, 0};
|
||||
return {{}, nullptr, 0, 0, 0, 0};
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
@@ -732,19 +702,16 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(IREmitter);
|
||||
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr;
|
||||
|
||||
// Debug
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP,
|
||||
Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
IRDumper(Thread, IREmitter, GuestRIP, RAData);
|
||||
}
|
||||
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->PullAllocationData() : nullptr;
|
||||
auto IRList = fextl::make_unique<IRListCopy>(IREmitter->ViewIR(), std::move(RAData));
|
||||
|
||||
IREmitter->DelayedDisownBuffer();
|
||||
|
||||
return {
|
||||
.IR = std::move(IRList),
|
||||
.IRView = IREmitter->ViewIR(),
|
||||
.RAData = RAData,
|
||||
.TotalInstructions = TotalInstructions,
|
||||
.TotalInstructionsLength = TotalInstructionsLength,
|
||||
.StartAddr = Thread->FrontendDecoder->DecodedMinAddress,
|
||||
@@ -761,9 +728,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
if (CompiledCode) {
|
||||
return {
|
||||
.CompiledCode = CompiledCode,
|
||||
.IR = nullptr, // No IR/RA data generated
|
||||
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
|
||||
.GeneratedIR = false, // nullptr here ensures IR cache mechanisms won't run
|
||||
.StartAddr = 0, // Unused
|
||||
.Length = 0, // Unused
|
||||
};
|
||||
@@ -778,53 +743,30 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
uint64_t TotalInstructions {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
auto IRFromAOT = IRCaptureCache.PreGenerateIRFetch(Thread, GuestRIP);
|
||||
if (IRFromAOT) {
|
||||
// Setup pointers to internal structures
|
||||
IR = std::move(IRFromAOT->IR);
|
||||
DebugData = IRFromAOT->DebugData;
|
||||
StartAddr = IRFromAOT->StartAddr;
|
||||
Length = IRFromAOT->Length;
|
||||
}
|
||||
}
|
||||
|
||||
if (!IR) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, _TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IR = std::move(IRCopy);
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
TotalInstructions = _TotalInstructions;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
}
|
||||
|
||||
if (!IR) {
|
||||
return {};
|
||||
// Generate IR + Meta Info
|
||||
auto [IRView, RAData, TotalInstructions, TotalInstructionsLength, StartAddr, Length] =
|
||||
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
if (!IRView) {
|
||||
return {nullptr, nullptr, 0, 0};
|
||||
}
|
||||
auto DebugData = fextl::make_unique<FEXCore::Core::DebugData>();
|
||||
|
||||
// If the trap flag is set we generate single instruction blocks that each check to generate a single step exception.
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
auto IRView = IR->GetIRView();
|
||||
|
||||
auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), RAData, TFSet);
|
||||
|
||||
// Release the IR
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
|
||||
return {
|
||||
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
|
||||
// In the future with code caching getting wired up, we will pass the rest of the data forward.
|
||||
// TODO: Pass the data forward when code caching is wired up to this.
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &IRView, DebugData, IR->RAData(), TFSet).BlockEntry,
|
||||
.IR = std::move(IR),
|
||||
.DebugData = DebugData,
|
||||
.GeneratedIR = true,
|
||||
.CompiledCode = CompiledCode.BlockEntry,
|
||||
.DebugData = std::move(DebugData),
|
||||
.StartAddr = StartAddr,
|
||||
.Length = Length,
|
||||
};
|
||||
@@ -843,7 +785,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
auto [CodePtr, IR, DebugData, GeneratedIR, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
@@ -894,7 +836,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, std::move(IR), DebugData, GeneratedIR)) {
|
||||
if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) {
|
||||
// Early exit
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
@@ -913,7 +855,7 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
auto [CodePtr, IR, DebugData, GeneratedIR, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
|
||||
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
@@ -999,11 +941,11 @@ ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandl
|
||||
}
|
||||
|
||||
void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) {
|
||||
LOGMAN_THROW_AA_FMT(Entrypoint, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_AA_FMT(GuestThunkEntrypoint, "Tried to link address to null pointer guest function");
|
||||
LOGMAN_THROW_A_FMT(Entrypoint, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_A_FMT(GuestThunkEntrypoint, "Tried to link address to null pointer guest function");
|
||||
if (!Config.Is64BitMode) {
|
||||
LOGMAN_THROW_AA_FMT((Entrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_AA_FMT((GuestThunkEntrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_A_FMT((Entrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_A_FMT((GuestThunkEntrypoint >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}", Entrypoint, GuestThunkEntrypoint);
|
||||
|
||||
@@ -63,7 +63,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
// }
|
||||
|
||||
ARMEmitter::ForwardLabel l_CTX;
|
||||
ARMEmitter::SingleUseForwardLabel l_Sleep;
|
||||
ARMEmitter::ForwardLabel l_Sleep;
|
||||
ARMEmitter::ForwardLabel l_CompileBlock;
|
||||
ARMEmitter::ForwardLabel l_CompileSingleStep;
|
||||
|
||||
@@ -274,7 +274,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
// Clobbers TMP1/2
|
||||
auto EmitECExitCheck = [&]() {
|
||||
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
|
||||
ARMEmitter::SingleUseForwardLabel l_NotECCode;
|
||||
ARMEmitter::ForwardLabel l_NotECCode;
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
|
||||
|
||||
|
||||
@@ -75,7 +75,7 @@ Decoder::~Decoder() {
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
uint8_t Byte = InstStream[InstructionSize];
|
||||
LOGMAN_THROW_AA_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
Instruction[InstructionSize] = Byte;
|
||||
InstructionSize++;
|
||||
return Byte;
|
||||
@@ -87,7 +87,7 @@ uint8_t Decoder::PeekByte(uint8_t Offset) const {
|
||||
}
|
||||
|
||||
uint64_t Decoder::ReadData(uint8_t Size) {
|
||||
LOGMAN_THROW_AA_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
LOGMAN_THROW_A_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
|
||||
uint64_t Res = 0;
|
||||
std::memcpy(&Res, &InstStream[InstructionSize], Size);
|
||||
@@ -235,7 +235,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
Operand->Data.SIB.Base = MapModRMToReg(BaseREX, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
LOGMAN_THROW_A_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
if (Displacement) {
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
@@ -282,10 +282,10 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P), "Group Ops "
|
||||
"should have "
|
||||
"been decoded "
|
||||
"before this!");
|
||||
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P), "Group Ops "
|
||||
"should have "
|
||||
"been decoded "
|
||||
"before this!");
|
||||
|
||||
uint8_t DestSize {};
|
||||
const bool HasWideningDisplacement =
|
||||
@@ -404,7 +404,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
|
||||
CurrentDest = &DecodeInst->Src[0];
|
||||
} else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_AA_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
// This also means that the destination is always a GPR on these ones
|
||||
@@ -522,7 +522,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_AA_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
|
||||
@@ -545,8 +545,8 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
|
||||
DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
|
||||
DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return true;
|
||||
}
|
||||
@@ -563,7 +563,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
|
||||
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
|
||||
|
||||
// A normal instruction is the most likely.
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INST) [[likely]] {
|
||||
@@ -613,7 +613,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
255, 0, 1, 2, 255, 255, 255, 3,
|
||||
};
|
||||
uint8_t Field = RegToField[ModRM.reg];
|
||||
LOGMAN_THROW_AA_FMT(Field != 255, "Invalid field selected!");
|
||||
LOGMAN_THROW_A_FMT(Field != 255, "Invalid field selected!");
|
||||
|
||||
LocalOp = (Field << 3) | ModRM.rm;
|
||||
return NormalOp(&SecondModRMTableOps[LocalOp], LocalOp);
|
||||
@@ -929,6 +929,7 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
uint64_t TargetRIP = 0;
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
bool Conditional = true;
|
||||
const auto InstEnd = DecodeInst->PC + DecodeInst->InstSize;
|
||||
|
||||
switch (DecodeInst->OP) {
|
||||
case 0x70 ... 0x7F: // Conditional JUMP
|
||||
@@ -937,17 +938,17 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
// auto RIPOffset = LoadSource(Op, Op->Src[0], Op->Flags);
|
||||
// auto RIPTargetConst = _Constant(Op->PC + Op->InstSize);
|
||||
// Target offset is PC + InstSize + Literal
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Literal();
|
||||
TargetRIP = InstEnd + DecodeInst->Src[0].Literal();
|
||||
break;
|
||||
}
|
||||
case 0xE9:
|
||||
case 0xEB: // Both are unconditional JMP instructions
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Literal();
|
||||
TargetRIP = InstEnd + DecodeInst->Src[0].Literal();
|
||||
Conditional = false;
|
||||
break;
|
||||
case 0xE8: // Call - Immediate target, We don't want to inline calls
|
||||
if (ExternalBranches) {
|
||||
ExternalBranches->insert(DecodeInst->PC + DecodeInst->InstSize);
|
||||
ExternalBranches->insert(InstEnd);
|
||||
}
|
||||
[[fallthrough]];
|
||||
case 0xC2: // RET imm
|
||||
@@ -961,7 +962,9 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
}
|
||||
|
||||
// If the target RIP is x86 code within the symbol ranges then we are golden
|
||||
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < SymbolMaxAddress;
|
||||
// Forbid cross-page branches to both avoid massive (range-wise) code blocks in highly fragmented code and trying to decode unmapped branch targets
|
||||
bool ValidMultiblockMember =
|
||||
TargetRIP >= SymbolMinAddress && TargetRIP < std::min(FEXCore::AlignUp(InstEnd, FEXCore::Utils::FEX_PAGE_SIZE), SymbolMaxAddress);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
|
||||
@@ -974,15 +977,10 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
MaxCondBranchBackwards = std::min(MaxCondBranchBackwards, TargetRIP);
|
||||
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
uint64_t FallthroughRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
if (!HasBlocks.contains(FallthroughRIP)) {
|
||||
CurrentBlockTargets.insert(FallthroughRIP);
|
||||
}
|
||||
AddBranchTarget(InstEnd);
|
||||
}
|
||||
|
||||
if (!HasBlocks.contains(TargetRIP)) {
|
||||
CurrentBlockTargets.insert(TargetRIP);
|
||||
}
|
||||
AddBranchTarget(TargetRIP);
|
||||
} else {
|
||||
if (ExternalBranches) {
|
||||
ExternalBranches->insert(TargetRIP);
|
||||
@@ -990,11 +988,15 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::BranchTargetCanContinue(bool FinalInstruction) const {
|
||||
if (FinalInstruction) {
|
||||
bool Decoder::InstCanContinue() const {
|
||||
if (DecodeInst->PC + DecodeInst->InstSize == NextBlockStartAddress) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!(DecodeInst->TableInfo->Flags & (FEXCore::X86Tables::InstFlags::FLAGS_BLOCK_END | FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP))) {
|
||||
return true;
|
||||
}
|
||||
|
||||
uint64_t TargetRIP = 0;
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
|
||||
@@ -1018,6 +1020,59 @@ bool Decoder::BranchTargetCanContinue(bool FinalInstruction) const {
|
||||
return false;
|
||||
}
|
||||
|
||||
void Decoder::AddBranchTarget(uint64_t Target) {
|
||||
if (VisitedBlocks.contains(Target)) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), Target,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != Target, "unexpected");
|
||||
|
||||
if (BlockSuccIt != BlockInfo.Blocks.begin()) {
|
||||
auto BlockIt = std::prev(BlockSuccIt);
|
||||
if (BlockIt->Entry + BlockIt->Size > Target) {
|
||||
uint64_t SplitIdx = 0;
|
||||
uint64_t SplitAddr = BlockIt->Entry;
|
||||
// Find the instruction boundary of the split
|
||||
for (; SplitIdx < BlockIt->NumInstructions && SplitAddr < Target; SplitIdx++) {
|
||||
SplitAddr += BlockIt->DecodedInstructions[SplitIdx].InstSize;
|
||||
}
|
||||
uint64_t SplitOffset = SplitAddr - BlockIt->Entry;
|
||||
|
||||
LOGMAN_THROW_A_FMT(SplitIdx != 0, "unexpected");
|
||||
|
||||
if (SplitAddr == Target) {
|
||||
// Split at the boundary
|
||||
DecodedBlocks SplitBlock {
|
||||
.Entry = SplitAddr,
|
||||
.Size = BlockIt->Size - SplitOffset,
|
||||
.NumInstructions = BlockIt->NumInstructions - SplitIdx,
|
||||
.DecodedInstructions = BlockIt->DecodedInstructions + SplitIdx,
|
||||
.HasInvalidInstruction = BlockIt->HasInvalidInstruction,
|
||||
};
|
||||
|
||||
BlockIt->Size = SplitOffset;
|
||||
BlockIt->NumInstructions = SplitIdx;
|
||||
|
||||
BlockInfo.Blocks.insert(BlockSuccIt, SplitBlock);
|
||||
} // else misaligned, leave as a branch out of the block
|
||||
|
||||
// If we split a block then the target has already been visited as part of that, if it was
|
||||
// misaligned the jump will just leave the multiblock, mark it as visited to avoid running
|
||||
// this code path again and just bail out early.
|
||||
VisitedBlocks.insert(Target);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
CurrentBlockTargets.insert(Target);
|
||||
if (Target >= DecodeInst->PC + DecodeInst->InstSize && Target < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = Target;
|
||||
}
|
||||
}
|
||||
|
||||
const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
|
||||
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
|
||||
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
|
||||
@@ -1041,7 +1096,7 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
VisitedBlocks.clear();
|
||||
// Reset internal state management
|
||||
DecodedSize = 0;
|
||||
MaxCondBranchForward = 0;
|
||||
@@ -1079,30 +1134,61 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
}
|
||||
|
||||
bool EntryBlock {true};
|
||||
bool FinalInstruction {false};
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
while (!FinalInstruction && !BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
BlockInfo.Blocks.emplace_back();
|
||||
DecodedBlocks& CurrentBlockDecoding = BlockInfo.Blocks.back();
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
VisitedBlocks.emplace(RIPToDecode);
|
||||
|
||||
CurrentBlockDecoding.Entry = RIPToDecode;
|
||||
auto BlockSuccIt = std::lower_bound(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(), RIPToDecode,
|
||||
[](const auto& a, uint64_t Address) { return a.Entry < Address; });
|
||||
|
||||
LOGMAN_THROW_A_FMT(BlockSuccIt == BlockInfo.Blocks.end() || BlockSuccIt->Entry != RIPToDecode, "unexpected");
|
||||
|
||||
NextBlockStartAddress = ~0ULL;
|
||||
if (!BlocksToDecode.empty()) {
|
||||
// We just erased the lowest, the front is then the second lowest
|
||||
NextBlockStartAddress = *BlocksToDecode.begin();
|
||||
}
|
||||
if (BlockSuccIt != BlockInfo.Blocks.end() && BlockSuccIt->Entry < NextBlockStartAddress) {
|
||||
NextBlockStartAddress = BlockSuccIt->Entry;
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(NextBlockStartAddress > RIPToDecode, "unexpected");
|
||||
|
||||
// Insert the block now so it can be looked up and split if necessary on a backward edge
|
||||
auto BlockIt = BlockInfo.Blocks.emplace(BlockSuccIt);
|
||||
|
||||
BlockIt->Entry = RIPToDecode;
|
||||
BlockIt->Size = 0;
|
||||
|
||||
uint64_t PCOffset = 0;
|
||||
uint64_t BlockNumberOfInstructions {};
|
||||
uint64_t BlockStartOffset = DecodedSize;
|
||||
bool EraseBlock = true; // Unset once the block contains an instruction
|
||||
|
||||
BlockIt->DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockIt->NumInstructions = 0;
|
||||
|
||||
// Do a bit of pointer math to figure out where we are in code
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
|
||||
while (1) {
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpMinAddress = RIPToDecode + PCOffset;
|
||||
auto OpMaxAddress = OpMinAddress + MAX_INST_SIZE;
|
||||
InstructionSize = 0;
|
||||
|
||||
auto OpMinPage = OpMinAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpAddress = RIPToDecode + PCOffset;
|
||||
auto OpMaxAddress = OpAddress + MAX_INST_SIZE;
|
||||
|
||||
auto OpMinPage = OpAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
auto OpMaxPage = OpMaxAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
|
||||
if (!EntryBlock && OpMinPage == OpMaxPage && PeekByte(0) == 0 && PeekByte(1) == 0) [[unlikely]] {
|
||||
// End the multiblock early if we hit 2 consecutive null bytes (add [rax], al) in the same page with the
|
||||
// assumption we are most likely trying to explore garbage code.
|
||||
break;
|
||||
}
|
||||
|
||||
if (OpMinPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMinPage;
|
||||
CodePages.insert(CurrentCodePage);
|
||||
@@ -1113,64 +1199,66 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
CodePages.insert(CurrentCodePage);
|
||||
}
|
||||
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(OpAddress);
|
||||
uint64_t OpEndAddress = OpAddress + DecodeInst->InstSize;
|
||||
|
||||
if (ErrorDuringDecoding) [[unlikely]] {
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
BlockIt->HasInvalidInstruction = true;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
DecodeInst->InstSize = 0;
|
||||
}
|
||||
|
||||
if (!ErrorDuringDecoding) {
|
||||
} else {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
auto TableInfo = DecodedBuffer[BlockStartOffset + BlockNumberOfInstructions].TableInfo;
|
||||
auto TableInfo = DecodeInst->TableInfo;
|
||||
if (!TableInfo || !TableInfo->OpcodeDispatcher) {
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
BlockIt->HasInvalidInstruction = true;
|
||||
}
|
||||
}
|
||||
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, RIPToDecode + PCOffset);
|
||||
DecodedMaxAddress = std::max(DecodedMaxAddress, RIPToDecode + PCOffset + DecodeInst->InstSize);
|
||||
DecodedMinAddress = std::min(DecodedMinAddress, OpAddress);
|
||||
DecodedMaxAddress = std::max(DecodedMaxAddress, OpEndAddress);
|
||||
|
||||
if (OpEndAddress > NextBlockStartAddress) {
|
||||
// This instruction would overlap with another so skip adding it to the multiblock
|
||||
break;
|
||||
}
|
||||
|
||||
EraseBlock = false; // Block contains at least one valid instruction, so unset erase
|
||||
++TotalInstructions;
|
||||
++BlockNumberOfInstructions;
|
||||
++DecodedSize;
|
||||
++BlockIt->NumInstructions;
|
||||
BlockIt->Size += DecodeInst->InstSize;
|
||||
|
||||
// Can not continue this block at all on invalid instruction
|
||||
if (CurrentBlockDecoding.HasInvalidInstruction) [[unlikely]] {
|
||||
if (BlockIt->HasInvalidInstruction) [[unlikely]] {
|
||||
if (!EntryBlock) {
|
||||
// In multiblock configurations, we can early terminate any non-entrypoint blocks with the expectation that this won't get hit.
|
||||
// Improves compile-times.
|
||||
// Just need to undo additions that this block decoding has caused.
|
||||
TotalInstructions -= CurrentBlockDecoding.NumInstructions;
|
||||
TotalInstructions -= BlockIt->NumInstructions;
|
||||
DecodedSize = BlockStartOffset;
|
||||
BlockNumberOfInstructions = 0;
|
||||
InstStream -= PCOffset;
|
||||
CurrentBlockTargets.clear();
|
||||
EraseBlock = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
bool CanContinue = false;
|
||||
if (!(DecodeInst->TableInfo->Flags & (FEXCore::X86Tables::InstFlags::FLAGS_BLOCK_END | FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP))) {
|
||||
// If this isn't a block ender then we can keep going regardless
|
||||
CanContinue = true;
|
||||
// Check if we need to end the entire multiblock
|
||||
FinalInstruction = DecodedSize >= MaxInst || DecodedSize >= DefaultDecodedBufferSize || TotalInstructions >= MaxInst;
|
||||
if (FinalInstruction) {
|
||||
break;
|
||||
}
|
||||
|
||||
bool FinalInstruction = DecodedSize >= MaxInst || DecodedSize >= DefaultDecodedBufferSize || TotalInstructions >= MaxInst;
|
||||
if (!InstCanContinue()) {
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP) {
|
||||
// If we have multiblock enabled
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
|
||||
BranchTargetInMultiblockRange();
|
||||
}
|
||||
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP) {
|
||||
// If we have multiblock enabled
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
BranchTargetInMultiblockRange();
|
||||
|
||||
// Bypass branches if we can continue through them in some cases.
|
||||
CanContinue |= BranchTargetCanContinue(FinalInstruction);
|
||||
}
|
||||
|
||||
if (FinalInstruction || !CanContinue) {
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1178,29 +1266,22 @@ void Decoder::DecodeInstructionsAtEntry(const uint8_t* _InstStream, uint64_t PC,
|
||||
InstStream += DecodeInst->InstSize;
|
||||
}
|
||||
|
||||
BlocksToDecode.merge(CurrentBlockTargets);
|
||||
// NOTE: BlockIt is only valid here in the EraseBlock case
|
||||
if (EraseBlock) {
|
||||
BlockInfo.Blocks.erase(BlockIt);
|
||||
} else {
|
||||
BlocksToDecode.merge(CurrentBlockTargets);
|
||||
}
|
||||
|
||||
CurrentBlockTargets.clear();
|
||||
|
||||
BlocksToDecode.erase(BlockDecodeIt);
|
||||
HasBlocks.emplace(RIPToDecode);
|
||||
|
||||
// Copy over only the number of instructions we decoded
|
||||
CurrentBlockDecoding.NumInstructions = BlockNumberOfInstructions;
|
||||
CurrentBlockDecoding.DecodedInstructions = &DecodedBuffer[BlockStartOffset];
|
||||
BlockInfo.TotalInstructionCount += BlockNumberOfInstructions;
|
||||
|
||||
EntryBlock = false;
|
||||
}
|
||||
|
||||
BlockInfo.TotalInstructionCount = TotalInstructions;
|
||||
|
||||
for (auto CodePage : CodePages) {
|
||||
AddContainedCodePage(PC, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
|
||||
// sort for better branching
|
||||
std::sort(BlockInfo.Blocks.begin(), BlockInfo.Blocks.end(),
|
||||
[](const FEXCore::Frontend::Decoder::DecodedBlocks& a, const FEXCore::Frontend::Decoder::DecodedBlocks& b) {
|
||||
return a.Entry < b.Entry;
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -22,6 +22,7 @@ public:
|
||||
// New Frontend decoding
|
||||
struct DecodedBlocks final {
|
||||
uint64_t Entry {};
|
||||
uint64_t Size {};
|
||||
uint64_t NumInstructions {};
|
||||
FEXCore::X86Tables::DecodedInst* DecodedInstructions;
|
||||
bool HasInvalidInstruction {};
|
||||
@@ -70,7 +71,9 @@ private:
|
||||
bool DecodeInstruction(uint64_t PC);
|
||||
|
||||
void BranchTargetInMultiblockRange();
|
||||
bool BranchTargetCanContinue(bool FinalInstruction) const;
|
||||
bool InstCanContinue() const;
|
||||
|
||||
void AddBranchTarget(uint64_t Target);
|
||||
|
||||
uint8_t ReadByte();
|
||||
uint8_t PeekByte(uint8_t Offset) const;
|
||||
@@ -102,11 +105,12 @@ private:
|
||||
uint64_t SymbolMaxAddress {};
|
||||
uint64_t SymbolMinAddress {~0ULL};
|
||||
uint64_t SectionMaxAddress {~0ULL};
|
||||
uint64_t NextBlockStartAddress {~0ULL};
|
||||
|
||||
DecodedBlockInformation BlockInfo;
|
||||
fextl::set<uint64_t> CurrentBlockTargets;
|
||||
fextl::set<uint64_t> BlocksToDecode;
|
||||
fextl::set<uint64_t> HasBlocks;
|
||||
fextl::set<uint64_t> VisitedBlocks;
|
||||
fextl::set<uint64_t>* ExternalBranches {nullptr};
|
||||
|
||||
// ModRM rm decoding
|
||||
|
||||
@@ -92,7 +92,7 @@ DEF_OP(AddNZCV) {
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
cmn(EmitSize, Src1, Const);
|
||||
} else if (IROp->Size < IR::OpSize::i32Bit) {
|
||||
unsigned Shift = 32 - IR::OpSizeAsBits(IROp->Size);
|
||||
@@ -193,7 +193,7 @@ DEF_OP(TestNZ) {
|
||||
|
||||
DEF_OP(TestZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestZ>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size < IR::OpSize::i32Bit, "TestNZ used at higher sizes");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size < IR::OpSize::i32Bit, "TestNZ used at higher sizes");
|
||||
const auto EmitSize = ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
@@ -202,7 +202,7 @@ DEF_OP(TestZ) {
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
// We can promote 8/16-bit tests to 32-bit since the constant is masked.
|
||||
LOGMAN_THROW_AA_FMT(!(Const & ~Mask), "constant is already masked");
|
||||
LOGMAN_THROW_A_FMT(!(Const & ~Mask), "constant is already masked");
|
||||
tst(EmitSize, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
@@ -228,7 +228,7 @@ DEF_OP(SubNZCV) {
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
LOGMAN_THROW_A_FMT(OpSize >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
cmp(EmitSize, GetReg(Op->Src1.ID()), Const);
|
||||
} else {
|
||||
unsigned Shift = OpSize < IR::OpSize::i32Bit ? (32 - IR::OpSizeAsBits(OpSize)) : 0;
|
||||
@@ -287,7 +287,7 @@ DEF_OP(SetSmallNZV) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
|
||||
const auto OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i8Bit || OpSize == IR::OpSize::i16Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i8Bit || OpSize == IR::OpSize::i16Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
setf8(GetReg(Op->Src.ID()).W());
|
||||
@@ -516,7 +516,7 @@ DEF_OP(MulH) {
|
||||
auto Op = IROp->C<IR::IROp_MulH>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
@@ -536,7 +536,7 @@ DEF_OP(UMulH) {
|
||||
auto Op = IROp->C<IR::IROp_UMulH>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
@@ -692,7 +692,7 @@ DEF_OP(ShiftFlags) {
|
||||
// updates for Src2=0 but anything that masks to zero.
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, TMP1, &Done);
|
||||
{
|
||||
// PF/SF/ZF/OF
|
||||
@@ -773,7 +773,7 @@ DEF_OP(RotateFlags) {
|
||||
const auto EmitSize = Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
@@ -862,7 +862,7 @@ DEF_OP(PDep) {
|
||||
const auto T1 = TMP4.R();
|
||||
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
|
||||
// First, copy the input/mask, since we'll be clobbering. Copy as 64-bit to
|
||||
// make this 0-uop on Firestorm.
|
||||
@@ -922,9 +922,9 @@ DEF_OP(PExt) {
|
||||
const auto BitReg = TMP2;
|
||||
const auto ValueReg = TMP3;
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel EarlyExit;
|
||||
ARMEmitter::ForwardLabel EarlyExit;
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
|
||||
cbz(EmitSize, Mask, &EarlyExit);
|
||||
mov(EmitSize, MaskReg, Mask);
|
||||
@@ -979,8 +979,8 @@ DEF_OP(LDiv) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
ARMEmitter::ForwardLabel Only64Bit {};
|
||||
ARMEmitter::ForwardLabel LongDIVRet {};
|
||||
|
||||
// Check if the upper bits match the top bit of the lower 64-bits
|
||||
// Sign extend the top bit of lower bits
|
||||
@@ -1047,8 +1047,8 @@ DEF_OP(LUDiv) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
ARMEmitter::ForwardLabel Only64Bit {};
|
||||
ARMEmitter::ForwardLabel LongDIVRet {};
|
||||
|
||||
// Check the upper bits for zero
|
||||
// If the upper bits are zero then we can do a 64-bit divide
|
||||
@@ -1115,8 +1115,8 @@ DEF_OP(LRem) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
ARMEmitter::ForwardLabel Only64Bit {};
|
||||
ARMEmitter::ForwardLabel LongDIVRet {};
|
||||
|
||||
// Check if the upper bits match the top bit of the lower 64-bits
|
||||
// Sign extend the top bit of lower bits
|
||||
@@ -1187,8 +1187,8 @@ DEF_OP(LURem) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
ARMEmitter::ForwardLabel Only64Bit {};
|
||||
ARMEmitter::ForwardLabel LongDIVRet {};
|
||||
|
||||
// Check the upper bits for zero
|
||||
// If the upper bits are zero then we can do a 64-bit divide
|
||||
@@ -1290,8 +1290,8 @@ DEF_OP(FindMSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindMSB>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1313,8 +1313,8 @@ DEF_OP(FindTrailingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_FindTrailingZeroes>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1338,8 +1338,8 @@ DEF_OP(CountLeadingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_CountLeadingZeroes>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1360,8 +1360,8 @@ DEF_OP(Rev) {
|
||||
auto Op = IROp->C<IR::IROp_Rev>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1428,8 +1428,8 @@ DEF_OP(Bfxil) {
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= IR::OpSize::i64Bit, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(Op->Width != 0, "Invalid BFE width of 0");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= IR::OpSize::i64Bit, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_A_FMT(Op->Width != 0, "Invalid BFE width of 0");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1438,7 +1438,7 @@ DEF_OP(Bfe) {
|
||||
if (Op->lsb == 0 && Op->Width == 32) {
|
||||
mov(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
} else if (Op->lsb == 0 && Op->Width == 64) {
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size == IR::OpSize::i64Bit, "Must be 64-bit wide register");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == IR::OpSize::i64Bit, "Must be 64-bit wide register");
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, Src);
|
||||
} else {
|
||||
ubfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
@@ -1549,12 +1549,12 @@ DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
[[maybe_unused]] constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(Op->Header.ElementSize);
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
[[maybe_unused]] const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1576,8 +1576,8 @@ DEF_OP(VExtractToGPR) {
|
||||
// when acting on larger register sizes.
|
||||
PerformMove(Vector, Op->Index);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(Is256Bit, "Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(Offset < AVXRegBitSize, "Trying to extract element outside bounds of register. Offset={}, Index={}", Offset, Op->Index);
|
||||
LOGMAN_THROW_A_FMT(Is256Bit, "Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_A_FMT(Offset < AVXRegBitSize, "Trying to extract element outside bounds of register. Offset={}, Index={}", Offset, Op->Index);
|
||||
|
||||
// We need to use the upper 128-bit lane, so lets move it down.
|
||||
// Inverting our dedicated predicate for 128-bit operations selects
|
||||
|
||||
@@ -86,7 +86,7 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uin
|
||||
size_t DataIndex {};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
|
||||
@@ -13,7 +13,7 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == IR::OpSize::i32Bit || IROp->ElementSize == IR::OpSize::i64Bit, "Wrong element size");
|
||||
LOGMAN_THROW_A_FMT(IROp->ElementSize == IR::OpSize::i32Bit || IROp->ElementSize == IR::OpSize::i64Bit, "Wrong element size");
|
||||
// Size is the size of each pair element
|
||||
auto Dst0 = GetReg(Op->OutLo.ID());
|
||||
auto Dst1 = GetReg(Op->OutHi.ID());
|
||||
@@ -61,8 +61,8 @@ DEF_OP(CASPair) {
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
|
||||
// This instruction sequence must be synced with HandleCASPAL_Armv8.
|
||||
@@ -108,8 +108,8 @@ DEF_OP(CAS) {
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::SingleUseForwardLabel LoopNotExpected;
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (IROp->Size == IR::OpSize::i8Bit) {
|
||||
@@ -274,7 +274,7 @@ DEF_OP(AtomicNeg) {
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
const auto OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(
|
||||
LOGMAN_THROW_A_FMT(
|
||||
OpSize == IR::OpSize::i64Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i8Bit, "Unexpecte"
|
||||
"d CAS "
|
||||
"size");
|
||||
|
||||
@@ -53,14 +53,14 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
#ifdef _M_ARM_64EC
|
||||
if (RtlIsEcCode(NewRIP)) {
|
||||
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
} else {
|
||||
#endif
|
||||
ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
ARMEmitter::ForwardLabel l_BranchHost;
|
||||
ldr(TMP1, &l_BranchHost);
|
||||
blr(TMP1);
|
||||
|
||||
@@ -72,7 +72,7 @@ DEF_OP(ExitFunction) {
|
||||
#endif
|
||||
} else {
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel FullLookup;
|
||||
ARMEmitter::ForwardLabel FullLookup;
|
||||
auto RipReg = GetReg(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
|
||||
@@ -17,14 +17,14 @@ DEF_OP(VAESImc) {
|
||||
|
||||
DEF_OP(VAESEnc) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
const auto OpSize = IROp->Size;
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -42,14 +42,14 @@ DEF_OP(VAESEnc) {
|
||||
|
||||
DEF_OP(VAESEncLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -65,14 +65,14 @@ DEF_OP(VAESEncLast) {
|
||||
|
||||
DEF_OP(VAESDec) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
const auto OpSize = IROp->Size;
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -90,14 +90,14 @@ DEF_OP(VAESDec) {
|
||||
|
||||
DEF_OP(VAESDecLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -187,13 +187,13 @@ DEF_OP(VSha256U0) {
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
const auto OpSize = IROp->Size;
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
|
||||
|
||||
@@ -21,6 +21,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
#include "Utils/variable_length_integer.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
@@ -35,6 +36,7 @@ $end_info$
|
||||
#include <stdio.h>
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
#include <limits>
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
@@ -469,7 +471,7 @@ static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Co
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
uintptr_t branch = (uintptr_t)(Record)-8;
|
||||
ARMEmitter::Emitter emit((uint8_t*)(branch), 8);
|
||||
ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
ARMEmitter::ForwardLabel l_BranchHost;
|
||||
emit.ldr(TMP1, &l_BranchHost);
|
||||
emit.blr(TMP1);
|
||||
emit.Bind(&l_BranchHost);
|
||||
@@ -539,7 +541,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::PREDClass, PredicateRegisters.size());
|
||||
RAPass->PairRegs = PairRegisters;
|
||||
|
||||
{
|
||||
@@ -662,8 +663,8 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
|
||||
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
if (CheckTF) {
|
||||
ARMEmitter::SingleUseForwardLabel l_TFUnset;
|
||||
ARMEmitter::SingleUseForwardLabel l_TFBlocked;
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
ARMEmitter::ForwardLabel l_TFBlocked;
|
||||
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
@@ -713,7 +714,7 @@ void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::SingleUseForwardLabel l_NoSuspend;
|
||||
ARMEmitter::ForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
@@ -799,7 +800,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
using namespace FEXCore::IR;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t*>();
|
||||
@@ -852,10 +853,23 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t*>();
|
||||
auto JITRIPEntries = GetCursorAddress<JITRIPReconstructEntries*>();
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
// Using these entries FEX is able to reconstruct the guest RIP accurately when an instruction cause a signal fault.
|
||||
// Packed using two variable length integer entries to ensure the size isn't too large.
|
||||
// These smaller sizes means that each entry is relative to each other instead of absolute offset from the start of the JIT block.
|
||||
// When reconstructing the RIP, each entry must be walked linearly and accumulated with the previous entries.
|
||||
// This is a trade-off between compression inside the JIT code space and execution time when reconstruction the RIP.
|
||||
// RIP reconstruction when faulting is less likely so we are requiring the accumulation.
|
||||
//
|
||||
// struct {
|
||||
// // The Host PC offset from the previous entry.
|
||||
// FEXCore::Utils::vl64 HostPCOffset;
|
||||
// // How much to offset the RIP from the previous entry.
|
||||
// FEXCore::Utils::vl64 GuestRIPOffset;
|
||||
// };
|
||||
|
||||
CursorIncrement(sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
auto JITRIPEntriesBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
@@ -866,22 +880,34 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
JITBlockTail->SingleInst = SingleInst;
|
||||
JITBlockTail->SpinLockFutex = 0;
|
||||
|
||||
auto JITRIPEntriesLocation = JITRIPEntriesBegin;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto& GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto& RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
int64_t HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
int64_t GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
|
||||
size_t Size = FEXCore::Utils::vl64::Encode(JITRIPEntriesLocation, HostPCOffset);
|
||||
JITRIPEntriesLocation += Size;
|
||||
|
||||
Size = FEXCore::Utils::vl64::Encode(JITRIPEntriesLocation, GuestRIPOffset);
|
||||
JITRIPEntriesLocation += Size;
|
||||
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CursorIncrement(JITRIPEntriesLocation - JITRIPEntriesBegin);
|
||||
Align();
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
@@ -893,7 +919,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
if (Disassemble() & FEXCore::Config::Disassemble::STATS) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
LOGMAN_THROW_AA_FMT(HeaderOp->Header.Op == IR::OP_IRHEADER, "First op wasn't IRHeader");
|
||||
LOGMAN_THROW_A_FMT(HeaderOp->Header.Op == IR::OP_IRHEADER, "First op wasn't IRHeader");
|
||||
|
||||
LogMan::Msg::IFmt("RIP: 0x{:x}", Entry);
|
||||
LogMan::Msg::IFmt("Guest Code instructions: {}", HeaderOp->NumHostInstructions);
|
||||
|
||||
@@ -69,7 +69,7 @@ private:
|
||||
ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return StaticRegisters[Reg.Reg];
|
||||
@@ -84,7 +84,7 @@ private:
|
||||
ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
@@ -95,19 +95,6 @@ private:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::PRegister GetPReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::PREDClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::PREDClass.Val) {
|
||||
return PredicateRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
|
||||
@@ -124,7 +111,7 @@ private:
|
||||
ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Src, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "Only valid constant");
|
||||
return ARMEmitter::Reg::zr;
|
||||
} else {
|
||||
return GetReg(Src.ID());
|
||||
@@ -148,15 +135,15 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSize(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
return ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
@@ -171,7 +158,7 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSize16(ElementSize);
|
||||
}
|
||||
|
||||
@@ -182,13 +169,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
@@ -199,13 +186,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair16(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair8(Op);
|
||||
}
|
||||
|
||||
@@ -245,6 +232,10 @@ private:
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register ApplyMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, ARMEmitter::Register Tmp,
|
||||
IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
// the limits of the scalar plus immediate variant of SVE load/stores.
|
||||
//
|
||||
|
||||
@@ -8,9 +8,9 @@ $end_info$
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
@@ -158,7 +158,7 @@ DEF_OP(LoadRegister) {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -210,7 +210,7 @@ DEF_OP(StoreRegister) {
|
||||
}
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -591,6 +591,44 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, ARMEmitter::Register Tmp,
|
||||
IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
if (Offset.IsInvalid()) {
|
||||
return Base;
|
||||
}
|
||||
|
||||
if (OffsetScale != 1 && OffsetScale != IR::OpSizeToSize(AccessSize)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled OffsetScale: {}", OffsetScale);
|
||||
}
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Offset, &Const)) {
|
||||
if (Const == 0) {
|
||||
return Base;
|
||||
}
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Tmp, Const);
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset.ID());
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType.Val); break;
|
||||
}
|
||||
}
|
||||
return Tmp;
|
||||
}
|
||||
|
||||
ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, [[maybe_unused]] uint8_t OffsetScale) {
|
||||
if (Offset.IsInvalid()) {
|
||||
@@ -861,7 +899,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
PerformMove(IROp->ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::SingleUseForwardLabel Skip {};
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
@@ -953,7 +991,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
PerformMove(IROp->ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::SingleUseForwardLabel Skip {};
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
@@ -1037,7 +1075,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
}
|
||||
|
||||
for (size_t i = DataElementOffsetStart, IndexElement = IndexElementOffsetStart; i < NumDataElements; ++i, ++IndexElement) {
|
||||
ARMEmitter::SingleUseForwardLabel Skip {};
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Extract mask element
|
||||
PerformMove(ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
@@ -1276,10 +1314,10 @@ DEF_OP(VLoadVectorElement) {
|
||||
const auto DstSrc = GetVReg(Op->DstSrc.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
|
||||
if (Is256Bit) {
|
||||
LOGMAN_MSG_A_FMT("Unsupported 256-bit VLoadVectorElement");
|
||||
@@ -1313,10 +1351,10 @@ DEF_OP(VStoreVectorElement) {
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
|
||||
// Emit a half-barrier if TSO is enabled.
|
||||
if (CTX->IsVectorAtomicTSOEnabled()) {
|
||||
@@ -1348,10 +1386,10 @@ DEF_OP(VBroadcastFromMem) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MemReg = GetReg(Op->Address.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid element "
|
||||
"size");
|
||||
|
||||
if (Is256Bit && HostSupportsSVE256) {
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
@@ -1552,21 +1590,14 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(InitPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_InitPredicate>();
|
||||
const auto OpSize = IROp->Size;
|
||||
ptrue(ConvertSubRegSize16(OpSize), GetPReg(Node), static_cast<ARMEmitter::PredicatePattern>(Op->Pattern));
|
||||
}
|
||||
DEF_OP(StoreMemX87SVEOptPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemX87SVEOptPredicate>();
|
||||
const auto Predicate = PRED_X87_SVEOPT;
|
||||
|
||||
DEF_OP(StoreMemPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemPredicate>();
|
||||
const auto Predicate = GetPReg(Op->Mask.ID());
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "StoreMemX87SVEOptPredicate needs SVE support");
|
||||
|
||||
const auto RegData = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "StoreMemPredicate needs SVE support");
|
||||
|
||||
const auto MemDst = ARMEmitter::SVEMemOperand(MemReg.X(), 0);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
@@ -1590,13 +1621,13 @@ DEF_OP(StoreMemPredicate) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMemPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPredicate>();
|
||||
DEF_OP(LoadMemX87SVEOptPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemX87SVEOptPredicate>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Predicate = GetPReg(Op->Mask.ID());
|
||||
const auto Predicate = PRED_X87_SVEOPT;
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "LoadMemPredicate needs SVE support");
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "LoadMemX87SVEOptPredicate needs SVE support");
|
||||
|
||||
const auto MemDst = ARMEmitter::SVEMemOperand(MemReg.X(), 0);
|
||||
|
||||
@@ -1750,8 +1781,8 @@ DEF_OP(MemSet) {
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel BackwardImpl {};
|
||||
ARMEmitter::SingleUseForwardLabel Done {};
|
||||
ARMEmitter::ForwardLabel BackwardImpl {};
|
||||
ARMEmitter::ForwardLabel Done {};
|
||||
|
||||
mov(TMP1, Length.X());
|
||||
if (Op->Prefix.IsInvalid()) {
|
||||
@@ -1788,7 +1819,6 @@ DEF_OP(MemSet) {
|
||||
case 8: stlr(Value.X(), TMP2); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
|
||||
if (Size >= 0) {
|
||||
@@ -1894,7 +1924,7 @@ DEF_OP(MemSet) {
|
||||
};
|
||||
|
||||
if (DirectionIsInline) {
|
||||
LOGMAN_THROW_AA_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
LOGMAN_THROW_A_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
EmitMemset(DirectionConstant);
|
||||
} else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
@@ -1943,8 +1973,8 @@ DEF_OP(MemCpy) {
|
||||
//
|
||||
// Counter is decremented regardless.
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel BackwardImpl {};
|
||||
ARMEmitter::SingleUseForwardLabel Done {};
|
||||
ARMEmitter::ForwardLabel BackwardImpl {};
|
||||
ARMEmitter::ForwardLabel Done {};
|
||||
|
||||
mov(TMP1, Length.X());
|
||||
mov(TMP2, MemRegDest.X());
|
||||
@@ -1993,23 +2023,23 @@ DEF_OP(MemCpy) {
|
||||
ldaprb(TMP4.W(), TMP3);
|
||||
stlrb(TMP4.W(), TMP2);
|
||||
} else {
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2: ldaprh(TMP4.W(), TMP3); break;
|
||||
case 4: ldapr(TMP4.W(), TMP3); break;
|
||||
case 8: ldapr(TMP4, TMP3); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
case 4: stlr(TMP4.W(), TMP2); break;
|
||||
case 8: stlr(TMP4, TMP2); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
} else {
|
||||
if (OpSize == 1) {
|
||||
@@ -2017,23 +2047,23 @@ DEF_OP(MemCpy) {
|
||||
ldarb(TMP4.W(), TMP3);
|
||||
stlrb(TMP4.W(), TMP2);
|
||||
} else {
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2: ldarh(TMP4.W(), TMP3); break;
|
||||
case 4: ldar(TMP4.W(), TMP3); break;
|
||||
case 8: ldar(TMP4, TMP3); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
case 4: stlr(TMP4.W(), TMP2); break;
|
||||
case 8: stlr(TMP4, TMP2); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2171,7 +2201,7 @@ DEF_OP(MemCpy) {
|
||||
};
|
||||
|
||||
if (DirectionIsInline) {
|
||||
LOGMAN_THROW_AA_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
LOGMAN_THROW_A_FMT(DirectionConstant == 1 || DirectionConstant == -1, "unexpected direction");
|
||||
EmitMemcpy(DirectionConstant);
|
||||
} else {
|
||||
// Emit forward direction memset then backward direction memset.
|
||||
@@ -2191,13 +2221,15 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -2214,6 +2246,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
@@ -2227,6 +2260,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: ldarb(Dst, MemReg); break;
|
||||
case IR::OpSize::i16Bit: ldarh(Dst, MemReg); break;
|
||||
@@ -2236,6 +2270,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
}
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
ldarb(TMP1, MemReg);
|
||||
@@ -2274,13 +2309,15 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -2296,6 +2333,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
|
||||
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
|
||||
@@ -2306,6 +2344,8 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
|
||||
|
||||
@@ -148,7 +148,7 @@ DEF_OP(PushRoundingMode) {
|
||||
} else if (Op->RoundMode == 0) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(3 << 22));
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
|
||||
LOGMAN_THROW_A_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(Op->RoundMode << 22));
|
||||
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, (Op->RoundMode == 2 ? 1 : 2) << 22);
|
||||
|
||||
@@ -265,8 +265,8 @@ void Arm64JITCore::VFScalarFMAOperation(IR::OpSize OpSize, IR::OpSize ElementSiz
|
||||
ARMEmitter::VRegister Addend) {
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "256-bit unsupported", __func__);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit || ElementSize == IR::OpSize::i64Bit, "Invalid"
|
||||
" size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit || ElementSize == IR::OpSize::i64Bit, "Invalid "
|
||||
"size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
@@ -299,8 +299,8 @@ void Arm64JITCore::VFScalarOperation(IR::OpSize OpSize, IR::OpSize ElementSize,
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from Vector1.
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit || ElementSize == IR::OpSize::i64Bit, "Invalid"
|
||||
" size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit || ElementSize == IR::OpSize::i64Bit, "Invalid "
|
||||
"size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
@@ -371,8 +371,8 @@ void Arm64JITCore::VFScalarUnaryOperation(IR::OpSize OpSize, IR::OpSize ElementS
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(Is256Bit || !ZeroUpperBits, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit || ElementSize == IR::OpSize::i64Bit, "Invalid"
|
||||
" size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit || ElementSize == IR::OpSize::i64Bit, "Invalid "
|
||||
"size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
@@ -630,9 +630,9 @@ DEF_OP(VSToFVectorInsert) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto HasTwoElements = Op->HasTwoElements;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i32Bit || ElementSize == IR::OpSize::i64Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i32Bit || ElementSize == IR::OpSize::i64Bit, "Invalid size");
|
||||
if (HasTwoElements) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i32Bit, "Can't have two elements for 8-byte size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i32Bit, "Can't have two elements for 8-byte size");
|
||||
}
|
||||
|
||||
auto ScalarEmit = [this, ElementSize, HasTwoElements](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
@@ -1122,8 +1122,7 @@ DEF_OP(VFAddV) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit || OpSize == IR::OpSize::i256Bit, "Only AVX and SSE size "
|
||||
"supported");
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit || OpSize == IR::OpSize::i256Bit, "Only AVX and SSE size supported");
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
|
||||
@@ -1349,7 +1348,7 @@ DEF_OP(VFMin) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
[[maybe_unused]] const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
@@ -1390,7 +1389,7 @@ DEF_OP(VFMin) {
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(!IsScalar, "should use VFMinScalarInsert instead");
|
||||
LOGMAN_THROW_A_FMT(!IsScalar, "should use VFMinScalarInsert instead");
|
||||
|
||||
if (Dst == Vector1) {
|
||||
// Destination is already Vector1, need to insert Vector2 on false.
|
||||
@@ -1415,7 +1414,7 @@ DEF_OP(VFMax) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
[[maybe_unused]] const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
@@ -1442,7 +1441,7 @@ DEF_OP(VFMax) {
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(!IsScalar, "should use VFMaxScalarInsert instead");
|
||||
LOGMAN_THROW_A_FMT(!IsScalar, "should use VFMaxScalarInsert instead");
|
||||
|
||||
if (Dst == Vector1) {
|
||||
// Destination is already Vector1, need to insert Vector2 on true.
|
||||
@@ -3912,7 +3911,7 @@ DEF_OP(VTBL1) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i256Bit: {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE256, "Host does not support SVE. Cannot perform 256-bit table lookup");
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Host does not support SVE. Cannot perform 256-bit table lookup");
|
||||
|
||||
tbl(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), VectorTable.Z(), VectorIndices.Z());
|
||||
break;
|
||||
@@ -3956,7 +3955,7 @@ DEF_OP(VTBL2) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i256Bit: {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE256, "Host does not support SVE. Cannot perform 256-bit table lookup");
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Host does not support SVE. Cannot perform 256-bit table lookup");
|
||||
|
||||
tbl(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), VectorTable1.Z(), VectorTable2.Z(), VectorIndices.Z());
|
||||
break;
|
||||
@@ -3989,7 +3988,7 @@ DEF_OP(VTBX1) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i256Bit: {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE256, "Host does not support SVE. Cannot perform 256-bit table lookup");
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Host does not support SVE. Cannot perform 256-bit table lookup");
|
||||
mov(VTMP1.Z(), VectorSrcDst.Z());
|
||||
tbx(ARMEmitter::SubRegSize::i8Bit, VTMP1.Z(), VectorTable.Z(), VectorIndices.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
@@ -4008,7 +4007,7 @@ DEF_OP(VTBX1) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i256Bit: {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE256, "Host does not support SVE. Cannot perform 256-bit table lookup");
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Host does not support SVE. Cannot perform 256-bit table lookup");
|
||||
|
||||
tbx(ARMEmitter::SubRegSize::i8Bit, VectorSrcDst.Z(), VectorTable.Z(), VectorIndices.Z());
|
||||
break;
|
||||
@@ -4029,7 +4028,7 @@ DEF_OP(VRev32) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit, "Invalid size");
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit, "Invalid size");
|
||||
const auto SubRegSize = ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
|
||||
@@ -44,11 +44,11 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / 4096 * 8;
|
||||
LOGMAN_THROW_AA_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = PageMemory + CODE_SIZE;
|
||||
LOGMAN_THROW_AA_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
}
|
||||
|
||||
@@ -90,7 +90,7 @@ public:
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_AA_FMT(Inserted, "Duplicate block mapping added");
|
||||
LOGMAN_THROW_A_FMT(Inserted, "Duplicate block mapping added");
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
|
||||
@@ -753,7 +753,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
auto OP = Op->OP & 0xF;
|
||||
auto [Complex, SimpleCond] = DecodeNZCVCondition(OP);
|
||||
if (Complex) {
|
||||
LOGMAN_THROW_AA_FMT(OP == 0xA || OP == 0xB, "only PF left");
|
||||
LOGMAN_THROW_A_FMT(OP == 0xA || OP == 0xB, "only PF left");
|
||||
CondJump_ = CondJumpBit(LoadPFRaw(false, false), 0, OP == 0xB);
|
||||
} else {
|
||||
CondJump_ = CondJumpNZCV(SimpleCond);
|
||||
@@ -3924,7 +3924,7 @@ void OpDispatchBuilder::Finalize() {
|
||||
Ref RealNode = reinterpret_cast<Ref>(GetNode(1));
|
||||
|
||||
[[maybe_unused]] const FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
LOGMAN_THROW_AA_FMT(IROp->Op == OP_IRHEADER, "First op in function must be our header");
|
||||
LOGMAN_THROW_A_FMT(IROp->Op == OP_IRHEADER, "First op in function must be our header");
|
||||
|
||||
// Let's walk the jump blocks and see if we have handled every block target
|
||||
for (auto& Handler : JumpTargets) {
|
||||
@@ -3940,13 +3940,13 @@ void OpDispatchBuilder::Finalize() {
|
||||
|
||||
uint8_t OpDispatchBuilder::GetDstSize(X86Tables::DecodedOp Op) const {
|
||||
const uint32_t DstSizeFlag = X86Tables::DecodeFlags::GetSizeDstFlags(Op->Flags);
|
||||
LOGMAN_THROW_AA_FMT(DstSizeFlag != 0 && DstSizeFlag != X86Tables::DecodeFlags::SIZE_MASK, "Invalid destination size for op");
|
||||
LOGMAN_THROW_A_FMT(DstSizeFlag != 0 && DstSizeFlag != X86Tables::DecodeFlags::SIZE_MASK, "Invalid destination size for op");
|
||||
return 1u << (DstSizeFlag - 1);
|
||||
}
|
||||
|
||||
uint8_t OpDispatchBuilder::GetSrcSize(X86Tables::DecodedOp Op) const {
|
||||
const uint32_t SrcSizeFlag = X86Tables::DecodeFlags::GetSizeSrcFlags(Op->Flags);
|
||||
LOGMAN_THROW_AA_FMT(SrcSizeFlag != 0 && SrcSizeFlag != X86Tables::DecodeFlags::SIZE_MASK, "Invalid destination size for op");
|
||||
LOGMAN_THROW_A_FMT(SrcSizeFlag != 0 && SrcSizeFlag != X86Tables::DecodeFlags::SIZE_MASK, "Invalid destination size for op");
|
||||
return 1u << (SrcSizeFlag - 1);
|
||||
}
|
||||
|
||||
@@ -4137,7 +4137,7 @@ Ref OpDispatchBuilder::LoadEffectiveAddress(AddressMode A, bool AddSegmentBase,
|
||||
|
||||
if (A.Index) {
|
||||
if (A.IndexScale != 1) {
|
||||
LOGMAN_THROW_AA_FMT((A.IndexScale & (A.IndexScale - 1)) == 0, "power of two");
|
||||
LOGMAN_THROW_A_FMT((A.IndexScale & (A.IndexScale - 1)) == 0, "power of two");
|
||||
uint32_t Log2 = FEXCore::ilog2(A.IndexScale);
|
||||
|
||||
if (Tmp) {
|
||||
@@ -4313,9 +4313,7 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemSrc = LoadEffectiveAddress(A, true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
// Using SVE we can load this with a single instruction.
|
||||
auto PReg = InitPredicateCached(OpSize::i16Bit, ARMEmitter::PredicatePattern::SVE_VL5);
|
||||
return _LoadMemPredicate(OpSize::i128Bit, OpSize::i16Bit, PReg, MemSrc);
|
||||
return _LoadMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, MemSrc);
|
||||
} else {
|
||||
// For X87 extended doubles, Split the load.
|
||||
auto Res = _LoadMem(Class, OpSize::i64Bit, MemSrc, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
@@ -4424,9 +4422,9 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
Ref Value = GetOpSize(Src) == OpSize::i64Bit ? _Bfe(OpSize::i32Bit, 32, 0, Src) : Src;
|
||||
StoreGPRRegister(gpr, Value, GPRSize);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
LOGMAN_THROW_A_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(!(GPRSize == OpSize::i32Bit && OpSize > OpSize::i32Bit), "Oops had a {} GPR load", OpSize);
|
||||
LOGMAN_THROW_A_FMT(!(GPRSize == OpSize::i32Bit && OpSize > OpSize::i32Bit), "Oops had a {} GPR load", OpSize);
|
||||
|
||||
if (GPRSize != OpSize) {
|
||||
// if the GPR isn't the full size then we need to insert.
|
||||
@@ -4448,8 +4446,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemStoreDst = LoadEffectiveAddress(A, true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
auto PReg = InitPredicateCached(OpSize::i16Bit, ARMEmitter::PredicatePattern::SVE_VL5);
|
||||
_StoreMemPredicate(OpSize::i128Bit, OpSize::i16Bit, Src, PReg, MemStoreDst);
|
||||
_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, Src, MemStoreDst);
|
||||
} else {
|
||||
// For X87 extended doubles, split before storing
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, MemStoreDst, Src, Align);
|
||||
@@ -4889,12 +4886,13 @@ void OpDispatchBuilder::BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDe
|
||||
_StoreContext(GPRSize, GPRClass, GetRelocatedPC(Op, -Op->InstSize), offsetof(FEXCore::Core::CPUState, rip));
|
||||
Break(BreakDefinition);
|
||||
|
||||
BlockSetRIP = true;
|
||||
|
||||
if (Multiblock) {
|
||||
auto NextBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetCurrentCodeBlock(NextBlock);
|
||||
StartNewBlock();
|
||||
} else {
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -125,9 +125,6 @@ public:
|
||||
|
||||
// Need to clear any named constants that were cached.
|
||||
ClearCachedNamedConstants();
|
||||
|
||||
// Clear predicate cache for x87 ldst
|
||||
ResetInitPredicateCache();
|
||||
}
|
||||
|
||||
IRPair<IROp_Jump> Jump() {
|
||||
@@ -712,32 +709,29 @@ public:
|
||||
RES_STI,
|
||||
};
|
||||
|
||||
void X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2);
|
||||
void FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FNINIT(OpcodeArgs);
|
||||
void FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FTST(OpcodeArgs);
|
||||
void FNINIT(OpcodeArgs);
|
||||
|
||||
void X87ModifySTP(OpcodeArgs, bool Inc);
|
||||
void X87SinCos(OpcodeArgs);
|
||||
void X87FYL2X(OpcodeArgs, bool IsFYL2XP1);
|
||||
void X87LDENV(OpcodeArgs);
|
||||
void FXCH(OpcodeArgs);
|
||||
void X87EMMS(OpcodeArgs);
|
||||
void X87FCMOV(OpcodeArgs);
|
||||
void X87FFREE(OpcodeArgs);
|
||||
void X87FLDCW(OpcodeArgs);
|
||||
void X87FNSTENV(OpcodeArgs);
|
||||
void X87FSTCW(OpcodeArgs);
|
||||
void X87LDSW(OpcodeArgs);
|
||||
void X87FNSTSW(OpcodeArgs);
|
||||
void X87FNSAVE(OpcodeArgs);
|
||||
void X87FNSTENV(OpcodeArgs);
|
||||
void X87FNSTSW(OpcodeArgs);
|
||||
void X87FRSTOR(OpcodeArgs);
|
||||
void X87FSTCW(OpcodeArgs);
|
||||
void X87FXAM(OpcodeArgs);
|
||||
void X87FXTRACT(OpcodeArgs);
|
||||
void X87FCMOV(OpcodeArgs);
|
||||
void X87EMMS(OpcodeArgs);
|
||||
void X87FFREE(OpcodeArgs);
|
||||
|
||||
void FXCH(OpcodeArgs);
|
||||
void X87FYL2X(OpcodeArgs, bool IsFYL2XP1);
|
||||
void X87LDENV(OpcodeArgs);
|
||||
void X87LDSW(OpcodeArgs);
|
||||
void X87ModifySTP(OpcodeArgs, bool Inc);
|
||||
void X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2);
|
||||
|
||||
enum class FCOMIFlags {
|
||||
FLAGS_X87,
|
||||
@@ -746,39 +740,23 @@ public:
|
||||
void FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, FCOMIFlags WhichFlags, bool PopTwice);
|
||||
|
||||
// F64 X87 Ops
|
||||
void FLDF64(OpcodeArgs, IR::OpSize Width);
|
||||
void FLDF64_Const(OpcodeArgs, uint64_t Num);
|
||||
|
||||
void FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FBLDF64(OpcodeArgs);
|
||||
void FBSTPF64(OpcodeArgs);
|
||||
|
||||
void FILDF64(OpcodeArgs);
|
||||
|
||||
void FSTF64(OpcodeArgs, IR::OpSize Width);
|
||||
|
||||
void FISTF64(OpcodeArgs, bool Truncate);
|
||||
|
||||
void FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FCOMIF64(OpcodeArgs, IR::OpSize width, bool Integer, FCOMIFlags whichflags, bool poptwice);
|
||||
void FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FILDF64(OpcodeArgs);
|
||||
void FISTF64(OpcodeArgs, bool Truncate);
|
||||
void FLDF64_Const(OpcodeArgs, uint64_t Num);
|
||||
void FLDF64(OpcodeArgs, IR::OpSize Width);
|
||||
void FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FSTF64(OpcodeArgs, IR::OpSize Width);
|
||||
void FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FCHSF64(OpcodeArgs);
|
||||
void FABSF64(OpcodeArgs);
|
||||
void FTSTF64(OpcodeArgs);
|
||||
void FRNDINTF64(OpcodeArgs);
|
||||
void FSQRTF64(OpcodeArgs);
|
||||
void X87UnaryOpF64(OpcodeArgs, FEXCore::IR::IROps IROp);
|
||||
void X87BinaryOpF64(OpcodeArgs, FEXCore::IR::IROps IROp);
|
||||
void X87SinCosF64(OpcodeArgs);
|
||||
void X87FLDCWF64(OpcodeArgs);
|
||||
void X87TANF64(OpcodeArgs);
|
||||
void X87ATANF64(OpcodeArgs);
|
||||
void X87FXAMF64(OpcodeArgs);
|
||||
void X87FXTRACTF64(OpcodeArgs);
|
||||
void X87LDENVF64(OpcodeArgs);
|
||||
|
||||
void FCOMIF64(OpcodeArgs, IR::OpSize width, bool Integer, FCOMIFlags whichflags, bool poptwice);
|
||||
|
||||
void FXSaveOp(OpcodeArgs);
|
||||
void FXRStoreOp(OpcodeArgs);
|
||||
|
||||
@@ -1551,7 +1529,7 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
static uint32_t GPROffset(X86State::X86Reg reg) {
|
||||
LOGMAN_THROW_AA_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
LOGMAN_THROW_A_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
return static_cast<uint32_t>(offsetof(Core::CPUState, gregs[static_cast<size_t>(reg)]));
|
||||
}
|
||||
|
||||
@@ -1710,7 +1688,7 @@ private:
|
||||
CFInverted ^= true;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(CFInverted == RequiredInvert, "post condition");
|
||||
LOGMAN_THROW_A_FMT(CFInverted == RequiredInvert, "post condition");
|
||||
}
|
||||
|
||||
void CarryInvert() {
|
||||
@@ -1892,7 +1870,7 @@ private:
|
||||
}
|
||||
|
||||
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
|
||||
LOGMAN_THROW_AA_FMT(Index < 64, "valid index");
|
||||
LOGMAN_THROW_A_FMT(Index < 64, "valid index");
|
||||
uint64_t Bit = (1ull << (uint64_t)Index);
|
||||
|
||||
if (Size == OpSize::i128Bit && (RegCache.Partial & Bit)) {
|
||||
@@ -1947,8 +1925,8 @@ private:
|
||||
}
|
||||
|
||||
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
|
||||
LOGMAN_THROW_AA_FMT(Index != DFIndex, "must be pairable");
|
||||
LOGMAN_THROW_AA_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
LOGMAN_THROW_A_FMT(Index != DFIndex, "must be pairable");
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
|
||||
// Try to load a pair into the cache
|
||||
uint64_t Bits = (3ull << (uint64_t)Index);
|
||||
@@ -1996,8 +1974,8 @@ private:
|
||||
}
|
||||
|
||||
void StoreContext(uint8_t Index, Ref Value) {
|
||||
LOGMAN_THROW_AA_FMT(Index < 64, "valid index");
|
||||
LOGMAN_THROW_AA_FMT(Value != InvalidNode, "storing valid");
|
||||
LOGMAN_THROW_A_FMT(Index < 64, "valid index");
|
||||
LOGMAN_THROW_A_FMT(Value != InvalidNode, "storing valid");
|
||||
|
||||
uint64_t Bit = (1ull << (uint64_t)Index);
|
||||
|
||||
@@ -2428,7 +2406,7 @@ private:
|
||||
}
|
||||
|
||||
AddressMode SelectPairAddressMode(AddressMode A, IR::OpSize Size) {
|
||||
LOGMAN_THROW_AA_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
const auto SizeInt = IR::OpSizeToSize(Size);
|
||||
AddressMode Out {};
|
||||
|
||||
|
||||
@@ -486,7 +486,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
|
||||
if (Operand.IsGPR()) {
|
||||
const auto gpr = Operand.Data.GPR.GPR;
|
||||
LOGMAN_THROW_AA_FMT(gpr >= FEXCore::X86State::REG_XMM_0 && gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_A_FMT(gpr >= FEXCore::X86State::REG_XMM_0 && gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
return {
|
||||
.Low = AVX128_LoadXMMRegister(gprIndex, false),
|
||||
@@ -501,8 +501,8 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
HighA.Offset += 16;
|
||||
|
||||
if (Operand.IsSIB()) {
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
LOGMAN_THROW_AA_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
|
||||
[[maybe_unused]] const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
LOGMAN_THROW_A_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
|
||||
}
|
||||
|
||||
if (NeedsHigh) {
|
||||
@@ -523,10 +523,9 @@ OpDispatchBuilder::AVX128_LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tabl
|
||||
|
||||
const auto Index_gpr = Operand.Data.SIB.Index;
|
||||
const auto Base_gpr = Operand.Data.SIB.Base;
|
||||
LOGMAN_THROW_AA_FMT(Index_gpr >= FEXCore::X86State::REG_XMM_0 && Index_gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_AA_FMT(
|
||||
Base_gpr == FEXCore::X86State::REG_INVALID || (Base_gpr >= FEXCore::X86State::REG_RAX && Base_gpr <= FEXCore::X86State::REG_R15),
|
||||
"Base must be a GPR.");
|
||||
LOGMAN_THROW_A_FMT(Index_gpr >= FEXCore::X86State::REG_XMM_0 && Index_gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_A_FMT(Base_gpr == FEXCore::X86State::REG_INVALID || (Base_gpr >= FEXCore::X86State::REG_RAX && Base_gpr <= FEXCore::X86State::REG_R15),
|
||||
"Base must be a GPR.");
|
||||
const auto Index_XMM_gpr = Index_gpr - X86State::REG_XMM_0;
|
||||
|
||||
return {
|
||||
@@ -542,7 +541,7 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
|
||||
const RefPair Src, MemoryAccessType AccessType) {
|
||||
if (Operand.IsGPR()) {
|
||||
const auto gpr = Operand.Data.GPR.GPR;
|
||||
LOGMAN_THROW_AA_FMT(gpr >= FEXCore::X86State::REG_XMM_0 && gpr <= FEXCore::X86State::REG_XMM_15, "expected AVX register");
|
||||
LOGMAN_THROW_A_FMT(gpr >= FEXCore::X86State::REG_XMM_0 && gpr <= FEXCore::X86State::REG_XMM_15, "expected AVX register");
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
|
||||
if (Src.Low) {
|
||||
@@ -1817,7 +1816,7 @@ void OpDispatchBuilder::AVX128_VPERMQ(OpcodeArgs) {
|
||||
uint8_t SelectorLow = Selector & 0b1111;
|
||||
uint8_t SelectorHigh = (Selector >> 4) & 0b1111;
|
||||
auto SelectLane = [this](uint8_t Selector, RefPair Src) -> Ref {
|
||||
LOGMAN_THROW_AA_FMT(Selector < 16, "Selector too large!");
|
||||
LOGMAN_THROW_A_FMT(Selector < 16, "Selector too large!");
|
||||
|
||||
switch (Selector) {
|
||||
case 0b00'00: return _VDupElement(OpSize::i128Bit, OpSize::i64Bit, Src.Low, 0);
|
||||
|
||||
@@ -4955,10 +4955,9 @@ OpDispatchBuilder::RefVSIB OpDispatchBuilder::LoadVSIB(const X86Tables::DecodedO
|
||||
|
||||
const auto Index_gpr = Operand.Data.SIB.Index;
|
||||
const auto Base_gpr = Operand.Data.SIB.Base;
|
||||
LOGMAN_THROW_AA_FMT(Index_gpr >= FEXCore::X86State::REG_XMM_0 && Index_gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_AA_FMT(
|
||||
Base_gpr == FEXCore::X86State::REG_INVALID || (Base_gpr >= FEXCore::X86State::REG_RAX && Base_gpr <= FEXCore::X86State::REG_R15),
|
||||
"Base must be a GPR.");
|
||||
LOGMAN_THROW_A_FMT(Index_gpr >= FEXCore::X86State::REG_XMM_0 && Index_gpr <= FEXCore::X86State::REG_XMM_15, "must be AVX reg");
|
||||
LOGMAN_THROW_A_FMT(Base_gpr == FEXCore::X86State::REG_INVALID || (Base_gpr >= FEXCore::X86State::REG_RAX && Base_gpr <= FEXCore::X86State::REG_R15),
|
||||
"Base must be a GPR.");
|
||||
const auto Index_XMM_gpr = Index_gpr - X86State::REG_XMM_0;
|
||||
|
||||
return {
|
||||
|
||||
@@ -16,6 +16,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/FPState.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
@@ -129,8 +130,23 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreStackMemory(Mem, OpSize::i128Bit, true, Width);
|
||||
// Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
// FIXME: Is TSO relevant for x87?
|
||||
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
|
||||
|
||||
// Index scale is a power of 2?
|
||||
LOGMAN_THROW_A_FMT(A.IndexScale > 0 && (A.IndexScale & (A.IndexScale - 1)) == 0, "Invalid index scale");
|
||||
|
||||
Ref Addr = A.Base ? A.Base : _Constant(0);
|
||||
if (A.Index) {
|
||||
Ref ScaledIndex = A.Index;
|
||||
if (A.IndexScale > 1) {
|
||||
ScaledIndex = _Lshl(A.AddrSize, ScaledIndex, _Constant(std::log2(A.IndexScale)));
|
||||
}
|
||||
Addr = _Add(A.AddrSize, Addr, ScaledIndex);
|
||||
}
|
||||
|
||||
_StoreStackMem(OpSize::i128Bit, Width, Addr, _Constant(A.Offset), /*Float=*/true);
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
@@ -226,9 +242,9 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
|
||||
|
||||
void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto Offset = Op->OP & 7;
|
||||
const auto St0 = 0;
|
||||
const auto Result = (ResInST0 == OpResult::RES_STI) ? Offset : St0;
|
||||
const uint8_t Offset = Op->OP & 7;
|
||||
const uint8_t St0 = 0;
|
||||
const uint8_t Result = (ResInST0 == OpResult::RES_STI) ? Offset : St0;
|
||||
|
||||
if (Reverse ^ (ResInST0 == OpResult::RES_STI)) {
|
||||
_F80DivStack(Result, Offset, St0);
|
||||
@@ -751,13 +767,11 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FFREE(OpcodeArgs) {
|
||||
|
||||
_InvalidateStack(Op->OP & 7);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87EMMS(OpcodeArgs) {
|
||||
// Tags all get set to 0b11
|
||||
|
||||
_InvalidateStack(0xff);
|
||||
}
|
||||
|
||||
|
||||
@@ -104,9 +104,21 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSTF64(OpcodeArgs, IR::OpSize Width) {
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreStackMemory(Mem, OpSize::i64Bit, true, Width);
|
||||
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
|
||||
|
||||
// Index scale is a power of 2?
|
||||
LOGMAN_THROW_A_FMT(A.IndexScale > 0 && (A.IndexScale & (A.IndexScale - 1)) == 0, "Invalid index scale");
|
||||
|
||||
Ref Addr = A.Base ? A.Base : _Constant(0);
|
||||
if (A.Index) {
|
||||
Ref ScaledIndex = A.Index;
|
||||
if (A.IndexScale > 1) {
|
||||
ScaledIndex = _Lshl(A.AddrSize, ScaledIndex, _Constant(std::log2(A.IndexScale)));
|
||||
}
|
||||
Addr = _Add(A.AddrSize, Addr, ScaledIndex);
|
||||
}
|
||||
|
||||
_StoreStackMem(OpSize::i64Bit, Width, Addr, _Constant(A.Offset), /*Float=*/true);
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
@@ -138,7 +138,7 @@ static bool LoadAOTIRCache(AOTIRCacheEntry* Entry, int streamfd) {
|
||||
|
||||
auto Array = (AOTIRInlineIndex*)((char*)FilePtr + IndexOffset);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr && Entry->FilePtr == nullptr, "Entry must not be initialized here");
|
||||
LOGMAN_THROW_A_FMT(Entry->Array == nullptr && Entry->FilePtr == nullptr, "Entry must not be initialized here");
|
||||
Entry->Array = Array;
|
||||
Entry->FilePtr = FilePtr;
|
||||
Entry->Size = Size;
|
||||
@@ -368,10 +368,6 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
}
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
@@ -392,7 +388,7 @@ AOTIRCacheEntry* AOTIRCaptureCache::LoadAOTIRCacheEntry(const fextl::string& fil
|
||||
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry {.FileId = fileid, .Filename = filename}});
|
||||
auto Entry = &(Inserted.first->second);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr, "Duplicate LoadAOTIRCacheEntry");
|
||||
LOGMAN_THROW_A_FMT(Entry->Array == nullptr, "Duplicate LoadAOTIRCacheEntry");
|
||||
|
||||
if (CTX->Config.AOTIRLoad && AOTIRLoader) {
|
||||
auto streamfd = AOTIRLoader(fileid);
|
||||
@@ -409,7 +405,7 @@ AOTIRCacheEntry* AOTIRCaptureCache::LoadAOTIRCacheEntry(const fextl::string& fil
|
||||
|
||||
void AOTIRCaptureCache::UnloadAOTIRCacheEntry(AOTIRCacheEntry* Entry) {
|
||||
#ifndef _WIN32
|
||||
LOGMAN_THROW_AA_FMT(Entry != nullptr, "Removing not existing entry");
|
||||
LOGMAN_THROW_A_FMT(Entry != nullptr, "Removing not existing entry");
|
||||
|
||||
if (Entry->Array) {
|
||||
FEXCore::Allocator::munmap(Entry->FilePtr, Entry->Size);
|
||||
|
||||
@@ -724,7 +724,7 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
|
||||
bool IsFragmentExit(FEXCore::IR::IROps Op);
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op);
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocationData* RAData);
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData);
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
template<>
|
||||
|
||||
@@ -7,7 +7,6 @@
|
||||
" SSA = untyped",
|
||||
" GPR = GPR class type",
|
||||
" FPR = FPR class type",
|
||||
" PRED = Predicate register class type",
|
||||
"Declaring the SSA types correctly will allow validation passes to ensure the op is getting passed correct arguments",
|
||||
"",
|
||||
"Arguments must always follow a particular order. <Type>:<Prefix><Name>",
|
||||
@@ -84,7 +83,6 @@
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {2}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRFixedClass {3}",
|
||||
"constexpr FEXCore::IR::RegisterClassType PREDClass {4}",
|
||||
"constexpr FEXCore::IR::RegisterClassType ComplexClass {5}",
|
||||
"constexpr FEXCore::IR::RegisterClassType InvalidClass {7}",
|
||||
"",
|
||||
@@ -150,7 +148,6 @@
|
||||
"SSA": "OrderedNode*",
|
||||
"GPR": "OrderedNode*",
|
||||
"FPR": "OrderedNode*",
|
||||
"PRED": "OrderedNode*",
|
||||
"FenceType": "FenceType",
|
||||
"RegisterClass": "RegisterClassType",
|
||||
"CondClass": "CondClassType",
|
||||
@@ -567,19 +564,16 @@
|
||||
]
|
||||
},
|
||||
|
||||
"PRED = InitPredicate OpSize:#Size, u8:$Pattern": {
|
||||
"Desc": ["Initialize predicate register from Pattern"],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StoreMemPredicate OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Value, PRED:$Mask, GPR:$Addr": {
|
||||
"Desc": [ "Stores a value to memory using SVE predicate mask." ],
|
||||
"StoreMemX87SVEOptPredicate OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Value, GPR:$Addr": {
|
||||
"Desc": [ "Stores a value to memory using SVE predicate mask that's designed",
|
||||
"specifically for use in the X87 SVE Ldst optimization." ],
|
||||
"DestSize": "RegisterSize",
|
||||
"HasSideEffects": true,
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = LoadMemPredicate OpSize:#RegisterSize, OpSize:#ElementSize, PRED:$Mask, GPR:$Addr": {
|
||||
"Desc": [ "Loads a value to memory using SVE predicate mask." ],
|
||||
"FPR = LoadMemX87SVEOptPredicate OpSize:#RegisterSize, OpSize:#ElementSize, GPR:$Addr": {
|
||||
"Desc": [ "Loads a value to memory using SVE predicate mask that's designed",
|
||||
"specifically for use in the X87 SVE Ldst optimization." ],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
@@ -2788,7 +2782,7 @@
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
},
|
||||
"StoreStackMemory GPR:$Addr, OpSize:$SourceSize, i1:$Float, OpSize:$StoreSize": {
|
||||
"StoreStackMem OpSize:$SourceSize, OpSize:$StoreSize, GPR:$Addr, GPR:$Offset, i1:$Float": {
|
||||
"Desc": [
|
||||
"Takes the top value off the x87 stack and stores it to memory.",
|
||||
"SourceSize is 128bit for F80 values, 64-bit for low precision.",
|
||||
|
||||
@@ -77,14 +77,12 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
*out << "FPR";
|
||||
} else if (Arg == FPRFixedClass.Val) {
|
||||
*out << "FPRFixed";
|
||||
} else if (Arg == PREDClass.Val) {
|
||||
*out << "PRED";
|
||||
} else {
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg, IR::RegisterAllocationData* RAData) {
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg, const IR::RegisterAllocationData* RAData) {
|
||||
auto [CodeNode, IROp] = IR->at(Arg)();
|
||||
const auto ArgID = Arg.ID();
|
||||
|
||||
@@ -100,7 +98,6 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::PREDClass.Val: *out << "(PRED"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
@@ -240,6 +237,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
case OpSize::i64Bit: *out << "i64"; break;
|
||||
case OpSize::i128Bit: *out << "i128"; break;
|
||||
case OpSize::i256Bit: *out << "i256"; break;
|
||||
case OpSize::f80Bit: *out << "f80"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
}
|
||||
@@ -273,7 +271,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocationData* RAData) {
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
int8_t CurrentIndent = 0;
|
||||
|
||||
@@ -41,7 +41,6 @@ FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
case FPRClass:
|
||||
case GPRFixedClass:
|
||||
case FPRFixedClass:
|
||||
case PREDClass:
|
||||
case InvalidClass: return Class;
|
||||
default: break;
|
||||
}
|
||||
@@ -161,7 +160,7 @@ IREmitter::IRPair<IROp_CodeBlock> IREmitter::CreateNewCodeBlockAfter(Ref insertA
|
||||
if (insertAfter) {
|
||||
LinkCodeBlocks(insertAfter, CodeNode);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(CurrentCodeBlock != nullptr, "CurrentCodeBlock must not be null here");
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBlock != nullptr, "CurrentCodeBlock must not be null here");
|
||||
|
||||
// Find last block
|
||||
auto LastBlock = CurrentCodeBlock;
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
|
||||
@@ -10,7 +9,6 @@
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <stdint.h>
|
||||
@@ -46,37 +44,6 @@ public:
|
||||
}
|
||||
void ResetWorkingList();
|
||||
|
||||
// Predicate Cache Implementation
|
||||
// This lives here rather than OpcodeDispatcher because x87StackOptimization Pass
|
||||
// also needs it.
|
||||
struct PredicateKey {
|
||||
ARMEmitter::PredicatePattern Pattern;
|
||||
OpSize Size;
|
||||
bool operator==(const PredicateKey& rhs) const = default;
|
||||
};
|
||||
|
||||
struct PredicateKeyHash {
|
||||
size_t operator()(const PredicateKey& key) const {
|
||||
return FEXCore::ToUnderlying(key.Pattern) + (FEXCore::ToUnderlying(key.Size) * FEXCore::ToUnderlying(OpSize::iInvalid));
|
||||
}
|
||||
};
|
||||
fextl::unordered_map<PredicateKey, Ref, PredicateKeyHash> InitPredicateCache;
|
||||
|
||||
Ref InitPredicateCached(OpSize Size, ARMEmitter::PredicatePattern Pattern) {
|
||||
PredicateKey Key {Pattern, Size};
|
||||
auto ValIt = InitPredicateCache.find(Key);
|
||||
if (ValIt == InitPredicateCache.end()) {
|
||||
auto Predicate = _InitPredicate(Size, static_cast<uint8_t>(FEXCore::ToUnderlying(Pattern)));
|
||||
InitPredicateCache[Key] = Predicate;
|
||||
return Predicate;
|
||||
}
|
||||
return ValIt->second;
|
||||
}
|
||||
|
||||
void ResetInitPredicateCache() {
|
||||
InitPredicateCache.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* @name IR allocation routines
|
||||
*
|
||||
@@ -238,7 +205,7 @@ public:
|
||||
|
||||
ReplaceAllUsesWithRange(Node, NewNode, Start, AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin()));
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Node->NumUses == 0, "Node still used");
|
||||
LOGMAN_THROW_A_FMT(Node->NumUses == 0, "Node still used");
|
||||
|
||||
auto IROp = Node->Op(DualListData.DataBegin())->CW<FEXCore::IR::IROp_Header>();
|
||||
// We can not remove the op if there are side-effects
|
||||
|
||||
@@ -147,7 +147,6 @@ private:
|
||||
class IRListView final {
|
||||
public:
|
||||
IRListView() = delete;
|
||||
IRListView(IRListView&&) = delete;
|
||||
|
||||
IRListView(DualIntrusiveAllocator* Data)
|
||||
: IRListView(reinterpret_cast<void*>(Data->DataBegin()), reinterpret_cast<void*>(Data->ListBegin()), Data->DataSize(), Data->ListSize()) {}
|
||||
|
||||
@@ -53,7 +53,7 @@ void IRDumper::Run(IREmitter* IREmit) {
|
||||
|
||||
auto IR = IREmit->ViewIR();
|
||||
auto HeaderOp = IR.GetHeader();
|
||||
LOGMAN_THROW_AA_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
LOGMAN_THROW_A_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpToFile) {
|
||||
|
||||
@@ -65,7 +65,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
if (!EntryBlock) {
|
||||
EntryBlock = BlockNode;
|
||||
|
||||
@@ -47,7 +47,6 @@ struct RegState {
|
||||
// On arm64, there are 16 Fixed and 12 normal
|
||||
FPRsFixed[Reg.Reg] = ssa;
|
||||
return true;
|
||||
case PREDClass: PREGs[Reg.Reg] = ssa; return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -60,7 +59,6 @@ struct RegState {
|
||||
case GPRFixedClass: return GPRsFixed[Reg.Reg];
|
||||
case FPRClass: return FPRs[Reg.Reg];
|
||||
case FPRFixedClass: return FPRsFixed[Reg.Reg];
|
||||
case PREDClass: return PREGs[Reg.Reg];
|
||||
}
|
||||
return InvalidReg;
|
||||
}
|
||||
@@ -84,7 +82,6 @@ private:
|
||||
std::array<IR::NodeID, 32> FPRsFixed = {};
|
||||
std::array<IR::NodeID, 32> GPRs = {};
|
||||
std::array<IR::NodeID, 32> FPRs = {};
|
||||
std::array<IR::NodeID, 32> PREGs = {};
|
||||
|
||||
fextl::unordered_map<uint32_t, IR::NodeID> Spills;
|
||||
};
|
||||
|
||||
@@ -191,7 +191,7 @@ unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType C
|
||||
case COND_FLEU:
|
||||
case COND_FGT: return FLAG_N | FLAG_Z | FLAG_V;
|
||||
|
||||
default: LOGMAN_THROW_AA_FMT(false, "unknown cond class type"); return FLAG_NZCV;
|
||||
default: LOGMAN_THROW_A_FMT(false, "unknown cond class type"); return FLAG_NZCV;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -435,7 +435,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
});
|
||||
}
|
||||
|
||||
default: LOGMAN_THROW_AA_FMT(false, "invalid special op"); FEX_UNREACHABLE;
|
||||
default: LOGMAN_THROW_A_FMT(false, "invalid special op"); FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
|
||||
@@ -21,7 +21,7 @@ using namespace FEXCore;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
namespace {
|
||||
constexpr uint32_t INVALID_REG = IR::InvalidReg;
|
||||
[[maybe_unused]] constexpr uint32_t INVALID_REG = IR::InvalidReg;
|
||||
constexpr uint32_t INVALID_CLASS = IR::InvalidClass.Val;
|
||||
|
||||
struct RegisterClass {
|
||||
@@ -160,7 +160,7 @@ private:
|
||||
|
||||
// Otherwise fill from stack
|
||||
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Old).Value];
|
||||
LOGMAN_THROW_AA_FMT(SlotPlusOne >= 1, "Old must have been spilled");
|
||||
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Old must have been spilled");
|
||||
|
||||
RegisterClassType RegClass = GetRegClassFromNode(IR, IROp);
|
||||
|
||||
@@ -214,7 +214,7 @@ private:
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!(Class->Available & RegBits), "Register double-free");
|
||||
LOGMAN_THROW_A_FMT(!(Class->Available & RegBits), "Register double-free");
|
||||
|
||||
Class->Available |= RegBits;
|
||||
};
|
||||
@@ -260,7 +260,7 @@ private:
|
||||
Class = Op->Class;
|
||||
Reg = Op->Reg;
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
LOGMAN_THROW_AA_FMT(IROp->Op == OP_STOREREGISTER, "node is SRA");
|
||||
LOGMAN_THROW_A_FMT(IROp->Op == OP_STOREREGISTER, "node is SRA");
|
||||
const IROp_StoreRegister* Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
Class = Op->Class;
|
||||
@@ -289,13 +289,13 @@ private:
|
||||
// next-use has the /smallest/ unsigned IP.
|
||||
Ref Candidate = nullptr;
|
||||
uint32_t BestDistance = UINT32_MAX;
|
||||
uint8_t BestReg = ~0;
|
||||
[[maybe_unused]] uint8_t BestReg = ~0;
|
||||
uint32_t Allocated = ((1u << Class->Count) - 1) & ~Class->Available;
|
||||
|
||||
foreach_bit(i, Allocated) {
|
||||
Ref Old = Class->RegToSSA[i];
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Old != nullptr, "Invariant3");
|
||||
LOGMAN_THROW_A_FMT(Old != nullptr, "Invariant3");
|
||||
LOGMAN_THROW_A_FMT(SSAToReg[IR->GetID(Map(Old)).Value].Reg == i, "Invariant4");
|
||||
|
||||
// Skip any source used by the current instruction, it is unspillable.
|
||||
@@ -316,11 +316,11 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Candidate != nullptr, "must've found something..");
|
||||
LOGMAN_THROW_A_FMT(Candidate != nullptr, "must've found something..");
|
||||
LOGMAN_THROW_A_FMT(IsOld(Candidate), "Invariant5");
|
||||
|
||||
PhysicalRegister Reg = SSAToReg[IR->GetID(Map(Candidate)).Value];
|
||||
LOGMAN_THROW_AA_FMT(Reg.Reg == BestReg, "Invariant6");
|
||||
LOGMAN_THROW_A_FMT(Reg.Reg == BestReg, "Invariant6");
|
||||
|
||||
IROp_Header* Header = IR->GetOp<IROp_Header>(Candidate);
|
||||
uint32_t Value = IR->GetID(Candidate).Value;
|
||||
@@ -357,7 +357,7 @@ private:
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_AA_FMT((Class->Available & RegBits) == RegBits, "Precondition");
|
||||
LOGMAN_THROW_A_FMT((Class->Available & RegBits) == RegBits, "Precondition");
|
||||
|
||||
Class->Available &= ~RegBits;
|
||||
Class->RegToSSA[Reg.Reg] = Unmap(Node);
|
||||
@@ -435,7 +435,7 @@ private:
|
||||
}
|
||||
|
||||
// Assign a free register in the appropriate class.
|
||||
LOGMAN_THROW_AA_FMT(Class->Available != 0, "Post-condition of spilling");
|
||||
LOGMAN_THROW_A_FMT(Class->Available != 0, "Post-condition of spilling");
|
||||
unsigned Reg = std::countr_zero(Class->Available);
|
||||
SetReg(CodeNode, PhysicalRegister(ClassType, Reg));
|
||||
};
|
||||
@@ -446,7 +446,7 @@ private:
|
||||
};
|
||||
|
||||
void ConstrainedRAPass::AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) {
|
||||
LOGMAN_THROW_AA_FMT(RegisterCount <= INVALID_REG, "Up to {} regs supported", INVALID_REG);
|
||||
LOGMAN_THROW_A_FMT(RegisterCount <= INVALID_REG, "Up to {} regs supported", INVALID_REG);
|
||||
|
||||
Classes[Class].Count = RegisterCount;
|
||||
}
|
||||
@@ -623,7 +623,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
|
||||
SourceIndex--;
|
||||
LOGMAN_THROW_AA_FMT(SourceIndex >= 0, "Consistent source count");
|
||||
LOGMAN_THROW_A_FMT(SourceIndex >= 0, "Consistent source count");
|
||||
|
||||
if (!SourcesNextUses[SourceIndex]) {
|
||||
Ref Old = IR->GetNode(IROp->Args[s]);
|
||||
@@ -654,11 +654,11 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(IP >= 1, "IP relative to end of block, iterating forward");
|
||||
LOGMAN_THROW_A_FMT(IP >= 1, "IP relative to end of block, iterating forward");
|
||||
--IP;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(SourceIndex == 0, "Consistent source count in block");
|
||||
LOGMAN_THROW_A_FMT(SourceIndex == 0, "Consistent source count in block");
|
||||
}
|
||||
|
||||
/* Now that we're done growing things, we can finalize our results.
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/Profiler.h"
|
||||
#include "FEXCore/Core/HostFeatures.h"
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
@@ -161,6 +161,18 @@ private:
|
||||
// Helpers
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
|
||||
// Helper to check if a Ref is a Zero constant
|
||||
bool IsZero(Ref Node) {
|
||||
auto Header = IR->GetOp<IR::IROp_Header>(Node);
|
||||
if (Header->Op != OP_CONSTANT) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Const = Header->C<IROp_Constant>();
|
||||
return Const->Constant == 0;
|
||||
}
|
||||
|
||||
|
||||
// Handles a Unary operation.
|
||||
// Takes the op we are handling, the Node for the reduced precision case and the node for the normal case.
|
||||
// Depending on the type of Op64, we might need to pass a couple of extra constant arguments, this happens
|
||||
@@ -245,6 +257,7 @@ private:
|
||||
bool SlowPath = false;
|
||||
// Keeping IREmitter not to pass arguments around
|
||||
IREmitter* IREmit = nullptr;
|
||||
IRListView* IR;
|
||||
};
|
||||
|
||||
inline void X87StackOptimization::InvalidateCaches() {
|
||||
@@ -528,7 +541,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
auto CurrentIR = Emit->ViewIR();
|
||||
auto* HeaderOp = CurrentIR.GetHeader();
|
||||
LOGMAN_THROW_AA_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
LOGMAN_THROW_A_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
|
||||
if (!HeaderOp->HasX87) {
|
||||
// If there is no x87 in this, just early exit.
|
||||
@@ -537,6 +550,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
// Initialize IREmit member
|
||||
IREmit = Emit;
|
||||
IR = &CurrentIR;
|
||||
|
||||
// Run optimization proper
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
@@ -780,11 +794,12 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_STORESTACKMEMORY: {
|
||||
const auto* Op = IROp->C<IROp_StoreStackMemory>();
|
||||
case OP_STORESTACKMEM: {
|
||||
const auto* Op = IROp->C<IROp_StoreStackMem>();
|
||||
const auto& Value = MigrateToSlowPath_IfInvalid();
|
||||
Ref StackNode = SlowPath ? LoadStackValueAtOffset_Slow() : Value->StackDataNode;
|
||||
Ref AddrNode = CurrentIR.GetNode(Op->Addr);
|
||||
Ref Offset = CurrentIR.GetNode(Op->Offset);
|
||||
|
||||
// On the fast path we can optimize memory copies.
|
||||
// If we are doing:
|
||||
@@ -796,45 +811,48 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// or similar. As long as the source size and dest size are one and the same.
|
||||
// This will avoid any conversions between source and stack element size and conversion back.
|
||||
if (!SlowPath && Value->Source && Value->Source->first == Op->StoreSize && Value->InterpretAsFloat) {
|
||||
IREmit->_StoreMem(Value->InterpretAsFloat ? FPRClass : GPRClass, Op->StoreSize, AddrNode, Value->Source->second);
|
||||
IREmit->_StoreMem(Value->InterpretAsFloat ? FPRClass : GPRClass, Op->StoreSize, Value->Source->second, AddrNode, Offset,
|
||||
OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
if (ReducedPrecisionMode) {
|
||||
switch (Op->StoreSize) {
|
||||
case OpSize::i32Bit: {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i32Bit, AddrNode, StackNode);
|
||||
break;
|
||||
}
|
||||
case OpSize::i32Bit:
|
||||
case OpSize::i64Bit: {
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
if (Op->StoreSize == OpSize::i32Bit) {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
}
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
break;
|
||||
}
|
||||
case OpSize::f80Bit: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, AddrNode, GetConstant(8), OpSize::i64Bit, MEM_OFFSET_SXTX, 1);
|
||||
auto NewOffset = IREmit->_Add(OpSize::i64Bit, Offset, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, AddrNode, NewOffset, OpSize::i64Bit, MEM_OFFSET_SXTX, 1);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
}
|
||||
} else {
|
||||
} else { // !ReducedPrecisionMode
|
||||
if (Op->StoreSize != OpSize::f80Bit) { // if it's not 80bits then convert
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
}
|
||||
if (Op->StoreSize == OpSize::f80Bit) { // Part of code from StoreResult_WithOpSize()
|
||||
if (Op->StoreSize == OpSize::f80Bit) {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
auto PReg = IREmit->InitPredicateCached(OpSize::i16Bit, ARMEmitter::PredicatePattern::SVE_VL5);
|
||||
IREmit->_StoreMemPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, PReg, AddrNode);
|
||||
if (!IsZero(Offset)) {
|
||||
AddrNode = IREmit->_Add(OpSize::i64Bit, AddrNode, Offset);
|
||||
}
|
||||
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
|
||||
} else {
|
||||
// For X87 extended doubles, split before storing
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
auto DestAddr = IREmit->_Add(OpSize::i64Bit, AddrNode, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, DestAddr, Upper, OpSize::i64Bit);
|
||||
auto NewOffset = IREmit->_Add(OpSize::i64Bit, Offset, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, AddrNode, NewOffset, OpSize::i64Bit, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, AddrNode, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -112,14 +112,18 @@ void ReenableSBRKAllocations(void* Ptr) {
|
||||
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wdeprecated-declarations"
|
||||
void SetupHooks() {
|
||||
Alloc64 = Alloc::OSAllocator::Create64BitAllocator();
|
||||
static void AssignHookOverrides() {
|
||||
SetJemallocMmapHook(FEX_mmap);
|
||||
SetJemallocMunmapHook(FEX_munmap);
|
||||
FEXCore::Allocator::mmap = FEX_mmap;
|
||||
FEXCore::Allocator::munmap = FEX_munmap;
|
||||
}
|
||||
|
||||
void SetupHooks() {
|
||||
Alloc64 = Alloc::OSAllocator::Create64BitAllocator();
|
||||
AssignHookOverrides();
|
||||
}
|
||||
|
||||
void ClearHooks() {
|
||||
SetJemallocMmapHook(::mmap);
|
||||
SetJemallocMunmapHook(::munmap);
|
||||
@@ -282,7 +286,7 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
auto Alloc =
|
||||
mmap(StackRegionIt->Ptr, StackRegionIt->Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", StackRegionIt->Ptr, StackRegionIt->Size);
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({},{:x}) failed", fmt::ptr(StackRegionIt->Ptr), StackRegionIt->Size);
|
||||
LogMan::Throw::AFmt(Alloc == StackRegionIt->Ptr, "mmap returned {} instead of {}", Alloc, fmt::ptr(StackRegionIt->Ptr));
|
||||
|
||||
Regions.erase(StackRegionIt);
|
||||
@@ -293,14 +297,14 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
for (auto RegionIt = Regions.begin(); RegionIt != Regions.end(); ++RegionIt) {
|
||||
auto Alloc = mmap(RegionIt->Ptr, RegionIt->Size, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", RegionIt->Ptr, RegionIt->Size);
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({},{:x}) failed", fmt::ptr(RegionIt->Ptr), RegionIt->Size);
|
||||
LogMan::Throw::AFmt(Alloc == RegionIt->Ptr, "mmap returned {} instead of {}", Alloc, fmt::ptr(RegionIt->Ptr));
|
||||
}
|
||||
|
||||
return Regions;
|
||||
}
|
||||
|
||||
fextl::vector<MemoryRegion> Steal48BitVA() {
|
||||
fextl::vector<MemoryRegion> Setup48BitAllocatorIfExists() {
|
||||
size_t Bits = FEXCore::Allocator::DetermineVASize();
|
||||
if (Bits < 48) {
|
||||
return {};
|
||||
@@ -308,7 +312,12 @@ fextl::vector<MemoryRegion> Steal48BitVA() {
|
||||
|
||||
uintptr_t Begin48BitVA = 0x0'8000'0000'0000ULL;
|
||||
uintptr_t End48BitVA = 0x1'0000'0000'0000ULL;
|
||||
return StealMemoryRegion(Begin48BitVA, End48BitVA);
|
||||
auto Regions = StealMemoryRegion(Begin48BitVA, End48BitVA);
|
||||
|
||||
Alloc64 = Alloc::OSAllocator::Create64BitAllocatorWithRegions(Regions);
|
||||
AssignHookOverrides();
|
||||
|
||||
return Regions;
|
||||
}
|
||||
|
||||
void ReclaimMemoryRegion(const fextl::vector<MemoryRegion>& Regions) {
|
||||
|
||||
@@ -7,6 +7,8 @@
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -35,6 +37,8 @@ thread_local FEXCore::Core::InternalThreadState* TLSThread {};
|
||||
class OSAllocator_64Bit final : public Alloc::HostAllocator {
|
||||
public:
|
||||
OSAllocator_64Bit();
|
||||
OSAllocator_64Bit(fextl::vector<FEXCore::Allocator::MemoryRegion>& Regions);
|
||||
|
||||
virtual ~OSAllocator_64Bit();
|
||||
void* AllocateSlab(size_t Size) override {
|
||||
return nullptr;
|
||||
@@ -99,19 +103,20 @@ private:
|
||||
// This returns the size of the LiveVMARegion in addition to the flex set that tracks the used data
|
||||
// The LiveVMARegion lives at the start of the VMA region which means on initialization we need to set that
|
||||
// tracked ranged as used immediately
|
||||
static size_t GetSizeWithFlexSet(size_t Size) {
|
||||
static size_t GetFEXManagedVMARegionSize(size_t Size) {
|
||||
// One element per page
|
||||
|
||||
// 0x10'0000'0000 bytes
|
||||
// 0x100'0000 Pages
|
||||
// 1 bit per page for tracking means 0x20'0000 (Pages / 8) bytes of flex space
|
||||
// Which is 2MB of tracking
|
||||
uint64_t NumElements = (Size >> FEXCore::Utils::FEX_PAGE_SHIFT) * sizeof(FlexBitElementType);
|
||||
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<FlexBitElementType>::Size(NumElements);
|
||||
const uint64_t NumElements = Size >> FEXCore::Utils::FEX_PAGE_SHIFT;
|
||||
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<FlexBitElementType>::SizeInBytes(NumElements);
|
||||
}
|
||||
|
||||
static void InitializeVMARegionUsed(LiveVMARegion* Region, size_t AdditionalSize) {
|
||||
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(Region->SlabInfo->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
size_t SizeOfLiveRegion =
|
||||
FEXCore::AlignUp(LiveVMARegion::GetFEXManagedVMARegionSize(Region->SlabInfo->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
size_t SizePlusManagedData = SizeOfLiveRegion + AdditionalSize;
|
||||
|
||||
Region->FreeSpace = Region->SlabInfo->RegionSize - SizePlusManagedData;
|
||||
@@ -155,7 +160,8 @@ private:
|
||||
ReservedRegions->erase(ReservedIterator);
|
||||
|
||||
// mprotect the new region we've allocated
|
||||
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(ReservedRegion->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
size_t SizeOfLiveRegion =
|
||||
FEXCore::AlignUp(LiveVMARegion::GetFEXManagedVMARegionSize(ReservedRegion->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
|
||||
|
||||
[[maybe_unused]] auto Res = mprotect(reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData, PROT_READ | PROT_WRITE);
|
||||
@@ -180,7 +186,7 @@ private:
|
||||
// 32-bit old kernel workarounds
|
||||
fextl::vector<FEXCore::Allocator::MemoryRegion> Steal32BitIfOldKernel();
|
||||
|
||||
void AllocateMemoryRegions(const fextl::vector<FEXCore::Allocator::MemoryRegion>& Ranges);
|
||||
void AllocateMemoryRegions(fextl::vector<FEXCore::Allocator::MemoryRegion>& Ranges);
|
||||
LiveVMARegion* FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd);
|
||||
};
|
||||
|
||||
@@ -383,7 +389,7 @@ again:
|
||||
if (!LiveRegion) {
|
||||
// Couldn't find a fit in the live regions
|
||||
// Allocate a new reserved region
|
||||
size_t lengthOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(length), FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
size_t lengthOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetFEXManagedVMARegionSize(length), FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
size_t lengthPlusManagedData = length + lengthOfLiveRegion;
|
||||
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
|
||||
if ((*it)->RegionSize >= lengthPlusManagedData) {
|
||||
@@ -515,27 +521,43 @@ fextl::vector<FEXCore::Allocator::MemoryRegion> OSAllocator_64Bit::Steal32BitIfO
|
||||
return FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND_32, UPPER_BOUND_32);
|
||||
}
|
||||
|
||||
void OSAllocator_64Bit::AllocateMemoryRegions(const fextl::vector<FEXCore::Allocator::MemoryRegion>& Ranges) {
|
||||
void OSAllocator_64Bit::AllocateMemoryRegions(fextl::vector<FEXCore::Allocator::MemoryRegion>& Ranges) {
|
||||
// Need to allocate the ObjectAlloc up front. Find a region that is larger than our minimum size first.
|
||||
const size_t ObjectAllocSize = 64 * 1024 * 1024;
|
||||
|
||||
for (auto& it : Ranges) {
|
||||
if (ObjectAllocSize > it.Size) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Allocate up to 64 MiB the first allocation for an intrusive allocator
|
||||
mprotect(it.Ptr, ObjectAllocSize, PROT_READ | PROT_WRITE);
|
||||
|
||||
// This enables the kernel to use transparent large pages in the allocator which can reduce memory pressure
|
||||
::madvise(it.Ptr, ObjectAllocSize, MADV_HUGEPAGE);
|
||||
|
||||
ObjectAlloc = new (it.Ptr) Alloc::ForwardOnlyIntrusiveArenaAllocator(it.Ptr, ObjectAllocSize);
|
||||
ReservedRegions = ObjectAlloc->new_construct(ReservedRegions, ObjectAlloc);
|
||||
LiveRegions = ObjectAlloc->new_construct(LiveRegions, ObjectAlloc);
|
||||
|
||||
if (it.Size >= ObjectAllocSize) {
|
||||
// Modify region size
|
||||
it.Size -= ObjectAllocSize;
|
||||
(uint8_t*&)it.Ptr += ObjectAllocSize;
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
if (!ObjectAlloc) {
|
||||
ERROR_AND_DIE_FMT("Couldn't allocate object allocator!");
|
||||
}
|
||||
|
||||
for (auto [Ptr, AllocationSize] : Ranges) {
|
||||
if (!ObjectAlloc) {
|
||||
auto MaxSize = std::min(size_t(64) * 1024 * 1024, AllocationSize);
|
||||
|
||||
// Allocate up to 64 MiB the first allocation for an intrusive allocator
|
||||
mprotect(Ptr, MaxSize, PROT_READ | PROT_WRITE);
|
||||
|
||||
// This enables the kernel to use transparent large pages in the allocator which can reduce memory pressure
|
||||
::madvise(Ptr, MaxSize, MADV_HUGEPAGE);
|
||||
|
||||
ObjectAlloc = new (Ptr) Alloc::ForwardOnlyIntrusiveArenaAllocator(Ptr, MaxSize);
|
||||
ReservedRegions = ObjectAlloc->new_construct(ReservedRegions, ObjectAlloc);
|
||||
LiveRegions = ObjectAlloc->new_construct(LiveRegions, ObjectAlloc);
|
||||
|
||||
if (AllocationSize > MaxSize) {
|
||||
AllocationSize -= MaxSize;
|
||||
(uint8_t*&)Ptr += MaxSize;
|
||||
} else {
|
||||
continue;
|
||||
}
|
||||
// Skip using any regions that are <= two pages. FEX's VMA allocator requires two pages
|
||||
// for tracking data. So three pages are minimum for a single page VMA allocation.
|
||||
if (AllocationSize <= (FEXCore::Utils::FEX_PAGE_SIZE * 2)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
ReservedVMARegion* Region = ObjectAlloc->new_construct<ReservedVMARegion>();
|
||||
@@ -557,6 +579,10 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
FEXCore::Allocator::ReclaimMemoryRegion(LowMem);
|
||||
}
|
||||
|
||||
OSAllocator_64Bit::OSAllocator_64Bit(fextl::vector<FEXCore::Allocator::MemoryRegion>& Regions) {
|
||||
AllocateMemoryRegions(Regions);
|
||||
}
|
||||
|
||||
OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
// This needs a mutex to be thread safe
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(AllocationMutex, TLSThread);
|
||||
@@ -576,6 +602,62 @@ OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocator() {
|
||||
return fextl::make_unique<OSAllocator_64Bit>();
|
||||
}
|
||||
|
||||
template<class T>
|
||||
struct alloc_delete : public std::default_delete<T> {
|
||||
void operator()(T* ptr) const {
|
||||
if (ptr) {
|
||||
const auto size = sizeof(T);
|
||||
const auto MinPage = FEXCore::AlignUp(size, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
std::destroy_at(ptr);
|
||||
::munmap(ptr, MinPage);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename U>
|
||||
requires (std::is_base_of_v<U, T>)
|
||||
operator fextl::default_delete<U>() {
|
||||
return fextl::default_delete<U>();
|
||||
}
|
||||
};
|
||||
|
||||
template<class T, class... Args>
|
||||
requires (!std::is_array_v<T>)
|
||||
fextl::unique_ptr<T> make_alloc_unique(FEXCore::Allocator::MemoryRegion& Base, Args&&... args) {
|
||||
const auto size = sizeof(T);
|
||||
const auto MinPage = FEXCore::AlignUp(size, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (Base.Size < size || MinPage != FEXCore::Utils::FEX_PAGE_SIZE) {
|
||||
ERROR_AND_DIE_FMT("Couldn't fit allocator in to page!");
|
||||
}
|
||||
|
||||
auto ptr = ::mmap(Base.Ptr, MinPage, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
|
||||
if (ptr == MAP_FAILED) {
|
||||
ERROR_AND_DIE_FMT("Couldn't allocate memory region");
|
||||
}
|
||||
|
||||
// Remove the page from the base region.
|
||||
// Could be zero after this.
|
||||
Base.Size -= MinPage;
|
||||
Base.Ptr = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(Base.Ptr) + MinPage);
|
||||
|
||||
auto Result = ::new (ptr) T(std::forward<Args>(args)...);
|
||||
return fextl::unique_ptr<T, alloc_delete<T>>(Result);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocatorWithRegions(fextl::vector<FEXCore::Allocator::MemoryRegion>& Regions) {
|
||||
// This is a bit tricky as we can't allocate memory safely except from the Regions provided. Otherwise we might overwrite memory pages we
|
||||
// don't own. Scan the memory regions and find the smallest one.
|
||||
FEXCore::Allocator::MemoryRegion& Smallest = Regions[0];
|
||||
for (auto& it : Regions) {
|
||||
if (it.Size <= Smallest.Size) {
|
||||
Smallest = it;
|
||||
}
|
||||
}
|
||||
|
||||
return make_alloc_unique<OSAllocator_64Bit>(Smallest, Regions);
|
||||
}
|
||||
|
||||
} // namespace Alloc::OSAllocator
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
|
||||
@@ -72,7 +72,7 @@ struct FlexBitSet final {
|
||||
bool FoundHole {};
|
||||
for (size_t CurrentPage = BeginningElement; CurrentPage >= (MinimumElement + ElementCount);) {
|
||||
size_t Remaining = ElementCount;
|
||||
LOGMAN_THROW_AA_FMT(Remaining <= CurrentPage, "Scanning less than available range");
|
||||
LOGMAN_THROW_A_FMT(Remaining <= CurrentPage, "Scanning less than available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentPage - Remaining) == WantUnset) {
|
||||
@@ -112,7 +112,7 @@ struct FlexBitSet final {
|
||||
// If we have enough free space, check if we have enough free pages that are contiguous
|
||||
size_t Remaining = ElementCount;
|
||||
|
||||
LOGMAN_THROW_AA_FMT((CurrentElement + Remaining - 1) < ElementsInSet, "Scanning less than available range");
|
||||
LOGMAN_THROW_A_FMT((CurrentElement + Remaining - 1) < ElementsInSet, "Scanning less than available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentElement + Remaining - 1) == WantUnset) {
|
||||
@@ -145,8 +145,14 @@ struct FlexBitSet final {
|
||||
return Get(Element);
|
||||
}
|
||||
|
||||
static size_t Size(uint64_t Elements) {
|
||||
return FEXCore::AlignUp(Elements / MinimumSizeBits, MinimumSizeBits);
|
||||
// Returns the number of bits required to hold the number of elements.
|
||||
// Just rounds up to the MinimumSizeInBits.
|
||||
constexpr static size_t SizeInBits(uint64_t Elements) {
|
||||
return FEXCore::AlignUp(Elements, MinimumSizeBits);
|
||||
}
|
||||
// Returns the number of bytes required to hold the number of elements.
|
||||
constexpr static size_t SizeInBytes(uint64_t Elements) {
|
||||
return SizeInBits(Elements) / 8;
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -2,9 +2,10 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/allocator.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <sys/types.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
@@ -49,4 +50,5 @@ public:
|
||||
|
||||
namespace Alloc::OSAllocator {
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocator();
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocatorWithRegions(fextl::vector<FEXCore::Allocator::MemoryRegion>& Regions);
|
||||
} // namespace Alloc::OSAllocator
|
||||
@@ -46,7 +46,7 @@ constexpr uint32_t LDSTREGISTER_MASK = 0b0011'1011'0010'0000'0000'1100'0000'0000
|
||||
constexpr uint32_t LDR_INST = 0b0011'1000'0111'1111'0110'1000'0000'0000;
|
||||
constexpr uint32_t STR_INST = 0b0011'1000'0011'1111'0110'1000'0000'0000;
|
||||
|
||||
constexpr uint32_t LDSTUNSCALED_MASK = 0b0011'1011'0010'0000'0000'1100'0000'0000;
|
||||
constexpr uint32_t LDSTUNSCALED_MASK = 0b0011'1011'1110'0000'0000'1100'0000'0000;
|
||||
constexpr uint32_t LDUR_INST = 0b0011'1000'0100'0000'0000'0000'0000'0000;
|
||||
constexpr uint32_t STUR_INST = 0b0011'1000'0000'0000'0000'0000'0000'0000;
|
||||
|
||||
|
||||
@@ -24,14 +24,13 @@ public:
|
||||
// Itanium C++ ABI (https://itanium-cxx-abi.github.io/cxx-abi/abi.html#member-function-pointers)
|
||||
// Low bit of ptr specifies if this Member function pointer is virtual or not
|
||||
// Throw an assert if we were trying to cast a virtual member
|
||||
LOGMAN_THROW_AA_FMT((PMF.ptr & 1) == 0, "C++ Pointer-To-Member representation didn't have low bit set to 0. Are you trying to cast a "
|
||||
"virtual member?");
|
||||
LOGMAN_THROW_A_FMT((PMF.ptr & 1) == 0, "C++ Pointer-To-Member representation didn't have low bit set to 0. Are you trying to cast a "
|
||||
"virtual member?");
|
||||
#elif defined(_M_ARM_64)
|
||||
// C++ ABI for the Arm 64-bit Architecture (IHI 0059E)
|
||||
// 4.2.1 Representation of pointer to member function
|
||||
// Differs from Itanium specification
|
||||
LOGMAN_THROW_AA_FMT(PMF.adj == 0, "C++ Pointer-To-Member representation didn't have adj == 0. Are you trying to cast a virtual "
|
||||
"member?");
|
||||
LOGMAN_THROW_A_FMT(PMF.adj == 0, "C++ Pointer-To-Member representation didn't have adj == 0. Are you trying to cast a virtual member?");
|
||||
#else
|
||||
#error Don't know how to cast Member to function here. Likely just Itanium
|
||||
#endif
|
||||
@@ -44,15 +43,15 @@ public:
|
||||
// Itanium C++ ABI (https://itanium-cxx-abi.github.io/cxx-abi/abi.html#member-function-pointers)
|
||||
// Low bit of ptr specifies if this Member function pointer is virtual or not
|
||||
// Throw an assert if we are not loading a virtual member.
|
||||
LOGMAN_THROW_AA_FMT((PMF.ptr & 1) == 1, "C++ Pointer-To-Member representation didn't have low bit set to 1. This cast only works for "
|
||||
"virtual members.");
|
||||
LOGMAN_THROW_A_FMT((PMF.ptr & 1) == 1, "C++ Pointer-To-Member representation didn't have low bit set to 1. This cast only works for "
|
||||
"virtual members.");
|
||||
return PMF.ptr & ~1ULL;
|
||||
#elif defined(_M_ARM_64)
|
||||
// C++ ABI for the Arm 64-bit Architecture (IHI 0059E)
|
||||
// 4.2.1 Representation of pointer to member function
|
||||
// Differs from Itanium specification
|
||||
LOGMAN_THROW_AA_FMT((PMF.adj & 1) == 1, "C++ Pointer-To-Member representation didn't have adj == 1. This cast only works for virtual "
|
||||
"members.");
|
||||
LOGMAN_THROW_A_FMT((PMF.adj & 1) == 1, "C++ Pointer-To-Member representation didn't have adj == 1. This cast only works for virtual "
|
||||
"members.");
|
||||
return PMF.ptr;
|
||||
#else
|
||||
#error Don't know how to cast Member to function here. Likely just Itanium
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#include <linux/magic.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/vfs.h>
|
||||
#include <time.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -18,6 +19,36 @@
|
||||
#define BACKEND_GPUVIS 1
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
#ifndef _WIN32
|
||||
static inline uint64_t GetTime() {
|
||||
// We want the time in the least amount of overhead possible
|
||||
// clock_gettime will do a VDSO call with the least amount of overhead
|
||||
struct timespec ts;
|
||||
clock_gettime(CLOCK_MONOTONIC, &ts);
|
||||
return ts.tv_sec * 1'000'000'000ULL + ts.tv_nsec;
|
||||
}
|
||||
#else
|
||||
|
||||
static inline uint64_t GetTime() {
|
||||
// GetTime needs to return nanoseconds, query the interface.
|
||||
static uint64_t FrequencyScale = {};
|
||||
if (!FrequencyScale) [[unlikely]] {
|
||||
LARGE_INTEGER Frequency {};
|
||||
while (!QueryPerformanceFrequency(&Frequency))
|
||||
;
|
||||
constexpr uint64_t NanosecondsInSecond = 1'000'000'000ULL;
|
||||
|
||||
// On WINE this will always result in a scale of 100.
|
||||
FrequencyScale = NanosecondsInSecond / Frequency.QuadPart;
|
||||
}
|
||||
LARGE_INTEGER ticks;
|
||||
while (!QueryPerformanceCounter(&ticks))
|
||||
;
|
||||
return ticks.QuadPart * FrequencyScale;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
namespace FEXCore::Profiler {
|
||||
ProfilerBlock::ProfilerBlock(std::string_view const Format)
|
||||
@@ -41,23 +72,18 @@ static std::array<const char*, 2> TraceFSDirectories {
|
||||
"/sys/kernel/debug/tracing",
|
||||
};
|
||||
|
||||
static bool IsTraceFS(const char* Path) {
|
||||
struct statfs stat;
|
||||
if (statfs(Path, &stat)) {
|
||||
return false;
|
||||
}
|
||||
return stat.f_type == TRACEFS_MAGIC;
|
||||
}
|
||||
|
||||
void Init() {
|
||||
for (auto Path : TraceFSDirectories) {
|
||||
if (IsTraceFS(Path)) {
|
||||
fextl::string FilePath = fextl::fmt::format("{}/trace_marker", Path);
|
||||
TraceFD = open(FilePath.c_str(), O_WRONLY | O_CLOEXEC);
|
||||
if (TraceFD != -1) {
|
||||
// Opened TraceFD, early exit
|
||||
break;
|
||||
}
|
||||
#ifdef _WIN32
|
||||
constexpr auto flags = O_WRONLY;
|
||||
#else
|
||||
constexpr auto flags = O_WRONLY | O_CLOEXEC;
|
||||
#endif
|
||||
fextl::string FilePath = fextl::fmt::format("{}/trace_marker", Path);
|
||||
TraceFD = open(FilePath.c_str(), flags);
|
||||
if (TraceFD != -1) {
|
||||
// Opened TraceFD, early exit
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -72,15 +98,19 @@ void Shutdown() {
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {
|
||||
if (TraceFD != -1) {
|
||||
// Print the duration as something that began negative duration ago
|
||||
fextl::string Event = fextl::fmt::format("{} (lduration=-{})\n", Format, Duration);
|
||||
write(TraceFD, Event.c_str(), Event.size());
|
||||
const auto StringSize = Format.size() + strlen(" (lduration=-)\n") + 22;
|
||||
auto Event = reinterpret_cast<char*>(alloca(StringSize));
|
||||
auto Res = ::fmt::format_to_n(Event, StringSize, "{} (lduration=-{})\n", Format, Duration);
|
||||
write(TraceFD, Event, Res.size);
|
||||
}
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
if (TraceFD != -1) {
|
||||
fextl::string Event = fextl::fmt::format("{}\n", Format);
|
||||
write(TraceFD, Format.data(), Format.size());
|
||||
const auto StringSize = Format.size() + 1;
|
||||
auto Event = reinterpret_cast<char*>(alloca(StringSize));
|
||||
auto Res = ::fmt::format_to_n(Event, StringSize, "{}\n", Format);
|
||||
write(TraceFD, Event, Res.size);
|
||||
}
|
||||
}
|
||||
} // namespace GPUVis
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstdint>
|
||||
#include <cstddef>
|
||||
#include <limits>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
// Variable length signed integer
|
||||
// The most common encoded size is 8-bit positive, but other values can occur
|
||||
//
|
||||
// 8-bit:
|
||||
// bit[7] = 0 - 8-bit
|
||||
// bit[6:0] = 7-bit encoding
|
||||
//
|
||||
// 16-bit:
|
||||
// byte1[7:6] = 0b10 - 16-bit
|
||||
// byte1[5:0] = top 6-bits
|
||||
// byte2[7:0] = Bottom 8-bits bits
|
||||
//
|
||||
// 32-bit
|
||||
// byte1[7:5] = 0b110 - 32-bit
|
||||
// byte1[4:0] = <reserved>
|
||||
// word[31:0] = signed word
|
||||
//
|
||||
// 64-bit
|
||||
// byte1[7:5] = 0b111 - 64-bit
|
||||
// byte1[4:0] = <reserved>
|
||||
// dword[63:0] = signed dword
|
||||
struct vl64 final {
|
||||
static size_t EncodedSize(int64_t Data) {
|
||||
if (Data >= vl8_min && Data <= vl8_max) {
|
||||
return sizeof(vl8_enc);
|
||||
} else if (Data >= vl16_min && Data <= vl16_max) {
|
||||
return sizeof(vl16_enc);
|
||||
} else if (Data >= vl32_min && Data <= vl32_max) {
|
||||
return sizeof(vl32_enc);
|
||||
}
|
||||
return sizeof(vl64_enc);
|
||||
}
|
||||
|
||||
struct Decoded {
|
||||
int64_t Integer;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
static Decoded Decode(const uint8_t* data) {
|
||||
auto vl8_type = reinterpret_cast<const vl8_enc*>(data);
|
||||
auto vl16_type = reinterpret_cast<const vl16_enc*>(data);
|
||||
auto vl32_type = reinterpret_cast<const vl32_enc*>(data);
|
||||
auto vl64_type = reinterpret_cast<const vl64_enc*>(data);
|
||||
|
||||
if (vl8_type->Type == vl8_type_header) {
|
||||
return {vl8_type->Integer, sizeof(vl8_enc)};
|
||||
} else if (vl16_type->HighBits.Type == vl16_type_header) {
|
||||
return {vl16_type->Integer(), sizeof(vl16_enc)};
|
||||
} else if (vl32_type->Type == vl32_type_header) {
|
||||
return {vl32_type->Integer, sizeof(vl32_enc)};
|
||||
}
|
||||
return {vl64_type->Integer, sizeof(vl64_enc)};
|
||||
}
|
||||
|
||||
static size_t Encode(uint8_t* dst, int64_t Data) {
|
||||
auto vl8_type = reinterpret_cast<vl8_enc*>(dst);
|
||||
auto vl16_type = reinterpret_cast<vl16_enc*>(dst);
|
||||
auto vl32_type = reinterpret_cast<vl32_enc*>(dst);
|
||||
auto vl64_type = reinterpret_cast<vl64_enc*>(dst);
|
||||
|
||||
if (Data >= vl8_min && Data <= vl8_max) {
|
||||
*vl8_type = {
|
||||
.Integer = static_cast<int8_t>(Data),
|
||||
.Type = vl8_type_header,
|
||||
};
|
||||
return sizeof(vl8_enc);
|
||||
} else if (Data >= vl16_min && Data <= vl16_max) {
|
||||
*vl16_type = {
|
||||
.HighBits {
|
||||
.Top = static_cast<int8_t>((Data >> 8) & 0xFF),
|
||||
.Type = vl16_type_header,
|
||||
},
|
||||
.LowBits = static_cast<uint8_t>(Data & 0xFF),
|
||||
};
|
||||
return sizeof(vl16_enc);
|
||||
} else if (Data >= vl32_min && Data <= vl32_max) {
|
||||
*vl32_type = {
|
||||
.Type = vl32_type_header,
|
||||
.Integer = static_cast<int32_t>(Data),
|
||||
};
|
||||
return sizeof(vl32_enc);
|
||||
}
|
||||
|
||||
*vl64_type = {
|
||||
.Type = vl64_type_header,
|
||||
.Integer = Data,
|
||||
};
|
||||
return sizeof(vl64_enc);
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
struct vl8_enc {
|
||||
int8_t Integer : 7;
|
||||
uint8_t Type : 1;
|
||||
};
|
||||
static_assert(sizeof(vl8_enc) == 1);
|
||||
|
||||
struct vl16_enc {
|
||||
struct {
|
||||
int8_t Top : 6;
|
||||
uint8_t Type : 2;
|
||||
} HighBits;
|
||||
uint8_t LowBits;
|
||||
|
||||
int64_t Integer() const {
|
||||
int16_t Value {};
|
||||
Value |= (HighBits.Top << 8);
|
||||
Value |= LowBits;
|
||||
return (Value << 2) >> 2;
|
||||
}
|
||||
};
|
||||
static_assert(sizeof(vl16_enc) == 2);
|
||||
|
||||
struct FEX_PACKED vl32_enc {
|
||||
uint8_t Type;
|
||||
int32_t Integer;
|
||||
};
|
||||
static_assert(sizeof(vl32_enc) == 5);
|
||||
|
||||
struct FEX_PACKED vl64_enc {
|
||||
uint8_t Type;
|
||||
int64_t Integer;
|
||||
};
|
||||
static_assert(sizeof(vl64_enc) == 9);
|
||||
|
||||
// Maximum ranges for encodings.
|
||||
|
||||
// vl8 can hold a signed 7-bit integer.
|
||||
// Encoded in one 8-bit value.
|
||||
constexpr static int64_t vl8_encoded_bits = 7;
|
||||
constexpr static int64_t vl8_type_header = 0;
|
||||
constexpr static int64_t vl8_min = std::numeric_limits<int64_t>::min() >> ((sizeof(int64_t) * 8) - vl8_encoded_bits);
|
||||
constexpr static int64_t vl8_max = std::numeric_limits<int64_t>::max() >> ((sizeof(int64_t) * 8) - vl8_encoded_bits);
|
||||
|
||||
// vl16 can hold a signed 14-bit integer.
|
||||
// Encoded in one 16-bit value.
|
||||
constexpr static int64_t vl16_encoded_bits = 14;
|
||||
constexpr static int64_t vl16_type_header = 0b10;
|
||||
constexpr static int64_t vl16_min = std::numeric_limits<int64_t>::min() >> ((sizeof(int64_t) * 8) - vl16_encoded_bits);
|
||||
constexpr static int64_t vl16_max = std::numeric_limits<int64_t>::max() >> ((sizeof(int64_t) * 8) - vl16_encoded_bits);
|
||||
|
||||
// vl32 can hold a signed 32-bit integer.
|
||||
// Encoded in 8-bit and 32-bit value;
|
||||
constexpr static int64_t vl32_encoded_bits = 32;
|
||||
constexpr static int64_t vl32_type_header = 0b1100'0000;
|
||||
constexpr static int64_t vl32_min = std::numeric_limits<int32_t>::min();
|
||||
constexpr static int64_t vl32_max = std::numeric_limits<int32_t>::max();
|
||||
|
||||
// vl64 can hold a signed 32-bit integer.
|
||||
// Encoded in 8-bit and 64-bit value.
|
||||
constexpr static int64_t vl64_encoded_bits = 64;
|
||||
constexpr static int64_t vl64_type_header = 0b1110'0000;
|
||||
constexpr static int64_t vl64_min = std::numeric_limits<int64_t>::min();
|
||||
constexpr static int64_t vl64_max = std::numeric_limits<int64_t>::max();
|
||||
};
|
||||
|
||||
} // namespace FEXCore::Utils
|
||||
@@ -170,7 +170,7 @@ public:
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, const char* Data) {
|
||||
LOGMAN_THROW_AA_FMT(Data != nullptr, "Data can't be null");
|
||||
LOGMAN_THROW_A_FMT(Data != nullptr, "Data can't be null");
|
||||
OptionMap[Option].emplace_back(fextl::string(Data));
|
||||
}
|
||||
|
||||
|
||||
@@ -86,7 +86,7 @@ FEX_DEFAULT_VISIBILITY void ReclaimMemoryRegion(const fextl::vector<MemoryRegion
|
||||
// AArch64 canonical addresses are only up to bits 48/52 with the remainder being other things
|
||||
// Use this to reserve the top 128TB of VA so the guest never see it
|
||||
// Returns nullptr on host VA < 48bits
|
||||
FEX_DEFAULT_VISIBILITY fextl::vector<MemoryRegion> Steal48BitVA();
|
||||
FEX_DEFAULT_VISIBILITY fextl::vector<MemoryRegion> Setup48BitAllocatorIfExists();
|
||||
|
||||
#ifndef _WIN32
|
||||
FEX_DEFAULT_VISIBILITY void RegisterTLSData(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
@@ -44,9 +44,6 @@ namespace Throw {
|
||||
[[noreturn]]
|
||||
void MFmt(const char* fmt, const fmt::format_args& args);
|
||||
|
||||
// AA_FMT and AAFmt are assume versions of {AA_FMT, AFmt} which will assert in debug builds if the assumption is incorrect.
|
||||
// In a release build these use __builtin_assume so compilers can optimize around the case that these cases always hold true.
|
||||
// The assume version should be preferred unless what is being checked has side effects.
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
template<typename... Args>
|
||||
static inline void AFmt(bool Value, const char* fmt, const Args&... args) {
|
||||
@@ -55,34 +52,16 @@ namespace Throw {
|
||||
}
|
||||
MFmt(fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
template<typename... Args>
|
||||
static inline void AAFmt(bool Value, const char* fmt, const Args&... args) {
|
||||
if (MSG_LEVEL < ASSERT || Value) {
|
||||
return;
|
||||
}
|
||||
MFmt(fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
#define LOGMAN_THROW_A_FMT(pred, ...) \
|
||||
do { \
|
||||
LogMan::Throw::AFmt(pred, __VA_ARGS__); \
|
||||
} while (0)
|
||||
#define LOGMAN_THROW_AA_FMT(pred, ...) \
|
||||
do { \
|
||||
LogMan::Throw::AFmt(pred, __VA_ARGS__); \
|
||||
} while (0)
|
||||
#else
|
||||
static inline void AFmt(bool, const char*, ...) {}
|
||||
#define LOGMAN_THROW_A_FMT(pred, ...) \
|
||||
do { \
|
||||
} while (0)
|
||||
static inline void AAFmt(bool pred, const char*, ...) {
|
||||
__builtin_assume(pred);
|
||||
}
|
||||
#define LOGMAN_THROW_AA_FMT(pred, ...) \
|
||||
do { \
|
||||
__builtin_assume(pred); \
|
||||
} while (0)
|
||||
#endif
|
||||
|
||||
} // namespace Throw
|
||||
|
||||
@@ -2,7 +2,6 @@
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <string_view>
|
||||
#include <time.h>
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
@@ -14,14 +13,6 @@ FEX_DEFAULT_VISIBILITY void Shutdown();
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format);
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format, uint64_t Duration);
|
||||
|
||||
static inline uint64_t GetTime() {
|
||||
// We want the time in the least amount of overhead possible
|
||||
// clock_gettime will do a VDSO call with the least amount of overhead
|
||||
struct timespec ts;
|
||||
clock_gettime(CLOCK_MONOTONIC, &ts);
|
||||
return ts.tv_sec * 1'000'000'000ULL + ts.tv_nsec;
|
||||
}
|
||||
|
||||
// A class that follows scoping rules to generate a profile duration block
|
||||
class ProfilerBlock final {
|
||||
public:
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
|
||||
#include "Utils/Allocator/FlexBitSet.h"
|
||||
|
||||
TEST_CASE("FlexBitSet - Sizing") {
|
||||
// Ensure that FlexBitSet sizing is correct.
|
||||
|
||||
// Size of zero shouldn't take any space.
|
||||
CHECK(FEXCore::FlexBitSet<uint8_t>::SizeInBytes(0) == 0);
|
||||
CHECK(FEXCore::FlexBitSet<uint16_t>::SizeInBytes(0) == 0);
|
||||
CHECK(FEXCore::FlexBitSet<uint32_t>::SizeInBytes(0) == 0);
|
||||
CHECK(FEXCore::FlexBitSet<uint64_t>::SizeInBytes(0) == 0);
|
||||
|
||||
CHECK(FEXCore::FlexBitSet<uint8_t>::SizeInBits(0) == 0);
|
||||
CHECK(FEXCore::FlexBitSet<uint16_t>::SizeInBits(0) == 0);
|
||||
CHECK(FEXCore::FlexBitSet<uint32_t>::SizeInBits(0) == 0);
|
||||
CHECK(FEXCore::FlexBitSet<uint64_t>::SizeInBits(0) == 0);
|
||||
|
||||
// Size of 1 should take one sizeof(ElementSize) size
|
||||
CHECK(FEXCore::FlexBitSet<uint8_t>::SizeInBytes(1) == sizeof(uint8_t));
|
||||
CHECK(FEXCore::FlexBitSet<uint16_t>::SizeInBytes(1) == sizeof(uint16_t));
|
||||
CHECK(FEXCore::FlexBitSet<uint32_t>::SizeInBytes(1) == sizeof(uint32_t));
|
||||
CHECK(FEXCore::FlexBitSet<uint64_t>::SizeInBytes(1) == sizeof(uint64_t));
|
||||
|
||||
CHECK(FEXCore::FlexBitSet<uint8_t>::SizeInBits(1) == sizeof(uint8_t) * 8);
|
||||
CHECK(FEXCore::FlexBitSet<uint16_t>::SizeInBits(1) == sizeof(uint16_t) * 8);
|
||||
CHECK(FEXCore::FlexBitSet<uint32_t>::SizeInBits(1) == sizeof(uint32_t) * 8);
|
||||
CHECK(FEXCore::FlexBitSet<uint64_t>::SizeInBits(1) == sizeof(uint64_t) * 8);
|
||||
|
||||
// Size of `sizeof(ElementSize) * 8` should take one sizeof(ElementSize) size
|
||||
CHECK(FEXCore::FlexBitSet<uint8_t>::SizeInBytes(sizeof(uint8_t) * 8) == sizeof(uint8_t));
|
||||
CHECK(FEXCore::FlexBitSet<uint16_t>::SizeInBytes(sizeof(uint16_t) * 8) == sizeof(uint16_t));
|
||||
CHECK(FEXCore::FlexBitSet<uint32_t>::SizeInBytes(sizeof(uint32_t) * 8) == sizeof(uint32_t));
|
||||
CHECK(FEXCore::FlexBitSet<uint64_t>::SizeInBytes(sizeof(uint64_t) * 8) == sizeof(uint64_t));
|
||||
|
||||
CHECK(FEXCore::FlexBitSet<uint8_t>::SizeInBits(sizeof(uint8_t) * 8) == sizeof(uint8_t) * 8);
|
||||
CHECK(FEXCore::FlexBitSet<uint16_t>::SizeInBits(sizeof(uint16_t) * 8) == sizeof(uint16_t) * 8);
|
||||
CHECK(FEXCore::FlexBitSet<uint32_t>::SizeInBits(sizeof(uint32_t) * 8) == sizeof(uint32_t) * 8);
|
||||
CHECK(FEXCore::FlexBitSet<uint64_t>::SizeInBits(sizeof(uint64_t) * 8) == sizeof(uint64_t) * 8);
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
#include <catch2/generators/catch_generators_random.hpp>
|
||||
|
||||
#include "Utils/variable_length_integer.h"
|
||||
|
||||
#include <limits>
|
||||
|
||||
TEST_CASE("vl-size") {
|
||||
// Check 8-bit minimum and maximum.
|
||||
CHECK(FEXCore::Utils::vl64::EncodedSize(-64) == 1);
|
||||
CHECK(FEXCore::Utils::vl64::EncodedSize(63) == 1);
|
||||
|
||||
// Check 16-bit minimum and maximum.
|
||||
CHECK(FEXCore::Utils::vl64::EncodedSize(-8192) == 2);
|
||||
CHECK(FEXCore::Utils::vl64::EncodedSize(8191) == 2);
|
||||
|
||||
// Check 32-bit minimum and maximum.
|
||||
CHECK(FEXCore::Utils::vl64::EncodedSize(std::numeric_limits<int32_t>::min()) == 5);
|
||||
CHECK(FEXCore::Utils::vl64::EncodedSize(std::numeric_limits<int32_t>::max()) == 5);
|
||||
|
||||
// Check 64-bit minimum and maximum.
|
||||
CHECK(FEXCore::Utils::vl64::EncodedSize(std::numeric_limits<int64_t>::min()) == 9);
|
||||
CHECK(FEXCore::Utils::vl64::EncodedSize(std::numeric_limits<int64_t>::max()) == 9);
|
||||
}
|
||||
|
||||
TEST_CASE("vl8 - in memory - encode/decode") {
|
||||
uint8_t data[1];
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, 0) == 1);
|
||||
CHECK(data[0] == 0);
|
||||
auto Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 1);
|
||||
CHECK(Dec.Integer == 0);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, 63) == 1);
|
||||
CHECK(data[0] == 0b0011'1111);
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 1);
|
||||
CHECK(Dec.Integer == 63);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, -1) == 1);
|
||||
CHECK(data[0] == 0b0111'1111);
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 1);
|
||||
CHECK(Dec.Integer == -1);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, -64) == 1);
|
||||
CHECK(data[0] == 0b0100'0000);
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 1);
|
||||
CHECK(Dec.Integer == -64);
|
||||
}
|
||||
|
||||
TEST_CASE("vl16 - in memory - encode/decode") {
|
||||
uint8_t data[2];
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, -65) == 2);
|
||||
CHECK((uint64_t)data[0] == 0b1011'1111);
|
||||
CHECK((uint64_t)data[1] == 0b1011'1111);
|
||||
auto Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 2);
|
||||
CHECK(Dec.Integer == -65);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, -66) == 2);
|
||||
CHECK((uint64_t)data[0] == 0b1011'1111);
|
||||
CHECK((uint64_t)data[1] == 0b1011'1110);
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 2);
|
||||
CHECK(Dec.Integer == -66);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, 64) == 2);
|
||||
CHECK((uint64_t)data[0] == 0b1000'0000);
|
||||
CHECK((uint64_t)data[1] == 0b0100'0000);
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 2);
|
||||
CHECK(Dec.Integer == 64);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, 8191) == 2);
|
||||
CHECK((uint64_t)data[0] == 0b1001'1111);
|
||||
CHECK((uint64_t)data[1] == 0b1111'1111);
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 2);
|
||||
CHECK(Dec.Integer == 8191);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, -8192) == 2);
|
||||
CHECK((uint64_t)data[0] == 0b1010'0000);
|
||||
CHECK((uint64_t)data[1] == 0b0000'0000);
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 2);
|
||||
CHECK(Dec.Integer == -8192);
|
||||
}
|
||||
|
||||
TEST_CASE("vl32 - in memory - encode/decode") {
|
||||
uint8_t data[5];
|
||||
int32_t result {};
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, 8192) == 5);
|
||||
CHECK(data[0] == 0b1100'0000);
|
||||
memcpy(&result, &data[1], sizeof(int32_t));
|
||||
CHECK(result == 8192);
|
||||
auto Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 5);
|
||||
CHECK(Dec.Integer == 8192);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, -8193) == 5);
|
||||
CHECK(data[0] == 0b1100'0000);
|
||||
memcpy(&result, &data[1], sizeof(int32_t));
|
||||
CHECK(result == -8193);
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 5);
|
||||
CHECK(Dec.Integer == -8193);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, std::numeric_limits<int32_t>::min()) == 5);
|
||||
CHECK(data[0] == 0b1100'0000);
|
||||
memcpy(&result, &data[1], sizeof(int32_t));
|
||||
CHECK(result == std::numeric_limits<int32_t>::min());
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 5);
|
||||
CHECK(Dec.Integer == std::numeric_limits<int32_t>::min());
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, std::numeric_limits<int32_t>::max()) == 5);
|
||||
CHECK(data[0] == 0b1100'0000);
|
||||
memcpy(&result, &data[1], sizeof(int32_t));
|
||||
CHECK(result == std::numeric_limits<int32_t>::max());
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 5);
|
||||
CHECK(Dec.Integer == std::numeric_limits<int32_t>::max());
|
||||
}
|
||||
|
||||
TEST_CASE("vl64 - in memory - encode/decode") {
|
||||
uint8_t data[9];
|
||||
int64_t result {};
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, static_cast<int64_t>(std::numeric_limits<int32_t>::min()) - 1) == 9);
|
||||
CHECK(data[0] == 0b1110'0000);
|
||||
memcpy(&result, &data[1], sizeof(int64_t));
|
||||
CHECK(result == static_cast<int64_t>(std::numeric_limits<int32_t>::min()) - 1);
|
||||
auto Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 9);
|
||||
CHECK(Dec.Integer == static_cast<int64_t>(std::numeric_limits<int32_t>::min()) - 1);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, static_cast<int64_t>(std::numeric_limits<int32_t>::max()) + 1) == 9);
|
||||
CHECK(data[0] == 0b1110'0000);
|
||||
memcpy(&result, &data[1], sizeof(int64_t));
|
||||
CHECK(result == static_cast<int64_t>(std::numeric_limits<int32_t>::max()) + 1);
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 9);
|
||||
CHECK(Dec.Integer == static_cast<int64_t>(std::numeric_limits<int32_t>::max()) + 1);
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, std::numeric_limits<int64_t>::min()) == 9);
|
||||
CHECK(data[0] == 0b1110'0000);
|
||||
memcpy(&result, &data[1], sizeof(int64_t));
|
||||
CHECK(result == std::numeric_limits<int64_t>::min());
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 9);
|
||||
CHECK(Dec.Integer == std::numeric_limits<int64_t>::min());
|
||||
|
||||
REQUIRE(FEXCore::Utils::vl64::Encode(data, std::numeric_limits<int64_t>::max()) == 9);
|
||||
CHECK(data[0] == 0b1110'0000);
|
||||
memcpy(&result, &data[1], sizeof(int64_t));
|
||||
CHECK(result == std::numeric_limits<int64_t>::max());
|
||||
Dec = FEXCore::Utils::vl64::Decode(data);
|
||||
CHECK(Dec.Size == 9);
|
||||
CHECK(Dec.Integer == std::numeric_limits<int64_t>::max());
|
||||
}
|
||||
@@ -1,21 +1,17 @@
|
||||
if (COMPILE_VIXL_DISASSEMBLER)
|
||||
file(GLOB_RECURSE TESTS CONFIGURE_DEPENDS *.cpp)
|
||||
file(GLOB_RECURSE TESTS CONFIGURE_DEPENDS *.cpp)
|
||||
|
||||
set (LIBS fmt::fmt vixl Catch2::Catch2WithMain FEXCore_Base JemallocLibs)
|
||||
foreach(TEST ${TESTS})
|
||||
get_filename_component(TEST_NAME ${TEST} NAME_WLE)
|
||||
add_executable(Emitter_${TEST_NAME} ${TEST})
|
||||
target_link_libraries(Emitter_${TEST_NAME} PRIVATE ${LIBS})
|
||||
target_include_directories(Emitter_${TEST_NAME} PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/../../Source/")
|
||||
set_target_properties(Emitter_${TEST_NAME} PROPERTIES RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/EmitterTests")
|
||||
catch_discover_tests(Emitter_${TEST_NAME} TEST_SUFFIX ".${TEST_NAME}.Emitter")
|
||||
endforeach()
|
||||
set (LIBS fmt::fmt vixl Catch2::Catch2WithMain FEXCore_Base JemallocLibs)
|
||||
foreach(TEST ${TESTS})
|
||||
get_filename_component(TEST_NAME ${TEST} NAME_WLE)
|
||||
add_executable(Emitter_${TEST_NAME} ${TEST})
|
||||
target_link_libraries(Emitter_${TEST_NAME} PRIVATE ${LIBS})
|
||||
target_include_directories(Emitter_${TEST_NAME} PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/../../Source/")
|
||||
set_target_properties(Emitter_${TEST_NAME} PROPERTIES RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/EmitterTests")
|
||||
catch_discover_tests(Emitter_${TEST_NAME} TEST_SUFFIX ".${TEST_NAME}.Emitter")
|
||||
endforeach()
|
||||
|
||||
add_custom_target(
|
||||
emitter_tests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}/"
|
||||
USES_TERMINAL
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" ${TEST_JOB_FLAG} "-R" "\.*.Emitter$$")
|
||||
else()
|
||||
message(AUTHOR_WARNING "Tests are enabled but vixl disassembler is not. Emitter tests won't be built.")
|
||||
endif()
|
||||
add_custom_target(
|
||||
emitter_tests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}/"
|
||||
USES_TERMINAL
|
||||
COMMAND "ctest" "--output-on-failure" "--timeout" "302" ${TEST_JOB_FLAG} "-R" "\.*.Emitter$$")
|
||||
@@ -47,10 +47,10 @@ for item in sorted(Meta.items()):
|
||||
if Tag != tag and tag != category:
|
||||
Tag = tag
|
||||
print("")
|
||||
print(" - " + tag.split("/")[1])
|
||||
|
||||
print(" - " + tag.split("/")[1])
|
||||
|
||||
for change in item[1]:
|
||||
if Tag == "":
|
||||
print(" - " + change)
|
||||
else:
|
||||
print(" - " + change)
|
||||
else:
|
||||
print(" - " + change)
|
||||
+1
-1
@@ -10,5 +10,5 @@ fi
|
||||
|
||||
# Reformat whole tree.
|
||||
# This is run by the reformat target.
|
||||
git ls-files -z '*.cpp' '*.h' | xargs -0 -n 1 -P $(nproc) python3 Scripts/clang-format.py -i
|
||||
git ls-files -z '*.cpp' '*.h' '*.inl' | xargs -0 -n 1 -P $(nproc) python3 Scripts/clang-format.py -i
|
||||
cd $DIR
|
||||
@@ -479,7 +479,7 @@ fextl::string GetDataDirectory(bool Global, const PortableInformation& PortableI
|
||||
const char* DataOverride = getenv("FEX_APP_DATA_LOCATION");
|
||||
|
||||
if (PortableInfo.IsPortable && (Global || !DataOverride)) {
|
||||
return fextl::fmt::format("{}fex-emu/", PortableInfo.InterpreterPath);
|
||||
return fextl::fmt::format("{}/fex-emu/", PortableInfo.InterpreterPath);
|
||||
}
|
||||
|
||||
fextl::string DataDir {};
|
||||
@@ -502,7 +502,7 @@ fextl::string GetDataDirectory(bool Global, const PortableInformation& PortableI
|
||||
fextl::string GetConfigDirectory(bool Global, const PortableInformation& PortableInfo) {
|
||||
const char* ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (PortableInfo.IsPortable && (Global || !ConfigOverride)) {
|
||||
return fextl::fmt::format("{}fex-emu/", PortableInfo.InterpreterPath);
|
||||
return fextl::fmt::format("{}/fex-emu/", PortableInfo.InterpreterPath);
|
||||
}
|
||||
|
||||
fextl::string ConfigDir;
|
||||
|
||||
@@ -378,6 +378,7 @@ static void SetFPCR(uint64_t Value) {
|
||||
__asm("msr FPCR, %[Value]" ::[Value] "r"(Value));
|
||||
}
|
||||
|
||||
#ifndef VIXL_SIMULATOR
|
||||
__attribute__((naked)) static uint64_t ReadSVEVectorLengthInBits() {
|
||||
///< Can't use rdvl instruction directly because compilers will complain that sve/sme is required.
|
||||
__asm(R"(
|
||||
@@ -385,6 +386,7 @@ __attribute__((naked)) static uint64_t ReadSVEVectorLengthInBits() {
|
||||
ret;
|
||||
)");
|
||||
}
|
||||
#endif
|
||||
#else
|
||||
[[maybe_unused]]
|
||||
static uint32_t GetDCZID() {
|
||||
|
||||
@@ -409,7 +409,7 @@ public:
|
||||
return reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(StackSize())) + StackSize();
|
||||
} else {
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(reinterpret_cast<void*>(STACK_OFFSET), StackSize()));
|
||||
LOGMAN_THROW_AA_FMT(Result != ~0ULL, "Stack Pointer mmap failed");
|
||||
LOGMAN_THROW_A_FMT(Result != ~0ULL, "Stack Pointer mmap failed");
|
||||
return Result + StackSize();
|
||||
}
|
||||
}
|
||||
@@ -422,7 +422,7 @@ public:
|
||||
bool LimitedSize = true;
|
||||
auto DoMMap = [](uint64_t Address, size_t Size) -> void* {
|
||||
void* Result = FEXCore::Allocator::VirtualAlloc(reinterpret_cast<void*>(Address), Size, true);
|
||||
LOGMAN_THROW_AA_FMT(Result == reinterpret_cast<void*>(Address), "Map Memory mmap failed");
|
||||
LOGMAN_THROW_A_FMT(Result == reinterpret_cast<void*>(Address), "Map Memory mmap failed");
|
||||
return Result;
|
||||
};
|
||||
|
||||
|
||||
@@ -147,7 +147,7 @@ ELFContainer::ELFContainer(const fextl::string& Filename, const fextl::string& R
|
||||
// PrintInitArray();
|
||||
// PrintDynamicTable();
|
||||
|
||||
// LOGMAN_THROW_AA_FMT(InterpreterHeader == nullptr, "Can only handle static programs");
|
||||
// LOGMAN_THROW_A_FMT(InterpreterHeader == nullptr, "Can only handle static programs");
|
||||
}
|
||||
|
||||
ELFContainer::~ELFContainer() {
|
||||
@@ -191,8 +191,8 @@ bool ELFContainer::LoadELF_32() {
|
||||
Mode = MODE_32BIT;
|
||||
|
||||
memcpy(&Header, reinterpret_cast<Elf32_Ehdr*>(&RawFile.at(0)), sizeof(Elf32_Ehdr));
|
||||
LOGMAN_THROW_AA_FMT(Header._32.e_phentsize == sizeof(Elf32_Phdr), "PH Entry size wasn't correct size");
|
||||
LOGMAN_THROW_AA_FMT(Header._32.e_shentsize == sizeof(Elf32_Shdr), "PH Entry size wasn't correct size");
|
||||
LOGMAN_THROW_A_FMT(Header._32.e_phentsize == sizeof(Elf32_Phdr), "PH Entry size wasn't correct size");
|
||||
LOGMAN_THROW_A_FMT(Header._32.e_shentsize == sizeof(Elf32_Shdr), "PH Entry size wasn't correct size");
|
||||
|
||||
if (Header._32.e_machine != EM_386) {
|
||||
LogMan::Msg::DFmt("32bit ELF wasn't x86 based");
|
||||
@@ -229,8 +229,8 @@ bool ELFContainer::LoadELF_64() {
|
||||
Mode = MODE_64BIT;
|
||||
|
||||
memcpy(&Header, reinterpret_cast<Elf64_Ehdr*>(&RawFile.at(0)), sizeof(Elf64_Ehdr));
|
||||
LOGMAN_THROW_AA_FMT(Header._64.e_phentsize == 56, "PH Entry size wasn't 56");
|
||||
LOGMAN_THROW_AA_FMT(Header._64.e_shentsize == 64, "PH Entry size wasn't 64");
|
||||
LOGMAN_THROW_A_FMT(Header._64.e_phentsize == 56, "PH Entry size wasn't 56");
|
||||
LOGMAN_THROW_A_FMT(Header._64.e_shentsize == 64, "PH Entry size wasn't 64");
|
||||
|
||||
if (Header._64.e_machine != EM_X86_64) {
|
||||
LogMan::Msg::DFmt("64bit ELF wasn't x86-64 based");
|
||||
@@ -402,7 +402,7 @@ void ELFContainer::CalculateSymbols() {
|
||||
uint64_t NumDynSymSymbols = 0;
|
||||
if (SymTabHeader) {
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_link < SectionHeaders.size(), "Symbol table string table section is wrong");
|
||||
LOGMAN_THROW_AA_FMT(SymTabHeader->sh_entsize == sizeof(Elf32_Sym), "Entry size doesn't match symbol entry");
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_entsize == sizeof(Elf32_Sym), "Entry size doesn't match symbol entry");
|
||||
|
||||
StringTableHeader = SectionHeaders.at(SymTabHeader->sh_link)._32;
|
||||
StrTab = &RawFile.at(StringTableHeader->sh_offset);
|
||||
@@ -411,7 +411,7 @@ void ELFContainer::CalculateSymbols() {
|
||||
|
||||
if (DynSymTabHeader) {
|
||||
LOGMAN_THROW_A_FMT(DynSymTabHeader->sh_link < SectionHeaders.size(), "Symbol table string table section is wrong");
|
||||
LOGMAN_THROW_AA_FMT(DynSymTabHeader->sh_entsize == sizeof(Elf32_Sym), "Entry size doesn't match symbol entry");
|
||||
LOGMAN_THROW_A_FMT(DynSymTabHeader->sh_entsize == sizeof(Elf32_Sym), "Entry size doesn't match symbol entry");
|
||||
|
||||
DynStringTableHeader = SectionHeaders.at(DynSymTabHeader->sh_link)._32;
|
||||
DynStrTab = &RawFile.at(DynStringTableHeader->sh_offset);
|
||||
@@ -526,7 +526,7 @@ void ELFContainer::CalculateSymbols() {
|
||||
uint64_t NumDynSymSymbols = 0;
|
||||
if (SymTabHeader) {
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_link < SectionHeaders.size(), "Symbol table string table section is wrong");
|
||||
LOGMAN_THROW_AA_FMT(SymTabHeader->sh_entsize == sizeof(Elf64_Sym), "Entry size doesn't match symbol entry");
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_entsize == sizeof(Elf64_Sym), "Entry size doesn't match symbol entry");
|
||||
|
||||
StringTableHeader = SectionHeaders.at(SymTabHeader->sh_link)._64;
|
||||
StrTab = &RawFile.at(StringTableHeader->sh_offset);
|
||||
@@ -535,7 +535,7 @@ void ELFContainer::CalculateSymbols() {
|
||||
|
||||
if (DynSymTabHeader) {
|
||||
LOGMAN_THROW_A_FMT(DynSymTabHeader->sh_link < SectionHeaders.size(), "Symbol table string table section is wrong");
|
||||
LOGMAN_THROW_AA_FMT(DynSymTabHeader->sh_entsize == sizeof(Elf64_Sym), "Entry size doesn't match symbol entry");
|
||||
LOGMAN_THROW_A_FMT(DynSymTabHeader->sh_entsize == sizeof(Elf64_Sym), "Entry size doesn't match symbol entry");
|
||||
|
||||
DynStringTableHeader = SectionHeaders.at(DynSymTabHeader->sh_link)._64;
|
||||
DynStrTab = &RawFile.at(DynStringTableHeader->sh_offset);
|
||||
@@ -795,7 +795,7 @@ void ELFContainer::PrintSymbolTable() const {
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_link < SectionHeaders.size(), "Symbol table string table section is wrong");
|
||||
LOGMAN_THROW_AA_FMT(SymTabHeader->sh_entsize == sizeof(Elf32_Sym), "Entry size doesn't match symbol entry");
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_entsize == sizeof(Elf32_Sym), "Entry size doesn't match symbol entry");
|
||||
|
||||
StringTableHeader = SectionHeaders.at(SymTabHeader->sh_link)._32;
|
||||
StrTab = &RawFile.at(StringTableHeader->sh_offset);
|
||||
@@ -826,7 +826,7 @@ void ELFContainer::PrintSymbolTable() const {
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_link < SectionHeaders.size(), "Symbol table string table section is wrong");
|
||||
LOGMAN_THROW_AA_FMT(SymTabHeader->sh_entsize == sizeof(Elf64_Sym), "Entry size doesn't match symbol entry");
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_entsize == sizeof(Elf64_Sym), "Entry size doesn't match symbol entry");
|
||||
|
||||
StringTableHeader = SectionHeaders.at(SymTabHeader->sh_link)._64;
|
||||
StrTab = &RawFile.at(StringTableHeader->sh_offset);
|
||||
@@ -885,7 +885,7 @@ void ELFContainer::PrintRelocationTable() const {
|
||||
LogMan::Msg::DFmt("\toffset: 0x{:x}", Entry->r_offset);
|
||||
LogMan::Msg::DFmt("\tSym: 0x{:x}", Sym);
|
||||
if (DynSymHeader && Sym != 0) {
|
||||
LOGMAN_THROW_AA_FMT(DynSymHeader->sh_entsize == sizeof(Elf64_Sym), "Oops, entry size doesn't match");
|
||||
LOGMAN_THROW_A_FMT(DynSymHeader->sh_entsize == sizeof(Elf64_Sym), "Oops, entry size doesn't match");
|
||||
|
||||
const uint64_t offset = DynSymHeader->sh_offset + Sym * DynSymHeader->sh_entsize;
|
||||
const auto* Symbol = reinterpret_cast<const Elf64_Sym*>(&RawFile.at(offset));
|
||||
@@ -957,7 +957,7 @@ void ELFContainer::FixupRelocations(void* ELFBase, uint64_t GuestELFBase, Symbol
|
||||
const Elf64_Sym* EntrySymbol {nullptr};
|
||||
const char* EntrySymbolName {nullptr};
|
||||
if (DynSymHeader && Sym != 0) {
|
||||
LOGMAN_THROW_AA_FMT(DynSymHeader->sh_entsize == sizeof(Elf64_Sym), "Oops, entry size doesn't match");
|
||||
LOGMAN_THROW_A_FMT(DynSymHeader->sh_entsize == sizeof(Elf64_Sym), "Oops, entry size doesn't match");
|
||||
|
||||
const uint64_t offset = DynSymHeader->sh_offset + Sym * DynSymHeader->sh_entsize;
|
||||
EntrySymbol = reinterpret_cast<const Elf64_Sym*>(&RawFile.at(offset));
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Common/Config.h"
|
||||
|
||||
namespace FEX {
|
||||
static inline FEX::Config::PortableInformation ReadPortabilityInformation() {
|
||||
const FEX::Config::PortableInformation BadResult {false, {}};
|
||||
const char* PortableConfig = getenv("FEX_PORTABLE");
|
||||
if (!PortableConfig) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
uint32_t Value {};
|
||||
std::string_view PortableView {PortableConfig};
|
||||
|
||||
if (std::from_chars(PortableView.data(), PortableView.data() + PortableView.size(), Value).ec != std::errc {} || Value == 0) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
// Read the FEXInterpreter path from `/proc/self/exe` which is always a symlink to the absolute path of the executable running.
|
||||
// This way we can get the parent path that the application is executing from.
|
||||
char SelfPath[PATH_MAX];
|
||||
auto Result = readlink("/proc/self/exe", SelfPath, PATH_MAX);
|
||||
if (Result == -1) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
std::string_view SelfPathView {SelfPath, std::min<size_t>(PATH_MAX, Result)};
|
||||
|
||||
// Extract the absolute path from the FEXInterpreter path
|
||||
return {true, fextl::string {SelfPathView.substr(0, SelfPathView.find_last_of('/') + 1)}};
|
||||
}
|
||||
} // namespace FEX
|
||||
@@ -222,6 +222,8 @@ ApplicationWindow {
|
||||
component ConfigSpinBox: SpinBox {
|
||||
property string config
|
||||
|
||||
editable: true
|
||||
|
||||
textFromValue: (val) => {
|
||||
if (valueFromConfig === "") {
|
||||
return qsTr("(not set)");
|
||||
|
||||
@@ -655,7 +655,7 @@ public:
|
||||
}
|
||||
|
||||
// Set the null terminator for the string
|
||||
*reinterpret_cast<uint8_t*>(ArgumentBackingBase + CurrentOffset + ArgSize + 1) = 0;
|
||||
*reinterpret_cast<uint8_t*>(ArgumentBackingBase + CurrentOffset + ArgSize) = 0;
|
||||
|
||||
CurrentOffset += ArgSize + 1;
|
||||
}
|
||||
@@ -667,10 +667,12 @@ public:
|
||||
EnvpPointers[i] = EnvpBackingBaseGuest + CurrentOffset;
|
||||
|
||||
// Copy the string in to the final location
|
||||
memcpy(reinterpret_cast<void*>(EnvpBackingBase + CurrentOffset), &EnvironmentVariables[i].at(0), EnvpSize);
|
||||
if (EnvpSize) {
|
||||
memcpy(reinterpret_cast<void*>(EnvpBackingBase + CurrentOffset), &EnvironmentVariables[i].at(0), EnvpSize);
|
||||
}
|
||||
|
||||
// Set the null terminator for the string
|
||||
*reinterpret_cast<uint8_t*>(EnvpBackingBase + CurrentOffset + EnvpSize + 1) = 0;
|
||||
*reinterpret_cast<uint8_t*>(EnvpBackingBase + CurrentOffset + EnvpSize) = 0;
|
||||
|
||||
CurrentOffset += EnvpSize + 1;
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include "Common/FEXServerClient.h"
|
||||
#include "Common/Config.h"
|
||||
#include "Common/HostFeatures.h"
|
||||
#include "PortabilityInfo.h"
|
||||
#include "ELFCodeLoader.h"
|
||||
#include "VDSO_Emulation.h"
|
||||
#include "LinuxSyscalls/GdbServer.h"
|
||||
@@ -176,34 +177,6 @@ bool InterpreterHandler(fextl::string* Filename, const fextl::string& RootFS, fe
|
||||
return true;
|
||||
}
|
||||
|
||||
FEX::Config::PortableInformation ReadPortabilityInformation() {
|
||||
const FEX::Config::PortableInformation BadResult {false, {}};
|
||||
const char* PortableConfig = getenv("FEX_PORTABLE");
|
||||
if (!PortableConfig) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
uint32_t Value {};
|
||||
std::string_view PortableView {PortableConfig};
|
||||
|
||||
if (std::from_chars(PortableView.data(), PortableView.data() + PortableView.size(), Value).ec != std::errc {} || Value == 0) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
// Read the FEXInterpreter path from `/proc/self/exe` which is always a symlink to the absolute path of the executable running.
|
||||
// This way we can get the parent path that the application is executing from.
|
||||
char SelfPath[PATH_MAX];
|
||||
auto Result = readlink("/proc/self/exe", SelfPath, PATH_MAX);
|
||||
if (Result == -1) {
|
||||
return BadResult;
|
||||
}
|
||||
|
||||
std::string_view SelfPathView {SelfPath, std::min<size_t>(PATH_MAX, Result)};
|
||||
|
||||
// Extract the absolute path from the FEXInterpreter path
|
||||
return {true, fextl::string {SelfPathView.substr(0, SelfPathView.find_last_of('/') + 1)}};
|
||||
}
|
||||
|
||||
bool RanAsInterpreter(bool ExecutedWithFD) {
|
||||
return ExecutedWithFD || FEXLOADER_AS_INTERPRETER;
|
||||
}
|
||||
@@ -323,7 +296,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
|
||||
const bool ExecutedWithFD = getauxval(AT_EXECFD) != 0;
|
||||
const bool IsInterpreter = RanAsInterpreter(ExecutedWithFD);
|
||||
const auto PortableInfo = ReadPortabilityInformation();
|
||||
const auto PortableInfo = FEX::ReadPortabilityInformation();
|
||||
const bool InterpreterInstalled = QueryInterpreterInstalled(ExecutedWithFD, PortableInfo);
|
||||
|
||||
int FEXFD {StealFEXFDFromEnv("FEX_EXECVEFD")};
|
||||
@@ -358,6 +331,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
// Doesn't use CONFIG_ROOTFS and we don't want it to spin up a squashfs instance
|
||||
FEX_CONFIG_OPT(StallProcess, STALLPROCESS);
|
||||
FEX_CONFIG_OPT(StartupSleep, STARTUPSLEEP);
|
||||
FEX_CONFIG_OPT(StartupSleepProcName, STARTUPSLEEPPROCNAME);
|
||||
if (StallProcess) {
|
||||
while (1) {
|
||||
// Stall this process out forever
|
||||
@@ -409,7 +383,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
}
|
||||
|
||||
if (StartupSleep()) {
|
||||
if (StartupSleep() && (StartupSleepProcName().empty() || Program.ProgramName == StartupSleepProcName())) {
|
||||
LogMan::Msg::IFmt("[{}][{}] Sleeping for {} seconds", ::getpid(), Program.ProgramName, StartupSleep());
|
||||
std::this_thread::sleep_for(std::chrono::seconds(StartupSleep()));
|
||||
}
|
||||
@@ -496,7 +470,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
|
||||
if (Loader.Is64BitMode()) {
|
||||
// Destroy the 48th bit if it exists
|
||||
Base48Bit = FEXCore::Allocator::Steal48BitVA();
|
||||
Base48Bit = FEXCore::Allocator::Setup48BitAllocatorIfExists();
|
||||
} else {
|
||||
// Reserve [0x1_0000_0000, 0x2_0000_0000).
|
||||
// Safety net if 32-bit address calculation overflows in to 64-bit range.
|
||||
|
||||
@@ -12,7 +12,7 @@ target_include_directories(${NAME} PRIVATE
|
||||
${CMAKE_BINARY_DIR}/generated
|
||||
${CMAKE_SOURCE_DIR}/Source/)
|
||||
|
||||
target_link_libraries(${NAME} PRIVATE FEXCore Common JemallocDummy ${PTHREAD_LIB})
|
||||
target_link_libraries(${NAME} PRIVATE FEXCore Common CommonTools JemallocDummy ${PTHREAD_LIB})
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
target_link_options(${NAME}
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
#include "ArgumentLoader.h"
|
||||
#include "Logger.h"
|
||||
#include "PipeScanner.h"
|
||||
#include "PortabilityInfo.h"
|
||||
#include "ProcessPipe.h"
|
||||
#include "SquashFS.h"
|
||||
#include "Common/ArgumentLoader.h"
|
||||
@@ -117,7 +118,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
|
||||
auto ArgsLoader = fextl::make_unique<FEX::ArgLoader::ArgLoader>(FEX::ArgLoader::ArgLoader::LoadType::WITHOUT_FEXLOADER_PARSER, argc, argv);
|
||||
FEX::Config::LoadConfig(std::move(ArgsLoader), {}, envp);
|
||||
FEX::Config::LoadConfig(std::move(ArgsLoader), {}, envp, FEX::ReadPortabilityInformation());
|
||||
|
||||
// Reload the meta layer
|
||||
FEXCore::Config::ReloadMetaLayer();
|
||||
@@ -200,7 +201,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
// This will let FEXInterpreter know we are ready
|
||||
PipeScanner::ClosePipes();
|
||||
|
||||
ProcessPipe::SetConfiguration(Options.Foreground, Options.PersistentTimeout ?: 10);
|
||||
ProcessPipe::SetConfiguration(Options.Foreground, Options.PersistentTimeout ?: 1);
|
||||
|
||||
// Actually spin up the request thread.
|
||||
// Any applications that were waiting for the socket to accept will then go through here.
|
||||
|
||||
@@ -158,7 +158,7 @@ static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
|
||||
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
auto MContext = GetMContext(ucontext);
|
||||
HostFPRState* HostState = reinterpret_cast<HostFPRState*>(&MContext->__reserved[0]);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
|
||||
return HostState->FPRs[id];
|
||||
}
|
||||
@@ -232,7 +232,7 @@ static inline void BackupContext(void* ucontext, T* Backup) {
|
||||
|
||||
// Host FPR state starts at _mcontext->reserved[0];
|
||||
HostFPRState* HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
Backup->FPSR = HostState->FPSR;
|
||||
Backup->FPCR = HostState->FPCR;
|
||||
memcpy(&Backup->FPRs[0], &HostState->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
@@ -258,7 +258,7 @@ static inline void RestoreContext(void* ucontext, T* Backup) {
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
HostFPRState* HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
memcpy(&HostState->FPRs[0], &Backup->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
HostState->FPCR = Backup->FPCR;
|
||||
HostState->FPSR = Backup->FPSR;
|
||||
|
||||
@@ -348,6 +348,10 @@ enum Syscalls_Arm64 {
|
||||
SYSCALL_Arm64_lsm_set_self_attr = 460,
|
||||
SYSCALL_Arm64_lsm_list_modules = 461,
|
||||
SYSCALL_Arm64_mseal = 462,
|
||||
SYSCALL_Arm64_setxattrat = 463,
|
||||
SYSCALL_Arm64_getxattrat = 464,
|
||||
SYSCALL_Arm64_listxattrat = 465,
|
||||
SYSCALL_Arm64_removexattrat = 466,
|
||||
SYSCALL_Arm64_MAX = 512,
|
||||
|
||||
// Unsupported syscalls on this host
|
||||
|
||||
@@ -1171,6 +1171,86 @@ uint64_t FileManager::LRemovexattr(const char* path, const char* name) {
|
||||
return ::lremovexattr(SelfPath, name);
|
||||
}
|
||||
|
||||
uint64_t FileManager::SetxattrAt(int dfd, const char* pathname, uint32_t at_flags, const char* name, const xattr_args* uargs, size_t usize) {
|
||||
if (IsSelfNoFollow(pathname, at_flags)) {
|
||||
// See Statx
|
||||
return syscall(SYSCALL_DEF(setxattrat), dfd, pathname, at_flags, name, uargs, usize);
|
||||
}
|
||||
|
||||
auto NewPath = GetSelf(pathname);
|
||||
const char* SelfPath = NewPath ? NewPath->data() : nullptr;
|
||||
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(dfd, SelfPath, (at_flags & AT_SYMLINK_NOFOLLOW) == 0, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
uint64_t Result = syscall(SYSCALL_DEF(setxattrat), Path.first, Path.second, at_flags, name, uargs, usize);
|
||||
if (Result != -1) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
return syscall(SYSCALL_DEF(setxattrat), dfd, SelfPath, at_flags, name, uargs, usize);
|
||||
}
|
||||
|
||||
uint64_t FileManager::GetxattrAt(int dfd, const char* pathname, uint32_t at_flags, const char* name, const xattr_args* uargs, size_t usize) {
|
||||
if (IsSelfNoFollow(pathname, at_flags)) {
|
||||
// See Statx
|
||||
return syscall(SYSCALL_DEF(getxattrat), dfd, pathname, at_flags, name, uargs, usize);
|
||||
}
|
||||
|
||||
auto NewPath = GetSelf(pathname);
|
||||
const char* SelfPath = NewPath ? NewPath->data() : nullptr;
|
||||
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(dfd, SelfPath, (at_flags & AT_SYMLINK_NOFOLLOW) == 0, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
uint64_t Result = syscall(SYSCALL_DEF(getxattrat), Path.first, Path.second, at_flags, name, uargs, usize);
|
||||
if (Result != -1) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
return syscall(SYSCALL_DEF(getxattrat), dfd, SelfPath, at_flags, name, uargs, usize);
|
||||
}
|
||||
|
||||
uint64_t FileManager::ListxattrAt(int dfd, const char* pathname, uint32_t at_flags, char* list, size_t size) {
|
||||
if (IsSelfNoFollow(pathname, at_flags)) {
|
||||
// See Statx
|
||||
return syscall(SYSCALL_DEF(listxattrat), dfd, pathname, at_flags, list, size);
|
||||
}
|
||||
|
||||
auto NewPath = GetSelf(pathname);
|
||||
const char* SelfPath = NewPath ? NewPath->data() : nullptr;
|
||||
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(dfd, SelfPath, (at_flags & AT_SYMLINK_NOFOLLOW) == 0, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
uint64_t Result = syscall(SYSCALL_DEF(listxattrat), Path.first, Path.second, at_flags, list, size);
|
||||
if (Result != -1) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
return syscall(SYSCALL_DEF(listxattrat), dfd, SelfPath, at_flags, list, size);
|
||||
}
|
||||
|
||||
uint64_t FileManager::RemovexattrAt(int dfd, const char* pathname, uint32_t at_flags, const char* name) {
|
||||
if (IsSelfNoFollow(pathname, at_flags)) {
|
||||
// See Statx
|
||||
return syscall(SYSCALL_DEF(removexattrat), dfd, pathname, at_flags, name);
|
||||
}
|
||||
|
||||
auto NewPath = GetSelf(pathname);
|
||||
const char* SelfPath = NewPath ? NewPath->data() : nullptr;
|
||||
|
||||
FDPathTmpData TmpFilename;
|
||||
auto Path = GetEmulatedFDPath(dfd, SelfPath, (at_flags & AT_SYMLINK_NOFOLLOW) == 0, TmpFilename);
|
||||
if (Path.first != -1) {
|
||||
uint64_t Result = syscall(SYSCALL_DEF(removexattrat), Path.first, Path.second, at_flags, name);
|
||||
if (Result != -1) {
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
return syscall(SYSCALL_DEF(removexattrat), dfd, SelfPath, at_flags, name);
|
||||
}
|
||||
|
||||
void FileManager::UpdatePID(uint32_t PID) {
|
||||
CurrentPID = PID;
|
||||
|
||||
|
||||
@@ -75,6 +75,17 @@ public:
|
||||
uint64_t LListxattr(const char* path, char* list, size_t size);
|
||||
uint64_t Removexattr(const char* path, const char* name);
|
||||
uint64_t LRemovexattr(const char* path, const char* name);
|
||||
struct xattr_args {
|
||||
uint64_t value;
|
||||
uint32_t size;
|
||||
uint32_t flags;
|
||||
};
|
||||
|
||||
uint64_t SetxattrAt(int dfd, const char* pathname, uint32_t at_flags, const char* name, const xattr_args* uargs, size_t usize);
|
||||
uint64_t GetxattrAt(int dfd, const char* pathname, uint32_t at_flags, const char* name, const xattr_args* uargs, size_t usize);
|
||||
uint64_t ListxattrAt(int dfd, const char* pathname, uint32_t at_flags, char* list, size_t size);
|
||||
uint64_t RemovexattrAt(int dfd, const char* pathname, uint32_t at_flags, const char* name);
|
||||
|
||||
// vfs
|
||||
uint64_t Statfs(const char* path, void* buf);
|
||||
|
||||
|
||||
@@ -64,7 +64,7 @@ namespace FEX {
|
||||
#ifndef _WIN32
|
||||
void GdbServer::Break(FEXCore::Core::InternalThreadState* Thread, int signal) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
if (!CommsStream) {
|
||||
if (!CommsStream.HasSocket()) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -73,7 +73,7 @@ void GdbServer::Break(FEXCore::Core::InternalThreadState* Thread, int signal) {
|
||||
CurrentDebuggingThread = ThreadObject->ThreadInfo.TID.load();
|
||||
|
||||
const auto str = fextl::fmt::format("T{:02x}thread:{:x};", signal, CurrentDebuggingThread);
|
||||
SendPacket(*CommsStream, str);
|
||||
SendPacket(str);
|
||||
}
|
||||
|
||||
void GdbServer::WaitForThreadWakeup() {
|
||||
@@ -178,7 +178,7 @@ static fextl::string encodeHex(std::string_view str) {
|
||||
// Takes a serial stream and reads a single packet
|
||||
// Un-escapes chars, checks the checksum and request a retransmit if it fails.
|
||||
// Once the checksum is validated, it acknowledges and returns the packet in a string
|
||||
fextl::string GdbServer::ReadPacket(std::iostream& stream) {
|
||||
fextl::string GdbServer::ReadPacket() {
|
||||
fextl::string packet {};
|
||||
|
||||
// The GDB "Remote Serial Protocal" was originally 7bit clean for use on serial ports.
|
||||
@@ -190,9 +190,9 @@ fextl::string GdbServer::ReadPacket(std::iostream& stream) {
|
||||
// where any $ or # in the packet body are escaped ('}' followed by the char XORed with 0x20)
|
||||
// The checksum is a single unsigned byte sum of the data, hex encoded.
|
||||
|
||||
int c;
|
||||
while ((c = stream.get()) > 0) {
|
||||
switch (c) {
|
||||
Utils::NetStream::ReturnGet c;
|
||||
while ((c = CommsStream.get()).HasData()) {
|
||||
switch (c.GetData()) {
|
||||
case '$': // start of packet
|
||||
if (packet.size() != 0) {
|
||||
LogMan::Msg::EFmt("Dropping unexpected data: \"{}\"", packet);
|
||||
@@ -203,15 +203,23 @@ fextl::string GdbServer::ReadPacket(std::iostream& stream) {
|
||||
break;
|
||||
case '}': // escape char
|
||||
{
|
||||
char escaped;
|
||||
stream >> escaped;
|
||||
packet.push_back(escaped ^ 0x20);
|
||||
Utils::NetStream::ReturnGet escaped;
|
||||
|
||||
do {
|
||||
escaped = CommsStream.get();
|
||||
} while (!escaped.HasData() && !escaped.HasHangup());
|
||||
|
||||
if (escaped.HasData()) {
|
||||
packet.push_back(escaped.GetData() ^ 0x20);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Received Invalid escape char: ${}", packet);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case '#': // end of packet
|
||||
{
|
||||
char hexString[3] = {0, 0, 0};
|
||||
stream.read(hexString, 2);
|
||||
CommsStream.read(hexString, 2, true);
|
||||
int expected_checksum = std::strtoul(hexString, nullptr, 16);
|
||||
|
||||
if (calculateChecksum(packet) == expected_checksum) {
|
||||
@@ -221,7 +229,7 @@ fextl::string GdbServer::ReadPacket(std::iostream& stream) {
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: packet.push_back((char)c); break;
|
||||
default: packet.push_back(c.GetData()); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -248,22 +256,22 @@ static fextl::string escapePacket(const fextl::string& packet) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
void GdbServer::SendPacket(std::ostream& stream, const fextl::string& packet) {
|
||||
void GdbServer::SendPacket(const fextl::string& packet) {
|
||||
const auto escaped = escapePacket(packet);
|
||||
const auto str = fextl::fmt::format("${}#{:02x}", escaped, calculateChecksum(escaped));
|
||||
|
||||
stream << str << std::flush;
|
||||
CommsStream.SendPacket(str);
|
||||
}
|
||||
|
||||
void GdbServer::SendACK(std::ostream& stream, bool NACK) {
|
||||
void GdbServer::SendACK(bool NACK) {
|
||||
if (NoAckMode) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (NACK) {
|
||||
stream << "-" << std::flush;
|
||||
CommsStream.SendPacket("-");
|
||||
} else {
|
||||
stream << "+" << std::flush;
|
||||
CommsStream.SendPacket("+");
|
||||
}
|
||||
|
||||
if (SettingNoAckMode) {
|
||||
@@ -1341,16 +1349,16 @@ GdbServer::HandledPacketType GdbServer::ProcessPacket(const fextl::string& packe
|
||||
void GdbServer::SendPacketPair(const HandledPacketType& response) {
|
||||
std::lock_guard lk(sendMutex);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_ACK || response.TypeResponse == HandledPacketType::TYPE_ONLYACK) {
|
||||
SendACK(*CommsStream, false);
|
||||
SendACK(false);
|
||||
} else if (response.TypeResponse == HandledPacketType::TYPE_NACK || response.TypeResponse == HandledPacketType::TYPE_ONLYNACK) {
|
||||
SendACK(*CommsStream, true);
|
||||
SendACK(true);
|
||||
}
|
||||
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
|
||||
SendPacket(*CommsStream, "");
|
||||
SendPacket("");
|
||||
} else if (response.TypeResponse != HandledPacketType::TYPE_ONLYNACK && response.TypeResponse != HandledPacketType::TYPE_ONLYACK &&
|
||||
response.TypeResponse != HandledPacketType::TYPE_NONE) {
|
||||
SendPacket(*CommsStream, response.Response);
|
||||
SendPacket(response.Response);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1362,7 +1370,7 @@ GdbServer::WaitForConnectionResult GdbServer::WaitForConnection() {
|
||||
int Result = ppoll(&PollFD, 1, nullptr, nullptr);
|
||||
if (Result > 0) {
|
||||
if (PollFD.revents & POLLIN) {
|
||||
CommsStream = OpenSocket();
|
||||
OpenSocket();
|
||||
return WaitForConnectionResult::CONNECTION;
|
||||
} else if (PollFD.revents & (POLLHUP | POLLERR | POLLNVAL)) {
|
||||
// Listen socket error or shutting down
|
||||
@@ -1392,47 +1400,52 @@ void GdbServer::GdbServerLoop() {
|
||||
|
||||
HandledPacketType response {};
|
||||
|
||||
// Outer server loop. Handles packet start, ACK/NAK and break
|
||||
while (!CoreShuttingDown.load()) {
|
||||
// Outer server loop. Handles packet start, ACK/NAK and break
|
||||
Utils::NetStream::ReturnGet c;
|
||||
while ((c = CommsStream.get()).HasData()) {
|
||||
switch (c.GetData()) {
|
||||
case '$': {
|
||||
auto packet = ReadPacket();
|
||||
response = ProcessPacket(packet);
|
||||
SendPacketPair(response);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown packet {}", packet);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case '+':
|
||||
// ACK, do nothing.
|
||||
break;
|
||||
case '-':
|
||||
// NAK, Resend requested
|
||||
{
|
||||
std::lock_guard lk(sendMutex);
|
||||
SendPacket(response.Response);
|
||||
}
|
||||
break;
|
||||
case '\x03': { // ASCII EOT
|
||||
SyscallHandler->TM.Pause();
|
||||
fextl::string str = fextl::fmt::format("T02thread:{:02x};", getpid());
|
||||
if (LibraryMapChanged) {
|
||||
// If libraries have changed then let gdb know
|
||||
str += "library:1;";
|
||||
}
|
||||
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("GdbServer: Unexpected byte {} ({:02x})", c.GetData(), c.GetData());
|
||||
}
|
||||
}
|
||||
|
||||
int c;
|
||||
while ((c = CommsStream->get()) >= 0) {
|
||||
switch (c) {
|
||||
case '$': {
|
||||
auto packet = ReadPacket(*CommsStream);
|
||||
response = ProcessPacket(packet);
|
||||
SendPacketPair(response);
|
||||
if (response.TypeResponse == HandledPacketType::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown packet {}", packet);
|
||||
}
|
||||
if (c.HasHangup()) {
|
||||
break;
|
||||
}
|
||||
case '+':
|
||||
// ACK, do nothing.
|
||||
break;
|
||||
case '-':
|
||||
// NAK, Resend requested
|
||||
{
|
||||
std::lock_guard lk(sendMutex);
|
||||
SendPacket(*CommsStream, response.Response);
|
||||
}
|
||||
break;
|
||||
case '\x03': { // ASCII EOT
|
||||
SyscallHandler->TM.Pause();
|
||||
fextl::string str = fextl::fmt::format("T02thread:{:02x};", getpid());
|
||||
if (LibraryMapChanged) {
|
||||
// If libraries have changed then let gdb know
|
||||
str += "library:1;";
|
||||
}
|
||||
SendPacketPair({std::move(str), HandledPacketType::TYPE_ACK});
|
||||
break;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("GdbServer: Unexpected byte {} ({:02x})", static_cast<char>(c), c);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
std::lock_guard lk(sendMutex);
|
||||
CommsStream.reset();
|
||||
CommsStream.InvalidateSocket();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1504,14 +1517,14 @@ void GdbServer::CloseListenSocket() {
|
||||
unlink(GdbUnixSocketPath.c_str());
|
||||
}
|
||||
|
||||
fextl::unique_ptr<std::iostream> GdbServer::OpenSocket() {
|
||||
void GdbServer::OpenSocket() {
|
||||
// Block until a connection arrives
|
||||
struct sockaddr_storage their_addr {};
|
||||
socklen_t addr_size {};
|
||||
|
||||
int new_fd = accept(ListenSocket, (struct sockaddr*)&their_addr, &addr_size);
|
||||
|
||||
return fextl::make_unique<FEXCore::Utils::NetStream>(new_fd);
|
||||
CommsStream.OpenSocket(new_fd);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -14,11 +14,11 @@ $end_info$
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <istream>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <stdint.h>
|
||||
|
||||
#include "LinuxSyscalls/NetStream.h"
|
||||
#include "LinuxSyscalls/SignalDelegator.h"
|
||||
|
||||
namespace FEX {
|
||||
@@ -45,12 +45,12 @@ private:
|
||||
ERROR,
|
||||
};
|
||||
WaitForConnectionResult WaitForConnection();
|
||||
fextl::unique_ptr<std::iostream> OpenSocket();
|
||||
void OpenSocket();
|
||||
void StartThread();
|
||||
fextl::string ReadPacket(std::iostream& stream);
|
||||
void SendPacket(std::ostream& stream, const fextl::string& packet);
|
||||
fextl::string ReadPacket();
|
||||
void SendPacket(const fextl::string& packet);
|
||||
|
||||
void SendACK(std::ostream& stream, bool NACK);
|
||||
void SendACK(bool NACK);
|
||||
|
||||
Event ThreadBreakEvent {};
|
||||
void WaitForThreadWakeup();
|
||||
@@ -147,7 +147,7 @@ private:
|
||||
FEX::HLE::SyscallHandler* const SyscallHandler;
|
||||
FEX::HLE::SignalDelegator* SignalDelegation;
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
fextl::unique_ptr<std::iostream> CommsStream;
|
||||
FEX::Utils::NetStream CommsStream;
|
||||
std::mutex sendMutex;
|
||||
bool SettingNoAckMode {false};
|
||||
bool NoAckMode {false};
|
||||
|
||||
Loaded 100 of 140 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user