FEXCore: Moves CodeEmitter to FHU

Now that the vixl dependency is gone, this gets moved to FHU since the
frontend is going to need it for a microjit.
This commit is contained in:
Ryan Houdek committed 2024-05-13 12:48:10 -07:00
1 parent 1f40590f9a
commit 9e1840e974
36 files changed
+3308 -3328

No files matched your search

+2
View File
@@ -0,0 +1,2 @@
add_library(CodeEmitter INTERFACE)
target_include_directories(CodeEmitter INTERFACE .)
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
+341
View File
@@ -0,0 +1,341 @@
// SPDX-License-Identifier: MIT
/* Branch instruction emitters.
*
* Most of these instructions will use `BackwardLabel`, `ForwardLabel`, or `BiDirectionLabel` to determine where a branch targets.
*/
public:
// Branches, Exception Generating and System instructions
public:
// Conditional branch immediate
///< Branch conditional
void b(ARMEmitter::Condition Cond, uint32_t Imm) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm);
}
void b(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 0, Cond, 0);
}
void b(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
b(Cond, &Label->Backward);
}
else {
b(Cond, &Label->Forward);
}
}
///< Branch consistent conditional
void bc(ARMEmitter::Condition Cond, uint32_t Imm) {
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm);
}
void bc(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bc(ARMEmitter::Condition Cond, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
constexpr uint32_t Op = 0b0101'010 << 25;
Branch_Conditional(Op, 0, 1, Cond, 0);
}
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
bc(Cond, &Label->Backward);
}
else {
bc(Cond, &Label->Forward);
}
}
// Unconditional branch register
void br(ARMEmitter::Register rn) {
constexpr uint32_t Op = 0b1101011 << 25 |
0b0'000 << 21 | // opc
0b1'1111 << 16 | // op2
0b0000'00 << 10 | // op3
0b0'0000; // op4
UnconditionalBranch(Op, rn);
}
void blr(ARMEmitter::Register rn) {
constexpr uint32_t Op = 0b1101011 << 25 |
0b0'001 << 21 | // opc
0b1'1111 << 16 | // op2
0b0000'00 << 10 | // op3
0b0'0000; // op4
UnconditionalBranch(Op, rn);
}
void ret(ARMEmitter::Register rn = ARMEmitter::Reg::r30) {
constexpr uint32_t Op = 0b1101011 << 25 |
0b0'010 << 21 | // opc
0b1'1111 << 16 | // op2
0b0000'00 << 10 | // op3
0b0'0000; // op4
UnconditionalBranch(Op, rn);
}
// Unconditional branch immediate
void b(uint32_t Imm) {
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, Imm);
}
void b(BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void b(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
constexpr uint32_t Op = 0b0001'01 << 26;
UnconditionalBranch(Op, 0);
}
void b(BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
b(&Label->Backward);
}
else {
b(&Label->Forward);
}
}
void bl(uint32_t Imm) {
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, Imm);
}
void bl(BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void bl(LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::B });
constexpr uint32_t Op = 0b1001'01 << 26;
UnconditionalBranch(Op, 0);
}
void bl(BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
bl(&Label->Backward);
}
else {
bl(&Label->Forward);
}
}
// Compare and branch
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, Imm);
}
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
constexpr uint32_t Op = 0b0011'0100 << 24;
CompareAndBranch(Op, s, rt, 0);
}
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
cbz(s, rt, &Label->Backward);
}
else {
cbz(s, rt, &Label->Forward);
}
}
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, Imm);
}
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
constexpr uint32_t Op = 0b0011'0101 << 24;
CompareAndBranch(Op, s, rt, 0);
}
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
cbnz(s, rt, &Label->Backward);
}
else {
cbnz(s, rt, &Label->Forward);
}
}
// Test and branch immediate
void tbz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, Imm);
}
void tbz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0110 << 24;
TestAndBranch(Op, rt, Bit, 0);
}
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
tbz(rt, Bit, &Label->Backward);
}
else {
tbz(rt, Bit, &Label->Forward);
}
}
void tbnz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, Imm);
}
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, Imm >> 2);
}
template<typename LabelType>
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
void tbnz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
constexpr uint32_t Op = 0b0011'0111 << 24;
TestAndBranch(Op, rt, Bit, 0);
}
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
if (Label->Backward.Location) {
tbnz(rt, Bit, &Label->Backward);
}
else {
tbnz(rt, Bit, &Label->Forward);
}
}
private:
// Conditional branch immediate
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, ARMEmitter::Condition Cond, uint32_t Imm) {
uint32_t Instr = Op;
Instr |= Op1 << 24;
Instr |= (Imm & 0x7'FFFF) << 5;
Instr |= Op0 << 4;
Instr |= FEXCore::ToUnderlying(Cond);
dc32(Instr);
}
// Unconditional branch register
void UnconditionalBranch(uint32_t Op, ARMEmitter::Register rn) {
uint32_t Instr = Op;
Instr |= Encode_rn(rn);
dc32(Instr);
}
// Unconditional branch - immediate
void UnconditionalBranch(uint32_t Op, uint32_t Imm) {
uint32_t Instr = Op;
Instr |= Imm & 0x3FF'FFFF;
dc32(Instr);
}
// Compare and branch
void CompareAndBranch(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
uint32_t Instr = Op;
Instr |= SF;
Instr |= (Imm & 0x7'FFFF) << 5;
Instr |= Encode_rt(rt);
dc32(Instr);
}
// Test and branch - immediate
void TestAndBranch(uint32_t Op, ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
uint32_t Instr = Op;
Instr |= (Bit >> 5) << 31;
Instr |= (Bit & 0b1'1111) << 19;
Instr |= (Imm & 0x3FFF) << 5;
Instr |= Encode_rt(rt);
dc32(Instr);
}
+106
View File
@@ -0,0 +1,106 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <cstddef>
#include <cstdint>
#include <cstring>
namespace ARMEmitter {
class Buffer {
public:
Buffer() {
SetBuffer(nullptr, 0);
}
Buffer(uint8_t* Base, uint64_t BaseSize) {
SetBuffer(Base, BaseSize);
}
void SetBuffer(uint8_t* Base, uint64_t BaseSize) {
BufferBase = Base;
CurrentOffset = BufferBase;
Size = BaseSize;
}
void dc8(uint8_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc16(uint16_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc32(uint32_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void dc64(uint64_t Data) {
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
*Memory = Data;
CurrentOffset += sizeof(Data);
}
void EmitString(const char* String) {
const auto StringLength = strlen(String);
memcpy(CurrentOffset, String, StringLength);
CurrentOffset += StringLength;
}
void Align() {
// Align the buffer to instruction size
auto CurrentAlignment = reinterpret_cast<uint64_t>(CurrentOffset) & 0b11;
if (!CurrentAlignment) {
return;
}
CurrentOffset += 4 - CurrentAlignment;
}
template<typename T>
T GetCursorAddress() const {
return reinterpret_cast<T>(CurrentOffset);
}
static void ClearICache(void* Begin, std::size_t Length) {
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
}
size_t GetCursorOffset() const {
return static_cast<size_t>(CurrentOffset - BufferBase);
}
uint8_t* GetBufferBase() const {
return BufferBase;
}
void CursorIncrement(size_t Size) {
CurrentOffset += Size;
}
void SetCursorOffset(size_t Offset) {
CurrentOffset = BufferBase + Offset;
}
uint64_t GetBufferSize() const {
return Size;
}
template<typename T>
size_t GetCursorOffsetFromAddress(const T* Address) const {
return static_cast<size_t>(reinterpret_cast<const uint8_t*>(Address) - BufferBase);
}
protected:
void ResetBuffer() {
CurrentOffset = BufferBase;
}
uint8_t* BufferBase;
uint8_t* CurrentOffset;
uint64_t Size;
};
} // namespace ARMEmitter
+905
View File
@@ -0,0 +1,905 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/fextl/vector.h>
#include <FEXHeaderUtils/BitUtils.h>
#include <CodeEmitter/Buffer.h>
#include <CodeEmitter/Registers.h>
#include <array>
#include <cstdint>
#include <utility>
#include <type_traits>
/*
* Welcome to FEX-Emu's custom AArch64 emitter.
* This was written specifically to avoid the performance cost of the vixl emitter.
*
* There are some specific design constraints in this design to target a couple features:
* - High performance
* - Low CPU cache performance hit
* - Significantly reduced code footprint
* - Low number of branches
*
* These requirements are mostly achieved by removing a bunch of developer conveniences
* that vixl provides. The developer needs to take a lot of care to not shoot themselves in the foot.
*
* Misc design decisions:
* - Registers are encoded as basic uint32_t enums.
* - Converting between different registers is zero-cost.
* - Passing around as arguments are as cheap as registers
* - Contrast to vixl where every register requires living on the stack.
* - Registers can get encoded in to instructions with a simple `BFM` instruction.
*
* - Instructions are very simply emitted, allowing direct inlining most of the time.
* - These are simple enough that multiple back-to-back instructions get optimized to 128-bit load-store operations.
* - Contrast to vixl where pretty much no instruction emitter gets inlined.
*
* - Instruction emitters are /mostly/ unsized. Most instructions take a size argument first, which gets encoded
* directly in to the instruction.
* - Contrast to vixl where the register arguments are how the instructions determine operating size.
* - Size argument allows FEX to use `CSEL` to select a size at runtime, instead of branching.
* - Some instructions are explicitly sized based on register type. Read comments in the respective `inl` files to
* see why.
* Some scalar/vector operations are an example of this.
*
* - Almost zero helper functions.
* - Primary exception to this rule is load-store operations. These will use a helper to make
* it easier to select the correct load-store instruction. Mostly because these are a nightmare selecting
* the right instruction.
*/
namespace ARMEmitter {
/*
* This `Size` enum is used for most ALU operations.
* These follow the AArch64 encoding style in most cases.
*/
enum class Size : uint32_t {
i32Bit = 0,
i64Bit,
};
// This allows us to get the `Size` enum in bits.
[[nodiscard]]
constexpr size_t RegSizeInBits(Size size) {
return size_t {32} << FEXCore::ToUnderlying(size);
}
/* This `SubRegSize` enum is used for most ASIMD operations.
* These follow the AArch64 encoding style in most cases.
*/
enum class SubRegSize : uint32_t {
i8Bit = 0b00,
i16Bit = 0b01,
i32Bit = 0b10,
i64Bit = 0b11,
i128Bit = 0b100,
};
// This allows us to get the `SubRegSize` in bits.
[[nodiscard]]
constexpr size_t SubRegSizeInBits(SubRegSize size) {
return size_t {8} << FEXCore::ToUnderlying(size);
}
/* This `ScalarRegSize` enum is used for most scalar float
* operations.
*
* This is specifically duplicated from `SubRegSize` to have strongly
* typed functions.
*
* `ScalarRegSize` specifically doesn't have `i128Bit` because scalar operations
* can't operate at 128-bit.
*/
enum class ScalarRegSize : uint32_t {
i8Bit = 0b00,
i16Bit = 0b01,
i32Bit = 0b10,
i64Bit = 0b11,
};
// This allows us to get the `ScalarRegSize` in bits.
[[nodiscard]]
constexpr size_t ScalarRegSizeInBits(ScalarRegSize size) {
return size_t {8} << FEXCore::ToUnderlying(size);
}
/* This `VectorRegSizePair` union allows us to have an overlapping type
* to select a scalar operation or a vector depending on which operation
* we pass in.
* Useful in FEX's vector operations that behave as scalar or vector
* depending on various factors. But since the operation will have the sa,e
* element size, we want to choose the operation more easily
*/
union VectorRegSizePair {
ScalarRegSize Scalar;
SubRegSize Vector;
};
// This allows us to create a `VectorRegSizePair` union.
[[nodiscard]]
constexpr VectorRegSizePair ToVectorSizePair(SubRegSize size) {
return VectorRegSizePair {.Vector = size};
}
[[nodiscard]]
constexpr VectorRegSizePair ToVectorSizePair(ScalarRegSize size) {
return VectorRegSizePair {.Scalar = size};
}
// This `ShiftType` enum is used for ALU shift-register encoded instructions.
enum class ShiftType : uint32_t {
LSL = 0,
LSR,
ASR,
ROR,
};
// This `ExtendedType` enum is used for ALU extended-register encoded instructions.
enum class ExtendedType : uint32_t {
UXTB = 0b000,
UXTH = 0b001,
UXTW = 0b010,
UXTX = 0b011,
SXTB = 0b100,
SXTH = 0b101,
SXTW = 0b110,
SXTX = 0b111,
LSL_32 = UXTW,
LSL_64 = UXTX,
};
// This `Condition` enum is used for various conditional instructions.
enum class Condition : uint32_t {
// Meaning: Int - Float
CC_EQ = 0, // Equal - Equal
CC_NE, // Not Eq - Not Eq or unordered
CC_CS, // Carry set - Greater than, equal, or unordered
CC_CC, // Carry clear - Less than
CC_MI, // Minus/Negative - Less than
CC_PL, // Plus, positive or zero - GT, equal, or unordered
CC_VS, // Overflow - Unordered
CC_VC, // No Overflow - Ordered
CC_HI, // Unsigned higher - GT, or unordered
CC_LS, // Unsigned lower or same - LT or EQ
CC_GE, // Signed GT or EQ - GT or EQ
CC_LT, // Signed LT - LT or Unordered
CC_GT, // Signed GT - GT
CC_LE, // Signed LT or EQ - LT, EQ, or Unordered
CC_AL, // Always - Always
CC_NV, // Always - Always
// Aliases
CC_HS = CC_CS,
CC_LO = CC_CC,
};
/*
* This `StatusFlags` enum is used for conditional compare encoded instructions.
* These directly encode to the `nzcv` flags.
*/
enum class StatusFlags : uint32_t {
None = 0,
Flag_V = 0b0001,
Flag_C = 0b0010,
Flag_Z = 0b0100,
Flag_N = 0b1000,
Flag_NZCV = Flag_N | Flag_Z | Flag_C | Flag_V,
};
/*
* This `IndexType` enum is used for load-store instructions.
* Not all load-store instructions use this, so the user needs to be careful.
*/
enum class IndexType {
POST,
OFFSET,
PRE,
UNPRIVILEGED,
};
// Used with adr and scalar + vector load/store variants to denote
// a modifier operation.
enum class SVEModType : uint8_t {
MOD_UXTW,
MOD_SXTW,
MOD_LSL,
MOD_NONE,
};
/* This `SVEMemOperand` class is used for the helper SVE load-store instructions.
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
*/
class SVEMemOperand final {
public:
enum class Type {
ScalarPlusScalar,
ScalarPlusImm,
ScalarPlusVector,
VectorPlusImm,
};
SVEMemOperand(XRegister rn, XRegister rm = XReg::zr)
: rn {rn}
, MemType {Type::ScalarPlusScalar}
, MetaType {.ScalarScalarType {
.rm = rm,
}} {}
SVEMemOperand(XRegister rn, int32_t imm = 0)
: rn {rn}
, MemType {Type::ScalarPlusImm}
, MetaType {.ScalarImmType {
.Imm = imm,
}} {}
SVEMemOperand(XRegister rn, ZRegister zm, SVEModType mod = SVEModType::MOD_NONE, uint8_t scale = 0)
: rn {rn}
, MemType {Type::ScalarPlusVector}
, MetaType {.ScalarVectorType {
.zm = zm,
.mod = mod,
.scale = scale,
}} {}
SVEMemOperand(ZRegister zn, uint32_t imm)
: rn {Register {zn.Idx()}}
, MemType {Type::VectorPlusImm}
, MetaType {.VectorImmType {
.Imm = imm,
}} {}
[[nodiscard]]
bool IsScalarPlusScalar() const {
return MemType == Type::ScalarPlusScalar;
}
[[nodiscard]]
bool IsScalarPlusImm() const {
return MemType == Type::ScalarPlusImm;
}
[[nodiscard]]
bool IsScalarPlusVector() const {
return MemType == Type::ScalarPlusVector;
}
[[nodiscard]]
bool IsVectorPlusImm() const {
return MemType == Type::VectorPlusImm;
}
union Data {
struct {
Register rm;
} ScalarScalarType;
struct {
int32_t Imm;
} ScalarImmType;
struct {
ZRegister zm;
SVEModType mod;
uint8_t scale;
} ScalarVectorType;
struct {
// rn will be a ZRegister
uint32_t Imm;
} VectorImmType;
};
Register rn;
Type MemType;
Data MetaType;
};
/* This `ExtendedMemOperand` class is used for the helper load-store instructions.
* Load-store instructions are quite expressive, so having a helper that handles these differences is worth it.
*/
class ExtendedMemOperand final {
public:
ExtendedMemOperand(XRegister rn, XRegister rm = XReg::zr, ExtendedType Option = ExtendedType::LSL_64, uint32_t Shift = 0)
: rn {rn}
, MetaType {.ExtendedType {
.Header = {.MemType = TYPE_EXTENDED},
.rm = rm,
.Option = Option,
.Shift = Shift,
}} {}
ExtendedMemOperand(XRegister rn, IndexType Index = IndexType::OFFSET, int32_t Imm = 0)
: rn {rn}
, MetaType {.ImmType {
.Header = {.MemType = TYPE_IMM},
.Index = Index,
.Imm = Imm,
}} {}
Register rn;
enum Type {
TYPE_EXTENDED,
TYPE_IMM,
};
struct HeaderStruct {
Type MemType;
};
union {
HeaderStruct Header;
struct {
HeaderStruct Header;
Register rm;
ExtendedType Option;
uint32_t Shift;
} ExtendedType;
struct {
HeaderStruct Header;
IndexType Index;
int32_t Imm;
} ImmType;
} MetaType;
};
template<uint32_t op0, uint32_t op1, uint32_t CRn, uint32_t CRm, uint32_t op2>
constexpr uint32_t GenSystemReg() {
return op0 << 19 | op1 << 16 | CRn << 12 | CRm << 8 | op2 << 5;
};
// This `SystemRegister` enum is used for the mrs/msr instructions.
enum class SystemRegister : uint32_t {
CTR_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b001>(),
DCZID_EL0 = GenSystemReg<0b11, 0b011, 0b0000, 0b0000, 0b111>(),
TPIDR_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b010>(),
RNDR = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b000>(),
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>(),
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>(),
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>(),
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>(),
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>(),
};
template<uint32_t op1, uint32_t CRm, uint32_t op2>
constexpr uint32_t GenDCReg() {
return op1 << 16 | CRm << 8 | op2 << 5;
};
// This `DataCacheOperation` enum is used for the dc instruction.
enum class DataCacheOperation : uint32_t {
IVAC = GenDCReg<0b000, 0b0110, 0b001>(),
ISW = GenDCReg<0b000, 0b0110, 0b010>(),
CSW = GenDCReg<0b000, 0b1010, 0b010>(),
CISW = GenDCReg<0b000, 0b1110, 0b010>(),
ZVA = GenDCReg<0b011, 0b0100, 0b001>(),
CVAC = GenDCReg<0b011, 0b1010, 0b001>(),
CVAU = GenDCReg<0b011, 0b1011, 0b001>(),
CIVAC = GenDCReg<0b011, 0b1110, 0b001>(),
// MTE2
IGVAC = GenDCReg<0b000, 0b0110, 0b011>(),
IGSW = GenDCReg<0b000, 0b0110, 0b100>(),
IGDVAC = GenDCReg<0b000, 0b0110, 0b101>(),
IGDSW = GenDCReg<0b000, 0b0110, 0b110>(),
CGSW = GenDCReg<0b000, 0b1010, 0b100>(),
CGDSW = GenDCReg<0b000, 0b1010, 0b110>(),
CIGSW = GenDCReg<0b000, 0b1110, 0b100>(),
CIGDSW = GenDCReg<0b000, 0b1110, 0b110>(),
// MTE
GVA = GenDCReg<0b011, 0b0100, 0b011>(),
GZVA = GenDCReg<0b011, 0b0100, 0b100>(),
CGVAC = GenDCReg<0b011, 0b1010, 0b011>(),
CGDVAC = GenDCReg<0b011, 0b1010, 0b101>(),
CGVAP = GenDCReg<0b011, 0b1100, 0b011>(),
CGDVAP = GenDCReg<0b011, 0b1100, 0b101>(),
CGVADP = GenDCReg<0b011, 0b1101, 0b011>(),
CGDVADP = GenDCReg<0b011, 0b1101, 0b101>(),
CIGVAC = GenDCReg<0b011, 0b1110, 0b011>(),
CIGDVAC = GenDCReg<0b011, 0b1110, 0b101>(),
// DPB
CVAP = GenDCReg<0b011, 0b1100, 0b001>(),
// DPB2
CVADP = GenDCReg<0b011, 0b1101, 0b001>(),
};
template<uint32_t CRm, uint32_t op2>
constexpr uint32_t GenHintBarrierReg() {
return CRm << 8 | op2 << 5;
}
// This `HintRegister` enum is used for the hint instruction.
enum class HintRegister : uint32_t {
NOP = GenHintBarrierReg<0b0000, 0b000>(),
YIELD = GenHintBarrierReg<0b0000, 0b001>(),
WFE = GenHintBarrierReg<0b0000, 0b010>(),
WFI = GenHintBarrierReg<0b0000, 0b011>(),
SEV = GenHintBarrierReg<0b0000, 0b100>(),
SEVL = GenHintBarrierReg<0b0000, 0b101>(),
DGH = GenHintBarrierReg<0b0000, 0b110>(),
CSDB = GenHintBarrierReg<0b0010, 0b100>(),
};
// This `BarrierRegister` enum is used for the various barrier instructions.
enum class BarrierRegister : uint32_t {
CLREX = GenHintBarrierReg<0b0000, 0b010>(),
TCOMMIT = GenHintBarrierReg<0b0000, 0b011>(),
DSB = GenHintBarrierReg<0b0000, 0b100>(),
DMB = GenHintBarrierReg<0b0000, 0b101>(),
ISB = GenHintBarrierReg<0b0000, 0b110>(),
SB = GenHintBarrierReg<0b0000, 0b111>(),
};
// This `BarrierScope` enum is used for the dsb/dmb instructions.
enum class BarrierScope : uint32_t {
// Outer shareable
OSHLD = 0b0001,
OSHST = 0b0010,
OSH = 0b0011,
// Non shareable
NSHLD = 0b0101,
NSHST = 0b0110,
NSH = 0b0111,
// Inner shareable
ISHLD = 0b1001,
ISHST = 0b1010,
ISH = 0b1011,
// Full System visibility
LD = 0b1101,
ST = 0b1110,
SY = 0b1111,
};
// This `Prefetch` enum is used for prefetch instructions.
enum class Prefetch : uint32_t {
// Prefetch for load
PLDL1KEEP = 0b00000,
PLDL1STRM = 0b00001,
PLDL2KEEP = 0b00010,
PLDL2STRM = 0b00011,
PLDL3KEEP = 0b00100,
PLDL3STRM = 0b00101,
// Preload instructions
PLIL1KEEP = 0b01000,
PLIL1STRM = 0b01001,
PLIL2KEEP = 0b01010,
PLIL2STRM = 0b01011,
PLIL3KEEP = 0b01100,
PLIL3STRM = 0b01101,
// Preload for store
PSTL1KEEP = 0b10000,
PSTL1STRM = 0b10001,
PSTL2KEEP = 0b10010,
PSTL2STRM = 0b10011,
PSTL3KEEP = 0b10100,
PSTL3STRM = 0b10101,
};
// This `PredicatePattern` enun is used for some SVE instructions.
enum class PredicatePattern : uint32_t {
SVE_POW2 = 0b00000,
SVE_VL1 = 0b00001,
SVE_VL2 = 0b00010,
SVE_VL3 = 0b00011,
SVE_VL4 = 0b00100,
SVE_VL5 = 0b00101,
SVE_VL6 = 0b00110,
SVE_VL7 = 0b00111,
SVE_VL8 = 0b01000,
SVE_VL16 = 0b01001,
SVE_VL32 = 0b01010,
SVE_VL64 = 0b01011,
SVE_VL128 = 0b01100,
SVE_VL256 = 0b01101,
SVE_MUL4 = 0b11101,
SVE_MUL3 = 0b11110,
SVE_ALL = 0b11111,
};
// Used with SVE FP immediate arithmetic instructions
enum class SVEFAddSubImm : uint32_t {
_0_5,
_1_0,
};
enum class SVEFMulImm : uint32_t {
_0_5,
_2_0,
};
enum class SVEFMaxMinImm : uint32_t {
_0_0,
_1_0,
};
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is logically `below` an instruction that uses it.
* Which means that a branch would jump backwards.
*/
struct BackwardLabel {
uint8_t* Location {};
};
/* This `SingleUseForwardLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is logically `above` an instruction that uses it.
* Which means that a branch would jump forwards.
*
* The `ForwardLabel` struct can be bound to multiple instructions, so it needs a vector for each bind instruction type.
*/
struct SingleUseForwardLabel {
enum class InstType {
UNKNOWN,
ADR,
ADRP,
B,
BC,
TEST_BRANCH,
RELATIVE_LOAD,
LONG_ADDRESS_GEN,
};
uint8_t* Location {};
InstType Type = InstType::UNKNOWN;
};
struct ForwardLabel {
fextl::vector<SingleUseForwardLabel> Insts {};
};
/* This `BiDirectionalLabel` struct used for retaining a location for PC-Relative instructions.
* This is specifically a label for a target that is in either direction of an instruction that uses it.
* Which means a branch could jump backwards or forwards depending on situation.
*/
struct BiDirectionalLabel {
BackwardLabel Backward;
ForwardLabel Forward;
};
static inline void AddLocationToLabel(SingleUseForwardLabel* Label, SingleUseForwardLabel&& Location) {
LOGMAN_THROW_A_FMT(Label->Type == SingleUseForwardLabel::InstType::UNKNOWN, "Trying to bind a SingleUseForwardLabel to multiple "
"locations. Use ForwardLabel instead.");
*Label = std::move(Location);
}
static inline void AddLocationToLabel(ForwardLabel* Label, SingleUseForwardLabel&& Location) {
Label->Insts.emplace_back(std::move(Location));
}
// Some FCMA ASIMD instructions support a rotation argument.
enum class Rotation : uint32_t {
ROTATE_0 = 0b00,
ROTATE_90 = 0b01,
ROTATE_180 = 0b10,
ROTATE_270 = 0b11,
};
// Concept for contraining some instructions to accept only an XRegister or WRegister.
// Particularly for operations that differ encodings depending on which one is used.
template<typename T>
concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegister>;
// Whether or not a given set of vector registers are sequential
// in increasing order as far as the register file is concerned (modulo its size)
//
// For example, a set of registers like:
//
// v1, v2, v3 and
// v31, v0, v1
//
// would both be considered sequential sequences, and some instructions in particular
// limit register lists to these kind of sequences.
//
template<typename T, typename... Args>
constexpr bool AreVectorsSequential(T first, const Args&... args) {
// Ensure we always have a pair of registers to compare against.
static_assert(sizeof...(args) >= 1, "Number of arguments must be greater than 1");
const auto fn = [](auto& lhs, const auto& rhs) {
const auto result = ((lhs.Idx() + 1) % 32) == rhs.Idx();
lhs = rhs;
return result;
};
return (fn(first, args) && ...);
}
// This is an emitter that is designed around the smallest code bloat as possible.
// Eschewing most developer convenience in order to keep code as small as possible.
// Choices:
// - Size of ops passed as an argument rather than template to let the compiler use csel instead of branching.
// - Registers are unsized so they can be passed in a GPR and not need conversion operations
class Emitter : public ARMEmitter::Buffer {
public:
Emitter() = default;
Emitter(uint8_t* Base, uint64_t BaseSize)
: Buffer(Base, BaseSize) {}
// Bind a backward label to an address.
// Address that is bound is the current emitter location.
void Bind(BackwardLabel* Label) {
LOGMAN_THROW_AA_FMT(Label->Location == nullptr, "Trying to bind a label twice");
Label->Location = GetCursorAddress<uint8_t*>();
}
void Bind(const SingleUseForwardLabel* Label) {
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
// Patch up the instructions
switch (Label->Type) {
case SingleUseForwardLabel::InstType::ADR: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::ADRP: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
Imm >>= 12;
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= (Offset & 0b11) << 29;
Inst |= (Offset >> 2) << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::B: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FF'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~InstMask;
Inst |= Offset;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::TEST_BRANCH: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x3FFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::BC:
case SingleUseForwardLabel::InstType::RELATIVE_LOAD: {
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
Imm >>= 2;
uint32_t InstMask = 0x7'FFFF;
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
uint32_t Inst = *Instruction & ~(InstMask << 5);
Inst |= Offset << 5;
*Instruction = Inst;
break;
}
case SingleUseForwardLabel::InstType::LONG_ADDRESS_GEN: {
uint32_t* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
auto OriginalOffset = GetCursorOffset();
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
SetCursorOffset(InstOffset);
// We encoded the destination register in to the first instruction space.
// Read it back.
ARMEmitter::Register DestReg(Instructions[0]);
if (IsADRRange(ImmInstTwo)) {
// If within ADR range from the second instruction, then we can emit NOP+ADR
nop();
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
} else if (IsADRPRange(ImmInstOne)) {
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
// First check if we are in non-offset range for second instruction.
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
// We can emit nop + adrp
nop();
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
} else {
// Not aligned, need adrp + add
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
}
} else {
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
FEX_UNREACHABLE;
}
SetCursorOffset(OriginalOffset);
break;
}
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
}
}
// Bind a forward label to a location.
// This walks all the instructions in the label's vector.
// Then backpatching all instructions that have used the label.
template<bool WarnAboutEmpty = false>
void Bind(ForwardLabel* Label) {
if constexpr (WarnAboutEmpty) {
LOGMAN_THROW_A_FMT(Label->Insts.empty() == false, "Binding forward label that didn't have any instructions using it");
}
for (auto& Inst : Label->Insts) {
Bind(&Inst);
}
}
// Bind a bidirectional location to a location.
// Binds both forwards and backwards depending on how the label was used.
void Bind(BiDirectionalLabel* Label) {
if (!Label->Backward.Location) {
Bind(&Label->Backward);
}
Bind<false>(&Label->Forward);
}
class Float16 final {
public:
// Float16 format:
// Bit[15] - Sign
// Bit[14:10] - Exponent
// Bit[9:0] - Fraction
Float16(double Value) {
const auto AsBits = ToBits(Value);
const uint16_t Sign = AsBits >> 63;
const uint16_t Exponent = (AsBits >> 52) & 0b0111'1111'1111;
const uint64_t Fraction = AsBits & ((1ULL << 52) - 1);
switch (std::fpclassify(Value)) {
case FP_NORMAL: {
// 11-bit Exponent converted down to 5-bits.
// - Exponent biased by 1023 in 64-bit.
// - Only biased by 15.
const uint16_t UnbiasedExponent = Exponent - 1023;
const uint16_t BiasedFP16Exponent = UnbiasedExponent + 15;
LOGMAN_THROW_A_FMT((BiasedFP16Exponent & ~0b1'1111U) == 0, "Exponent too large to fit in to raw bits");
// 52-bit Fraction converted down to 10-bits.
uint32_t FractionAdjusted = Fraction >> 42; // Only retain upper 10-bits
if ((Fraction >> 41) & 1) {
// Round up if the bottom bit being shifted out is set.
FractionAdjusted += 1;
}
RawBits = (Sign << 15) | (BiasedFP16Exponent << 10) | FractionAdjusted;
break;
}
case FP_ZERO:
// Positive or negative zero.
// Exponent and Fraction as zero.
RawBits = Sign << 15;
break;
case FP_INFINITE:
// Positive or negative infinite
// Exponent is fully set, Fraction must be zero.
RawBits = (Sign << 15) | (0b1'1111 << 10);
break;
case FP_SUBNORMAL:
case FP_NAN:
// TODO: Can be encoded if necessary, but only handle when necessary.
default: LOGMAN_MSG_A_FMT("Invalid FP type");
}
}
uint8_t ToImm8() const {
// ARM imm8 float encoding
// Bit[7] - Sign
// Bit[6] - Exponent
// Bit[5:0] - Fraction
uint8_t Result {};
// Sign bit
Result |= ((RawBits >> 15) & 1) << 7;
// Exponent - Cuts off the bottom 4 bits and the top 1 bit.
Result |= ((RawBits >> 13) & 1) << 6;
// Bottom Exponent & Fraction (Cuts off bottom 6 bits)
Result |= (RawBits >> 6) & 0b11'1111;
return Result;
}
uint16_t RawBits {};
private:
uint64_t ToBits(double Value) {
uint64_t Result {};
memcpy(&Result, &Value, sizeof(Value));
return Result;
}
};
#include <CodeEmitter/VixlUtils.inl>
public:
// TODO: Implement SME when it matters.
#include <CodeEmitter/ALUOps.inl>
#include <CodeEmitter/BranchOps.inl>
#include <CodeEmitter/LoadstoreOps.inl>
#include <CodeEmitter/SystemOps.inl>
#include <CodeEmitter/ScalarOps.inl>
#include <CodeEmitter/ASIMDOps.inl>
#include <CodeEmitter/SVEOps.inl>
private:
template<typename T>
uint32_t Encode_ra(T Reg) const {
return Reg.Idx() << 10;
}
uint32_t Encode_ra(uint32_t Reg) const {
return Reg << 10;
}
template<typename T>
uint32_t Encode_rt2(T Reg) const {
return Reg.Idx() << 10;
}
template<>
uint32_t Encode_rt2(uint32_t Reg) const {
return Reg << 10;
}
template<typename T>
uint32_t Encode_rm(T Reg) const {
return Reg.Idx() << 16;
}
uint32_t Encode_rm(uint32_t Reg) const {
return Reg << 16;
}
template<typename T>
uint32_t Encode_rs(T Reg) const {
return Reg.Idx() << 16;
}
uint32_t Encode_rs(uint32_t Reg) const {
return Reg << 16;
}
template<typename T>
uint32_t Encode_rn(T Reg) const {
return Reg.Idx() << 5;
}
uint32_t Encode_rn(uint32_t Reg) const {
return Reg << 5;
}
template<typename T>
uint32_t Encode_rd(T Reg) const {
return Reg.Idx();
}
uint32_t Encode_rd(uint32_t Reg) const {
return Reg;
}
template<typename T>
uint32_t Encode_rt(T Reg) const {
return Reg.Idx();
}
template<>
uint32_t Encode_rt(Prefetch Reg) const {
return FEXCore::ToUnderlying(Reg);
}
uint32_t Encode_rt(uint32_t Reg) const {
return Reg;
}
template<typename T>
uint32_t Encode_pd(T Reg) const {
return FEXCore::ToUnderlying(Reg);
}
};
} // namespace ARMEmitter
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
+176
View File
@@ -0,0 +1,176 @@
// SPDX-License-Identifier: MIT
/* System instruction emitters.
*
* This is mostly a mashup of various instruction types.
* Nothing follows an explicit pattern since they are mostly different.
*/
public:
// System with result
// TODO: SYSL
// System Instruction
// TODO: AT
// TODO: CFP
// TODO: CPP
void dc(ARMEmitter::DataCacheOperation DCOp, ARMEmitter::Register rt) {
constexpr uint32_t Op = 0b1101'0101'0000'1000'0111 << 12;
SystemInstruction(Op, 0, FEXCore::ToUnderlying(DCOp), rt);
}
// TODO: DVP
// TODO: IC
// TODO: TLBI
// Exception generation
void svc(uint32_t Imm) {
ExceptionGeneration(0b000, 0b000, 0b01, Imm);
}
void hvc(uint32_t Imm) {
ExceptionGeneration(0b000, 0b000, 0b10, Imm);
}
void smc(uint32_t Imm) {
ExceptionGeneration(0b000, 0b000, 0b11, Imm);
}
void brk(uint32_t Imm) {
ExceptionGeneration(0b001, 0b000, 0b00, Imm);
}
void hlt(uint32_t Imm) {
ExceptionGeneration(0b010, 0b000, 0b00, Imm);
}
void tcancel(uint32_t Imm) {
ExceptionGeneration(0b011, 0b000, 0b00, Imm);
}
void dcps1(uint32_t Imm) {
ExceptionGeneration(0b101, 0b000, 0b01, Imm);
}
void dcps2(uint32_t Imm) {
ExceptionGeneration(0b101, 0b000, 0b10, Imm);
}
void dcps3(uint32_t Imm) {
ExceptionGeneration(0b101, 0b000, 0b11, Imm);
}
// System instructions with register argument
void wfet(ARMEmitter::Register rt) {
SystemInstructionWithReg(0b0000, 0b000, rt);
}
void wfit(ARMEmitter::Register rt) {
SystemInstructionWithReg(0b0000, 0b001, rt);
}
// Hints
void nop() {
Hint(ARMEmitter::HintRegister::NOP);
}
void yield() {
Hint(ARMEmitter::HintRegister::YIELD);
}
void wfe() {
Hint(ARMEmitter::HintRegister::WFE);
}
void wfi() {
Hint(ARMEmitter::HintRegister::WFI);
}
void sev() {
Hint(ARMEmitter::HintRegister::SEV);
}
void sevl() {
Hint(ARMEmitter::HintRegister::SEVL);
}
void dgh() {
Hint(ARMEmitter::HintRegister::DGH);
}
void csdb() {
Hint(ARMEmitter::HintRegister::CSDB);
}
// Barriers
void clrex(uint32_t imm = 15) {
LOGMAN_THROW_AA_FMT(imm < 16, "Immediate out of range");
Barrier(ARMEmitter::BarrierRegister::CLREX, imm);
}
void dsb(ARMEmitter::BarrierScope Scope) {
Barrier(ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
}
void dmb(ARMEmitter::BarrierScope Scope) {
Barrier(ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
}
void isb() {
Barrier(ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(ARMEmitter::BarrierScope::SY));
}
void sb() {
Barrier(ARMEmitter::BarrierRegister::SB, 0);
}
void tcommit() {
Barrier(ARMEmitter::BarrierRegister::TCOMMIT, 0);
}
// System register move
void msr(ARMEmitter::SystemRegister reg, ARMEmitter::Register rt) {
constexpr uint32_t Op = 0b1101'0101'0001 << 20;
SystemRegisterMove(Op, rt, reg);
}
void mrs(ARMEmitter::Register rd, ARMEmitter::SystemRegister reg) {
constexpr uint32_t Op = 0b1101'0101'0011 << 20;
SystemRegisterMove(Op, rd, reg);
}
private:
// Exception Generation
void ExceptionGeneration(uint32_t opc, uint32_t op2, uint32_t LL, uint32_t Imm) {
LOGMAN_THROW_AA_FMT((Imm & 0xFFFF'0000) == 0, "Imm amount too large");
uint32_t Instr = 0b1101'0100 << 24;
Instr |= opc << 21;
Instr |= Imm << 5;
Instr |= op2 << 2;
Instr |= LL;
dc32(Instr);
}
// System instructions with register argument
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, ARMEmitter::Register rt) {
uint32_t Instr = 0b1101'0101'0000'0011'0001 << 12;
Instr |= CRm << 8;
Instr |= op2 << 5;
Instr |= Encode_rt(rt);
dc32(Instr);
}
// Hints
void Hint(ARMEmitter::HintRegister Reg) {
uint32_t Instr = 0b1101'0101'0000'0011'0010'0000'0001'1111U;
Instr |= FEXCore::ToUnderlying(Reg);
dc32(Instr);
}
// Barriers
void Barrier(ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
uint32_t Instr = 0b1101'0101'0000'0011'0011'0000'0001'1111U;
Instr |= CRm << 8;
Instr |= FEXCore::ToUnderlying(Reg);
dc32(Instr);
}
// System Instruction
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, ARMEmitter::Register rt) {
uint32_t Instr = Op;
Instr |= L << 21;
Instr |= SubOp;
Instr |= Encode_rt(rt);
dc32(Instr);
}
// System register move
void SystemRegisterMove(uint32_t Op, ARMEmitter::Register rt, ARMEmitter::SystemRegister reg) {
uint32_t Instr = Op;
Instr |= FEXCore::ToUnderlying(reg);
Instr |= Encode_rt(rt);
dc32(Instr);
}
+311
View File
@@ -0,0 +1,311 @@
// Collection of utilities from vixl.
// Following is the vixl license.
// Copyright 2015, VIXL authors
// All rights reserved.
//
// Redistribution and use in source and binary forms, with or without
// modification, are permitted provided that the following conditions are met:
//
// * Redistributions of source code must retain the above copyright notice,
// this list of conditions and the following disclaimer.
// * Redistributions in binary form must reproduce the above copyright notice,
// this list of conditions and the following disclaimer in the documentation
// and/or other materials provided with the distribution.
// * Neither the name of ARM Limited nor the names of its contributors may be
// used to endorse or promote products derived from this software without
// specific prior written permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS CONTRIBUTORS "AS IS" AND
// ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
// WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
// DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE
// FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
// DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
// SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
// CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
// OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
// Test if a given value can be encoded in the immediate field of a logical
// instruction.
// If it can be encoded, the function returns true, and values pointed to by n,
// imm_s and imm_r are updated with immediates encoded in the format required
// by the corresponding fields in the logical instruction.
// If it can not be encoded, the function returns false, and the values pointed
// to by n, imm_s and imm_r are undefined.
static bool IsImmLogical(uint64_t value,
unsigned width,
unsigned* n,
unsigned* imm_s,
unsigned* imm_r) {
constexpr auto kBRegSize = 8;
constexpr auto kHRegSize = 16;
constexpr auto kSRegSize = 32;
constexpr auto kDRegSize = 64;
constexpr auto kWRegSize = 32;
constexpr auto kXRegSize = 64;
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) ||
(width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
bool negate = false;
// Logical immediates are encoded using parameters n, imm_s and imm_r using
// the following table:
//
// N imms immr size S R
// 1 ssssss rrrrrr 64 UInt(ssssss) UInt(rrrrrr)
// 0 0sssss xrrrrr 32 UInt(sssss) UInt(rrrrr)
// 0 10ssss xxrrrr 16 UInt(ssss) UInt(rrrr)
// 0 110sss xxxrrr 8 UInt(sss) UInt(rrr)
// 0 1110ss xxxxrr 4 UInt(ss) UInt(rr)
// 0 11110s xxxxxr 2 UInt(s) UInt(r)
// (s bits must not be all set)
//
// A pattern is constructed of size bits, where the least significant S+1 bits
// are set. The pattern is rotated right by R, and repeated across a 32 or
// 64-bit value, depending on destination register width.
//
// Put another way: the basic format of a logical immediate is a single
// contiguous stretch of 1 bits, repeated across the whole word at intervals
// given by a power of 2. To identify them quickly, we first locate the
// lowest stretch of 1 bits, then the next 1 bit above that; that combination
// is different for every logical immediate, so it gives us all the
// information we need to identify the only logical immediate that our input
// could be, and then we simply check if that's the value we actually have.
//
// (The rotation parameter does give the possibility of the stretch of 1 bits
// going 'round the end' of the word. To deal with that, we observe that in
// any situation where that happens the bitwise NOT of the value is also a
// valid logical immediate. So we simply invert the input whenever its low bit
// is set, and then we know that the rotated case can't arise.)
if (value & 1) {
// If the low bit is 1, negate the value, and set a flag to remember that we
// did (so that we can adjust the return values appropriately).
negate = true;
value = ~value;
}
if (width <= kWRegSize) {
// To handle 8/16/32-bit logical immediates, the very easiest thing is to repeat
// the input value to fill a 64-bit word. The correct encoding of that as a
// logical immediate will also be the correct encoding of the value.
// Avoid making the assumption that the most-significant 56/48/32 bits are zero by
// shifting the value left and duplicating it.
for (unsigned bits = width; bits <= kWRegSize; bits *= 2) {
value <<= bits;
uint64_t mask = (UINT64_C(1) << bits) - 1;
value |= ((value >> bits) & mask);
}
}
// The basic analysis idea: imagine our input word looks like this.
//
// 0011111000111110001111100011111000111110001111100011111000111110
// c b a
// |<--d-->|
//
// We find the lowest set bit (as an actual power-of-2 value, not its index)
// and call it a. Then we add a to our original number, which wipes out the
// bottommost stretch of set bits and replaces it with a 1 carried into the
// next zero bit. Then we look for the new lowest set bit, which is in
// position b, and subtract it, so now our number is just like the original
// but with the lowest stretch of set bits completely gone. Now we find the
// lowest set bit again, which is position c in the diagram above. Then we'll
// measure the distance d between bit positions a and c (using CLZ), and that
// tells us that the only valid logical immediate that could possibly be equal
// to this number is the one in which a stretch of bits running from a to just
// below b is replicated every d bits.
uint64_t a = LowestSetBit(value);
uint64_t value_plus_a = value + a;
uint64_t b = LowestSetBit(value_plus_a);
uint64_t value_plus_a_minus_b = value_plus_a - b;
uint64_t c = LowestSetBit(value_plus_a_minus_b);
int d, clz_a, out_n;
uint64_t mask;
if (c != 0) {
// The general case, in which there is more than one stretch of set bits.
// Compute the repeat distance d, and set up a bitmask covering the basic
// unit of repetition (i.e. a word with the bottom d bits set). Also, in all
// of these cases the N bit of the output will be zero.
clz_a = CountLeadingZeros(a, kXRegSize);
int clz_c = CountLeadingZeros(c, kXRegSize);
d = clz_a - clz_c;
mask = ((UINT64_C(1) << d) - 1);
out_n = 0;
} else {
// Handle degenerate cases.
//
// If any of those 'find lowest set bit' operations didn't find a set bit at
// all, then the word will have been zero thereafter, so in particular the
// last lowest_set_bit operation will have returned zero. So we can test for
// all the special case conditions in one go by seeing if c is zero.
if (a == 0) {
// The input was zero (or all 1 bits, which will come to here too after we
// inverted it at the start of the function), for which we just return
// false.
return false;
} else {
// Otherwise, if c was zero but a was not, then there's just one stretch
// of set bits in our word, meaning that we have the trivial case of
// d == 64 and only one 'repetition'. Set up all the same variables as in
// the general case above, and set the N bit in the output.
clz_a = CountLeadingZeros(a, kXRegSize);
d = 64;
mask = ~UINT64_C(0);
out_n = 1;
}
}
// If the repeat period d is not a power of two, it can't be encoded.
if (!IsPowerOf2(d)) {
return false;
}
if (((b - a) & ~mask) != 0) {
// If the bit stretch (b - a) does not fit within the mask derived from the
// repeat period, then fail.
return false;
}
// The only possible option is b - a repeated every d bits. Now we're going to
// actually construct the valid logical immediate derived from that
// specification, and see if it equals our original input.
//
// To repeat a value every d bits, we multiply it by a number of the form
// (1 + 2^d + 2^(2d) + ...), i.e. 0x0001000100010001 or similar. These can
// be derived using a table lookup on CLZ(d).
static const uint64_t multipliers[] = {
0x0000000000000001UL,
0x0000000100000001UL,
0x0001000100010001UL,
0x0101010101010101UL,
0x1111111111111111UL,
0x5555555555555555UL,
};
uint64_t multiplier = multipliers[CountLeadingZeros(d, kXRegSize) - 57];
uint64_t candidate = (b - a) * multiplier;
if (value != candidate) {
// The candidate pattern doesn't match our input value, so fail.
return false;
}
// We have a match! This is a valid logical immediate, so now we have to
// construct the bits and pieces of the instruction encoding that generates
// it.
// Count the set bits in our basic stretch. The special case of clz(0) == -1
// makes the answer come out right for stretches that reach the very top of
// the word (e.g. numbers like 0xffffc00000000000).
int clz_b = (b == 0) ? -1 : CountLeadingZeros(b, kXRegSize);
int s = clz_a - clz_b;
// Decide how many bits to rotate right by, to put the low bit of that basic
// stretch in position a.
int r;
if (negate) {
// If we inverted the input right at the start of this function, here's
// where we compensate: the number of set bits becomes the number of clear
// bits, and the rotation count is based on position b rather than position
// a (since b is the location of the 'lowest' 1 bit after inversion).
s = d - s;
r = (clz_b + 1) & (d - 1);
} else {
r = (clz_a + 1) & (d - 1);
}
// Now we're done, except for having to encode the S output in such a way that
// it gives both the number of set bits and the length of the repeated
// segment. The s field is encoded like this:
//
// imms size S
// ssssss 64 UInt(ssssss)
// 0sssss 32 UInt(sssss)
// 10ssss 16 UInt(ssss)
// 110sss 8 UInt(sss)
// 1110ss 4 UInt(ss)
// 11110s 2 UInt(s)
//
// So we 'or' (2 * -d) with our computed s to form imms.
if ((n != NULL) || (imm_s != NULL) || (imm_r != NULL)) {
*n = out_n;
*imm_s = ((2 * -d) | (s - 1)) & 0x3f;
*imm_r = r;
}
return true;
}
private:
template <typename V>
static inline bool IsPowerOf2(V value) {
return (value != 0) && ((value & (value - 1)) == 0);
}
// Some compilers dislike negating unsigned integers,
// so we provide an equivalent.
template <typename T>
static inline T UnsignedNegate(T value) {
static_assert(std::is_unsigned<T>::value);
return ~value + 1;
}
static inline uint64_t LowestSetBit(uint64_t value) {
return value & UnsignedNegate(value);
}
template <typename V>
static inline int CountLeadingZeros(V value, int width = (sizeof(V) * 8)) {
#if COMPILER_HAS_BUILTIN_CLZ
if (width == 32) {
return (value == 0) ? 32 : __builtin_clz(static_cast<unsigned>(value));
} else if (width == 64) {
return (value == 0) ? 64 : __builtin_clzll(value);
}
#endif
return CountLeadingZerosFallBack(value, width);
}
static inline int CountLeadingZerosFallBack(uint64_t value, int width) {
LOGMAN_THROW_A_FMT(IsPowerOf2(width) && (width <= 64), "Invalid width");
if (value == 0) {
return width;
}
int count = 0;
value = value << (64 - width);
if ((value & UINT64_C(0xffffffff00000000)) == 0) {
count += 32;
value = value << 32;
}
if ((value & UINT64_C(0xffff000000000000)) == 0) {
count += 16;
value = value << 16;
}
if ((value & UINT64_C(0xff00000000000000)) == 0) {
count += 8;
value = value << 8;
}
if ((value & UINT64_C(0xf000000000000000)) == 0) {
count += 4;
value = value << 4;
}
if ((value & UINT64_C(0xc000000000000000)) == 0) {
count += 2;
value = value << 2;
}
if ((value & UINT64_C(0x8000000000000000)) == 0) {
count += 1;
}
count += (value == 0);
return count;
}
public: