mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 10:00:16 +02:00
Merge pull request #3145 from Sonicadvance1/optimize_inline_calls
PassManager: Optimize out CPUID and XGetBV calls
This commit is contained in:
9 files changed
+405
-156
No files matched your search
@@ -152,7 +152,7 @@ set (SRCS
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Interface/IR/Passes/InlineCallOptimization.cpp
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
|
||||
@@ -347,7 +347,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
#endif
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
// EBX, EDX, ECX become the manufacturer id string
|
||||
@@ -366,7 +366,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// Processor Info and Features bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
|
||||
@@ -451,7 +451,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// 2: Cache and TLB information
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
// returns default values from i7 model 1Ah
|
||||
@@ -476,7 +476,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_02h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// 4: Deterministic cache parameters for each level
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
constexpr uint32_t CacheType_Data = 1;
|
||||
constexpr uint32_t CacheType_Instruction = 2;
|
||||
@@ -582,14 +582,14 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_04h(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax = (1 << 2); // Always running APIC
|
||||
Res.ecx = (0 << 3); // Intel performance energy bias preference (EPB)
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
if (Leaf == 0) {
|
||||
// Disable Enhanced REP MOVS when TSO is enabled.
|
||||
@@ -705,7 +705,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) const {
|
||||
// Leaf 0
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
@@ -759,7 +759,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
// TSC frequency = ECX * EBX / EAX
|
||||
uint32_t FrequencyHz = GetCycleCounterFrequency();
|
||||
@@ -771,7 +771,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_1Ah(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_1Ah(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
if (Hybrid) {
|
||||
uint32_t CPU = GetCPUID();
|
||||
@@ -784,7 +784,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_1Ah(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// Hypervisor CPUID information leaf
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0000h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0000h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
// Maximum supported hypervisor leafs
|
||||
// We only expose the information leaf
|
||||
@@ -806,7 +806,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0000h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// Hypervisor CPUID information leaf
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0001h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0001h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
if (Leaf == 0) {
|
||||
// EAX[3:0] Is the host architecture that FEX is running under
|
||||
@@ -825,7 +825,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0001h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// Highest extended function implemented
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax = 0x8000001F;
|
||||
|
||||
@@ -844,7 +844,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// Extended processor and feature bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) const {
|
||||
|
||||
// RDTSCP is disabled on WIN32/Wine because there is no sane way to query processor ID.
|
||||
#ifndef _WIN32
|
||||
@@ -934,33 +934,33 @@ constexpr ssize_t DESCRIBE_STR_SIZE = std::char_traits<char>::length(GIT_DESCRIB
|
||||
static_assert(DESCRIBE_STR_SIZE < 32);
|
||||
|
||||
//Processor brand string
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf) const {
|
||||
return Function_8000_0002h(Leaf, GetCPUID());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf) const {
|
||||
return Function_8000_0003h(Leaf, GetCPUID());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf) const {
|
||||
return Function_8000_0004h(Leaf, GetCPUID());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf, uint32_t CPU) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[0], std::min(ssize_t{16L}, DESCRIBE_STR_SIZE));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf, uint32_t CPU) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[16], std::max(ssize_t{0L}, DESCRIBE_STR_SIZE - 16));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf, uint32_t CPU) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
auto &Data = PerCPUData[CPU];
|
||||
memcpy(&Res, Data.ProductName, std::min(strlen(Data.ProductName), sizeof(FEXCore::CPUID::FunctionResults)));
|
||||
@@ -968,7 +968,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf, uin
|
||||
}
|
||||
|
||||
// L1 Cache and TLB identifiers
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0005h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0005h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
// L1 TLB Information for 2MB and 4MB pages
|
||||
@@ -1003,7 +1003,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0005h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// L2 Cache identifiers
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0006h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0006h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
// L2 TLB Information for 2MB and 4MB pages
|
||||
@@ -1037,7 +1037,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0006h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// Advanced power management
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0007h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0007h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax = (1 << 2); // APIC timer not affected by p-state
|
||||
Res.edx =
|
||||
@@ -1046,7 +1046,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0007h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// Virtual and physical address sizes
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax =
|
||||
(48 << 0) | // PhysAddrSize = 48-bit
|
||||
@@ -1068,7 +1068,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// TLB 1GB page identifiers
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0019h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0019h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
Res.eax =
|
||||
(0xF << 28) | // L1 DTLB associativity for 1GB pages
|
||||
@@ -1085,7 +1085,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0019h(uint32_t Leaf) {
|
||||
}
|
||||
|
||||
// Deterministic cache parameters for each level
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_001Dh(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_001Dh(uint32_t Leaf) const {
|
||||
// This is nearly a copy of CPUID function 4h
|
||||
// There are some minor changes though
|
||||
|
||||
@@ -1180,12 +1180,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_001Dh(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() {
|
||||
FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
// This just returns XCR0
|
||||
FEXCore::CPUID::XCRResults Res{
|
||||
.eax = static_cast<uint32_t>(XCR0),
|
||||
|
||||
@@ -34,7 +34,7 @@ public:
|
||||
|
||||
void Init(FEXCore::Context::ContextImpl *ctx);
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) const {
|
||||
if (Function < Primary.size()) {
|
||||
const auto Handler = Primary[Function];
|
||||
return (this->*Handler)(Leaf);
|
||||
@@ -55,7 +55,7 @@ public:
|
||||
return Function_Reserved(Leaf);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) const {
|
||||
if (Function == 0x8000'0002U)
|
||||
return Function_8000_0002h(Leaf, CPU % PerCPUData.size());
|
||||
else if (Function == 0x8000'0003U)
|
||||
@@ -64,7 +64,7 @@ public:
|
||||
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) {
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) const {
|
||||
if (Function >= 1) {
|
||||
// XCR function 1 is not yet supported.
|
||||
return {};
|
||||
@@ -73,6 +73,43 @@ public:
|
||||
return XCRFunction_0h();
|
||||
}
|
||||
|
||||
bool DoesXCRFunctionReportConstantData(uint32_t Function) const {
|
||||
// Every function currently returns constant data.
|
||||
return true;
|
||||
}
|
||||
|
||||
enum class SupportsConstant {
|
||||
CONSTANT,
|
||||
NONCONSTANT,
|
||||
};
|
||||
enum class NeedsLeafConstant {
|
||||
NEEDSLEAFCONSTANT,
|
||||
NOLEAFCONSTANT,
|
||||
};
|
||||
struct FunctionConstant {
|
||||
SupportsConstant SupportsConstantFunction;
|
||||
NeedsLeafConstant NeedsLeaf;
|
||||
};
|
||||
|
||||
static constexpr FunctionConstant DoesFunctionReportConstantData(uint32_t Function) {
|
||||
if (Function < Primary.size()) {
|
||||
return Primary_Constant[Function];
|
||||
}
|
||||
|
||||
constexpr uint32_t HypervisorBase = 0x4000'0000;
|
||||
if (Function >= HypervisorBase && Function < (HypervisorBase + Hypervisor.size())) {
|
||||
return Hypervisor_Constant[Function - HypervisorBase];
|
||||
}
|
||||
|
||||
constexpr uint32_t ExtendedBase = 0x8000'0000;
|
||||
if (Function >= ExtendedBase && Function < (ExtendedBase + Extended.size())) {
|
||||
return Extended_Constant[Function - ExtendedBase];
|
||||
}
|
||||
|
||||
// Anything unsupported is known constant return of reserved data.
|
||||
return {SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT};
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
bool Hybrid{};
|
||||
@@ -108,7 +145,7 @@ private:
|
||||
return (XCR0 & XCR0_AVX) ? 1 : 0;
|
||||
}
|
||||
|
||||
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
|
||||
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf) const;
|
||||
|
||||
struct CPUData {
|
||||
const char *ProductName{};
|
||||
@@ -120,39 +157,42 @@ private:
|
||||
fextl::vector<CPUData> PerCPUData{};
|
||||
|
||||
// Functions
|
||||
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_01h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_02h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_04h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_06h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_07h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_01h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_02h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_04h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_06h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_07h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_4000_0001h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf) const;
|
||||
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf, uint32_t CPU);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf, uint32_t CPU);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf, uint32_t CPU);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf, uint32_t CPU) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf, uint32_t CPU) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf, uint32_t CPU) const;
|
||||
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0005h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0008h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0005h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0008h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf) const;
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf) const;
|
||||
|
||||
FEXCore::CPUID::XCRResults XCRFunction_0h();
|
||||
FEXCore::CPUID::XCRResults XCRFunction_0h() const;
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
static constexpr std::array<FunctionHandler, 27> Primary = {
|
||||
static constexpr size_t PRIMARY_FUNCTION_COUNT = 27;
|
||||
static constexpr size_t HYPERVISOR_FUNCTION_COUNT = 2;
|
||||
static constexpr size_t EXTENDED_FUNCTION_COUNT = 32;
|
||||
static constexpr std::array<FunctionHandler, PRIMARY_FUNCTION_COUNT> Primary = {
|
||||
// 0: Highest function parameter and ID
|
||||
&CPUIDEmu::Function_0h,
|
||||
// 1: Processor info
|
||||
@@ -222,14 +262,94 @@ private:
|
||||
#endif
|
||||
};
|
||||
|
||||
static constexpr std::array<FunctionHandler, 2> Hypervisor = {
|
||||
static constexpr std::array<FunctionConstant, PRIMARY_FUNCTION_COUNT> Primary_Constant = {{
|
||||
// 0: Highest function parameter and ID
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 1: Processor info
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 2: Cache and TLB info
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 3: Serial Number(previously), now reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#ifndef CPUID_AMD
|
||||
// 4: Deterministic cache parameters for each level
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
#else
|
||||
// 4: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#endif
|
||||
// 5: Monitor/mwait
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 6: Thermal and power management
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 7: Extended feature flags
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
// 0x08: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 9: Direct Cache Access information
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x0A: Architectural performance monitoring
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x0B: Extended topology enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x0C: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x0D: Processor extended state enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
// 0x0E: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x0F: Intel RDT monitoring
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x10: Intel RDT allocation enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x12: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x12: Intel SGX capability enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x13: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x14: Intel Processor trace
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#ifndef CPUID_AMD
|
||||
// 0x15: Timestamp counter information
|
||||
// Doesn't exist on AMD hardware
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#else
|
||||
// 0x15: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#endif
|
||||
// 0x16: Processor frequency information
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x17: SoC vendor attribute enumeration
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x18: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x19: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#ifndef CPUID_AMD
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#else
|
||||
// 0x1A: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#endif
|
||||
}};
|
||||
|
||||
static constexpr std::array<FunctionHandler, HYPERVISOR_FUNCTION_COUNT> Hypervisor = {
|
||||
// Hypervisor CPUID information leaf
|
||||
&CPUIDEmu::Function_4000_0000h,
|
||||
// FEX-Emu specific leaf
|
||||
&CPUIDEmu::Function_4000_0001h,
|
||||
};
|
||||
|
||||
static constexpr std::array<FunctionHandler, 32> Extended = {
|
||||
static constexpr std::array<FunctionConstant, HYPERVISOR_FUNCTION_COUNT> Hypervisor_Constant = {{
|
||||
// Hypervisor CPUID information leaf
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// FEX-Emu specific leaf
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
}};
|
||||
|
||||
static constexpr std::array<FunctionHandler, EXTENDED_FUNCTION_COUNT> Extended = {
|
||||
// Largest extended function number
|
||||
&CPUIDEmu::Function_8000_0000h,
|
||||
// Processor vendor
|
||||
@@ -303,5 +423,82 @@ private:
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
&CPUIDEmu::Function_Reserved,
|
||||
};
|
||||
|
||||
static constexpr std::array<FunctionConstant, EXTENDED_FUNCTION_COUNT> Extended_Constant = {{
|
||||
// Largest extended function number
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// Processor vendor
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// Processor brand string
|
||||
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// Processor brand string continued
|
||||
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// Processor brand string continued
|
||||
{ SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#ifdef CPUID_AMD
|
||||
// 0x8000'0005: L1 Cache and TLB identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#else
|
||||
// 0x8000'0005: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#endif
|
||||
// 0x8000'0006: L2 Cache identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0007: Advanced power management information
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0008: Virtual and physical address sizes
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0009: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'000A: SVM Revision
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'000B: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'000C: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'000D: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'000E: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'000F: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0010: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0011: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0012: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0013: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0014: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0015: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0016: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0017: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0018: Reserved?
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'0019: TLB 1GB page identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'001A: Performance optimization identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'001B: Instruction based sampling identifiers
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'001C: Lightweight profiling capabilities
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#ifdef CPUID_AMD
|
||||
// 0x8000'001D: Cache properties
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NEEDSLEAFCONSTANT },
|
||||
#else
|
||||
// 0x8000'001D: Reserved
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
#endif
|
||||
// 0x8000'001E: Extended APIC ID
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
{ SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT },
|
||||
}};
|
||||
};
|
||||
}
|
||||
@@ -1758,7 +1758,9 @@ void OpDispatchBuilder::MOVOffsetOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CPUIDOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
const auto GPRSize = CTX->GetGPRSize();
|
||||
|
||||
OrderedNode *Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], GPRSize, Op->Flags, -1);
|
||||
OrderedNode *Leaf = LoadGPRRegister(X86State::REG_RCX);
|
||||
|
||||
auto Res = _CPUID(Src, Leaf);
|
||||
|
||||
@@ -84,7 +84,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool Inli
|
||||
|
||||
////// InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
|
||||
InsertPass(CreateSyscallOptimization());
|
||||
InsertPass(CreateInlineCallOptimization(&ctx->CPUID));
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
}
|
||||
|
||||
|
||||
@@ -45,7 +45,7 @@ protected:
|
||||
};
|
||||
|
||||
class PassManager final {
|
||||
friend class SyscallOptimization;
|
||||
friend class InlineCallOptimization;
|
||||
public:
|
||||
void AddDefaultPasses(FEXCore::Context::ContextImpl *ctx, bool InlineConstants, bool StaticRegisterAllocation);
|
||||
void AddDefaultValidationPasses();
|
||||
|
||||
@@ -3,6 +3,10 @@
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
namespace FEXCore {
|
||||
class CPUIDEmu;
|
||||
}
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
class IntrusivePooledAllocator;
|
||||
}
|
||||
@@ -14,7 +18,7 @@ class RegisterAllocationData;
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants, bool SupportsTSOImm9);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateSyscallOptimization();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateInlineCallOptimization(const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination(bool SupportsAVX);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination();
|
||||
|
||||
@@ -0,0 +1,136 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Removes unused arguments if known syscall number
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class InlineCallOptimization final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
InlineCallOptimization(const FEXCore::CPUIDEmu* CPUID)
|
||||
: CPUID {CPUID} {}
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
private:
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
};
|
||||
|
||||
bool InlineCallOptimization::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::SyscallOpt");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
|
||||
if (IROp->Op == FEXCore::IR::OP_SYSCALL) {
|
||||
auto Op = IROp->CW<IR::IROp_Syscall>();
|
||||
|
||||
// Is the first argument a constant?
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->SyscallID, &Constant)) {
|
||||
auto SyscallDef = Manager->SyscallHandler->GetSyscallABI(Constant);
|
||||
auto SyscallFlags = Manager->SyscallHandler->GetSyscallFlags(Constant);
|
||||
|
||||
// Update the syscall flags
|
||||
Op->Flags = SyscallFlags;
|
||||
|
||||
// XXX: Once we have the ability to do real function calls then we can call directly in to the syscall handler
|
||||
if (SyscallDef.NumArgs < FEXCore::HLE::SyscallArguments::MAX_ARGS) {
|
||||
// If the number of args are less than what the IR op supports then we can remove arg usage
|
||||
// We need +1 since we are still passing in syscall number here
|
||||
for (uint8_t Arg = (SyscallDef.NumArgs + 1); Arg < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++Arg) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Arg, IREmit->Invalid());
|
||||
}
|
||||
#ifdef _M_ARM_64
|
||||
// Replace syscall with inline passthrough syscall if we can
|
||||
if (SyscallDef.HostSyscallNumber != -1) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// Skip Args[0] since that is the syscallid
|
||||
auto InlineSyscall = IREmit->_InlineSyscall(
|
||||
CurrentIR.GetNode(IROp->Args[1]),
|
||||
CurrentIR.GetNode(IROp->Args[2]),
|
||||
CurrentIR.GetNode(IROp->Args[3]),
|
||||
CurrentIR.GetNode(IROp->Args[4]),
|
||||
CurrentIR.GetNode(IROp->Args[5]),
|
||||
CurrentIR.GetNode(IROp->Args[6]),
|
||||
SyscallDef.HostSyscallNumber,
|
||||
Op->Flags);
|
||||
|
||||
// Replace all syscall uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, InlineSyscall);
|
||||
|
||||
// We must remove here since DCE can't remove a IROp with sideeffects
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
else if (IROp->Op == FEXCore::IR::OP_CPUID) {
|
||||
auto Op = IROp->CW<IR::IROp_CPUID>();
|
||||
|
||||
uint64_t ConstantFunction{}, ConstantLeaf{};
|
||||
bool IsConstantFunction = IREmit->IsValueConstant(Op->Function, &ConstantFunction);
|
||||
bool IsConstantLeaf = IREmit->IsValueConstant(Op->Leaf, &ConstantLeaf);
|
||||
// If the CPUID function is constant then we can try and optimize.
|
||||
if (IsConstantFunction) { // && ConstantFunction != 1) {
|
||||
// Check if it supports constant data reporting for this function.
|
||||
const auto SupportsConstant = CPUID->DoesFunctionReportConstantData(ConstantFunction);
|
||||
if (SupportsConstant.SupportsConstantFunction == CPUIDEmu::SupportsConstant::CONSTANT) {
|
||||
// If the CPUID needs a constant leaf to be optimized then this can't work if we didn't const-prop the leaf register.
|
||||
if (!(SupportsConstant.NeedsLeaf == CPUIDEmu::NeedsLeafConstant::NEEDSLEAFCONSTANT && !IsConstantLeaf)) {
|
||||
// Calculate the constant data and replace all uses.
|
||||
// DCE will remove the CPUID IR operation.
|
||||
const auto ConstantCPUIDResult = CPUID->RunFunction(ConstantFunction, ConstantLeaf);
|
||||
uint64_t ResultsLower = (static_cast<uint64_t>(ConstantCPUIDResult.ebx) << 32) | ConstantCPUIDResult.eax;
|
||||
uint64_t ResultsUpper = (static_cast<uint64_t>(ConstantCPUIDResult.edx) << 32) | ConstantCPUIDResult.ecx;
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto ElementPair = IREmit->_CreateElementPair(IR::OpSize::i128Bit, IREmit->_Constant(ResultsLower), IREmit->_Constant(ResultsUpper));
|
||||
// Replace all CPUID uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ElementPair);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
else if (IROp->Op == FEXCore::IR::OP_XGETBV) {
|
||||
auto Op = IROp->CW<IR::IROp_XGetBV>();
|
||||
|
||||
uint64_t ConstantFunction{};
|
||||
if (IREmit->IsValueConstant(Op->Function, &ConstantFunction) &&
|
||||
CPUID->DoesXCRFunctionReportConstantData(ConstantFunction)) {
|
||||
const auto ConstantXCRResult = CPUID->RunXCRFunction(ConstantFunction);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto ElementPair = IREmit->_CreateElementPair(IR::OpSize::i64Bit, IREmit->_Constant(ConstantXCRResult.eax), IREmit->_Constant(ConstantXCRResult.edx));
|
||||
// Replace all xgetbv uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ElementPair);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return Changed;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateInlineCallOptimization(const FEXCore::CPUIDEmu* CPUID) {
|
||||
return fextl::make_unique<InlineCallOptimization>(CPUID);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,90 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Removes unused arguments if known syscall number
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class SyscallOptimization final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool Run(IREmitter *IREmit) override;
|
||||
};
|
||||
|
||||
bool SyscallOptimization::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::SyscallOpt");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
|
||||
if (IROp->Op == FEXCore::IR::OP_SYSCALL) {
|
||||
auto Op = IROp->CW<IR::IROp_Syscall>();
|
||||
|
||||
// Is the first argument a constant?
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->SyscallID, &Constant)) {
|
||||
auto SyscallDef = Manager->SyscallHandler->GetSyscallABI(Constant);
|
||||
auto SyscallFlags = Manager->SyscallHandler->GetSyscallFlags(Constant);
|
||||
|
||||
// Update the syscall flags
|
||||
Op->Flags = SyscallFlags;
|
||||
|
||||
// XXX: Once we have the ability to do real function calls then we can call directly in to the syscall handler
|
||||
if (SyscallDef.NumArgs < FEXCore::HLE::SyscallArguments::MAX_ARGS) {
|
||||
// If the number of args are less than what the IR op supports then we can remove arg usage
|
||||
// We need +1 since we are still passing in syscall number here
|
||||
for (uint8_t Arg = (SyscallDef.NumArgs + 1); Arg < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++Arg) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Arg, IREmit->Invalid());
|
||||
}
|
||||
#ifdef _M_ARM_64
|
||||
// Replace syscall with inline passthrough syscall if we can
|
||||
if (SyscallDef.HostSyscallNumber != -1) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// Skip Args[0] since that is the syscallid
|
||||
auto InlineSyscall = IREmit->_InlineSyscall(
|
||||
CurrentIR.GetNode(IROp->Args[1]),
|
||||
CurrentIR.GetNode(IROp->Args[2]),
|
||||
CurrentIR.GetNode(IROp->Args[3]),
|
||||
CurrentIR.GetNode(IROp->Args[4]),
|
||||
CurrentIR.GetNode(IROp->Args[5]),
|
||||
CurrentIR.GetNode(IROp->Args[6]),
|
||||
SyscallDef.HostSyscallNumber,
|
||||
Op->Flags);
|
||||
|
||||
// Replace all syscall uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, InlineSyscall);
|
||||
|
||||
// We must remove here since DCE can't remove a IROp with sideeffects
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateSyscallOptimization() {
|
||||
return fextl::make_unique<SyscallOptimization>();
|
||||
}
|
||||
|
||||
}
|
||||
Reference in new issue
Block a user