Adds an inblock time sampling routine to the x86-64 JIT

Useful for a quick overview of the time spent in a function. It isn't
amazing for profiling since it effects the time in the function by
itself.
Only useful as a reference.
This commit is contained in:
Ryan Houdek authored and Stefanos Kornilios Mitsis Poiitidis committed 2020-03-06 07:55:35 +02:00
1 parent 0a959475c5
commit 6890ae4e20
6 files changed
+139

No files matched your search

+1
View File
@@ -51,6 +51,7 @@ set (SRCS
Interface/Config/Config.cpp
Interface/Context/Context.cpp
Interface/Core/BlockCache.cpp
Interface/Core/BlockSamplingData.cpp
Interface/Core/Core.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
+6
View File
@@ -15,6 +15,8 @@
namespace FEXCore {
class SyscallHandler;
class BlockSamplingData;
namespace CPU {
class JITCore;
}
@@ -65,6 +67,10 @@ namespace FEXCore::Context {
CustomCPUFactoryType CustomCPUFactory;
CustomCPUFactoryType FallbackCPUFactory;
#ifdef BLOCKSTATS
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
#endif
Context();
~Context();
@@ -0,0 +1,51 @@
#include "Interface/Core/BlockSamplingData.h"
#include "LogManager.h"
#include <cstring>
#include <fstream>
namespace FEXCore {
void BlockSamplingData::DumpBlockData() {
std::fstream Output;
Output.open("output.csv", std::fstream::out | std::fstream::binary);
if (!Output.is_open())
return;
Output << "Entry, Min, Max, Total, Calls, Average" << std::endl;
for (auto it : SamplingMap) {
if (!it.second->TotalCalls)
continue;
Output << "0x" << std::hex << it.first
<< ", " << std::dec << it.second->Min
<< ", " << std::dec << it.second->Max
<< ", " << std::dec << it.second->TotalTime
<< ", " << std::dec << it.second->TotalCalls
<< ", " << std::dec << ((double)it.second->TotalTime / (double)it.second->TotalCalls)
<< std::endl;
}
Output.close();
LogMan::Msg::D("Dumped %d blocks of sampling data", SamplingMap.size());
}
BlockSamplingData::BlockData *BlockSamplingData::GetBlockData(uint64_t RIP) {
auto it = SamplingMap.find(RIP);
if (it != SamplingMap.end()) {
return it->second;
}
BlockData *NewData = new BlockData{};
memset(NewData, 0, sizeof(BlockData));
NewData->Min = ~0ULL;
SamplingMap[RIP] = NewData;
return NewData;
}
BlockSamplingData::~BlockSamplingData() {
DumpBlockData();
for (auto it : SamplingMap) {
delete it.second;
}
SamplingMap.clear();
}
}
+22
View File
@@ -0,0 +1,22 @@
#pragma once
#include <unordered_map>
namespace FEXCore {
class BlockSamplingData {
public:
struct BlockData {
uint64_t Start, End;
uint64_t Min, Max;
uint64_t TotalTime;
uint64_t TotalCalls;
};
BlockData *GetBlockData(uint64_t RIP);
~BlockSamplingData();
void DumpBlockData();
private:
std::unordered_map<uint64_t, BlockData*> SamplingMap;
};
}
+4
View File
@@ -3,6 +3,7 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/BlockCache.h"
#include "Interface/Core/BlockSamplingData.h"
#include "Interface/Core/Core.h"
#include "Interface/Core/DebugData.h"
#include "Interface/Core/OpcodeDispatcher.h"
@@ -102,6 +103,9 @@ namespace FEXCore::Context {
FallbackCPUFactory = FEXCore::Core::DefaultFallbackCore::CPUCreationFactory;
PassManager.AddDefaultPasses();
PassManager.AddDefaultValidationPasses();
#ifdef BLOCKSTATS
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
#endif
}
bool Context::GetFilenameHash(std::string const &Filename, std::string &Hash) {
+55
View File
@@ -1,4 +1,5 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/BlockSamplingData.h"
#include "Interface/Core/InternalThreadState.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
@@ -99,6 +100,10 @@ private:
using CustomDispatch = void(*)(FEXCore::Core::InternalThreadState *Thread);
CustomDispatch DispatchPtr{};
IR::RegisterAllocationPass *RAPass;
#ifdef BLOCKSTATS
bool GetSamplingData {true};
#endif
};
JITCore::JITCore(FEXCore::Context::Context *ctx)
@@ -226,6 +231,49 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const
auto HeaderOp = HeaderNode->Op(DataBegin)->CW<FEXCore::IR::IROp_IRHeader>();
LogMan::Throw::A(HeaderOp->Header.Op == IR::OP_IRHEADER, "First op wasn't IRHeader");
#ifdef BLOCKSTATS
BlockSamplingData::BlockData *SamplingData = CTX->BlockData->GetBlockData(HeaderOp->Entry);
if (GetSamplingData) {
mov(rcx, reinterpret_cast<uintptr_t>(SamplingData));
rdtsc();
shl(rdx, 32);
or(rax, rdx);
mov(qword [rcx + offsetof(BlockSamplingData::BlockData, Start)], rax);
}
auto ExitBlock = [&]() {
if (GetSamplingData) {
mov(rcx, reinterpret_cast<uintptr_t>(SamplingData));
// Get time
rdtsc();
shl(rdx, 32);
or(rax, rdx);
// Calculate time spent in block
mov(rdx, qword [rcx + offsetof(BlockSamplingData::BlockData, Start)]);
sub(rax, rdx);
// Add time to total time
add(qword [rcx + offsetof(BlockSamplingData::BlockData, TotalTime)], rax);
// Increment call count
inc(qword [rcx + offsetof(BlockSamplingData::BlockData, TotalCalls)]);
// Calculate min
mov(rdx, qword [rcx + offsetof(BlockSamplingData::BlockData, Min)]);
cmp(rdx, rax);
cmova(rdx, rax);
mov(qword [rcx + offsetof(BlockSamplingData::BlockData, Min)], rdx);
// Calculate max
mov(rdx, qword [rcx + offsetof(BlockSamplingData::BlockData, Max)]);
cmp(rdx, rax);
cmovb(rdx, rax);
mov(qword [rcx + offsetof(BlockSamplingData::BlockData, Max)], rdx);
}
};
#endif
IR::OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin);
while (1) {
using namespace FEXCore::IR;
@@ -317,6 +365,9 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const
pop(rbp);
pop(rbx);
#ifdef BLOCKSTATS
ExitBlock();
#endif
ret();
break;
}
@@ -342,6 +393,10 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView<true> const
pop(r12);
pop(rbp);
pop(rbx);
#ifdef BLOCKSTATS
ExitBlock();
#endif
ret();
break;
default: LogMan::Msg::A("Unknown Break reason: %d", Op->Reason);