Merge pull request #2722 from Sonicadvance1/rip_reconstruction

JIT: Implement support for per-instruction RIP reconstruction
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-06-16 13:02:14 -07:00
commit 9dcc1deec0
6 files changed
+115 -8

No files matched your search

+22 -4
View File
@@ -195,17 +195,35 @@ namespace FEXCore::Context {
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) {
const auto Frame = Thread->CurrentFrame;
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
const CPU::CPUBackend::JITCodeHeader *InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
if (InlineHeader) {
const CPU::CPUBackend::JITCodeTail *InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
auto InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
// Check if the host PC is currently within a code block.
// If it is then RIP can be reconstructed from the beginning of the code block.
// This is currently as close as FEX can get RIP reconstructions.
if (HostPC >= reinterpret_cast<uint64_t>(BlockBegin) &&
HostPC < reinterpret_cast<uint64_t>(BlockBegin + InlineTail->Size)) {
return InlineTail->RIP;
// Reconstruct RIP from JIT entries for this block.
uint64_t StartingHostPC = BlockBegin;
uint64_t StartingGuestRIP = InlineTail->RIP;
for (uint32_t i = 0; i < InlineTail->NumberOfRIPEntries; ++i) {
const auto &RIPEntry = RIPEntries[i];
if (HostPC >= (StartingHostPC + RIPEntry.HostPCOffset)) {
// We are beyond this entry, keep going forward.
StartingHostPC += RIPEntry.HostPCOffset;
StartingGuestRIP += RIPEntry.GuestRIPOffset;
}
else {
// Passed where the Host PC is at. Break now.
break;
}
}
return StartingGuestRIP;
}
}
@@ -811,7 +829,7 @@ namespace FEXCore::Context {
DecodedInfo = &Block.DecodedInstructions[i];
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
if (ExtendedDebugInfo) {
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
}
@@ -1102,12 +1102,33 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
CursorIncrement(sizeof(JITCodeTail));
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t *>();
auto JITRIPEntries = GetCursorAddress<JITRIPReconstructEntries*>();
CursorIncrement(sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
// Put the block's RIP entry in the tail.
// This will be used for RIP reconstruction in the future.
// TODO: This needs to be a data RIP relocation once code caching works.
// Current relocation code doesn't support this feature yet.
JITBlockTail->RIP = Entry;
{
// Store the RIP entries.
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
uintptr_t CurrentRIPOffset = 0;
uint64_t CurrentPCOffset = 0;
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
auto &RIPEntry = JITRIPEntries[i];
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
CurrentPCOffset = GuestOpcode.HostEntryOffset;
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
}
}
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
CodeData.Size = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
@@ -817,12 +817,33 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
auto JITBlockTail = getCurr<JITCodeTail*>();
setSize(getSize() + sizeof(JITCodeTail));
auto JITRIPEntriesLocation = getCurr<uint8_t *>();
auto JITRIPEntries = getCurr<JITRIPReconstructEntries*>();
setSize(getSize() + sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
// Put the block's RIP entry in the tail.
// This will be used for RIP reconstruction in the future.
// TODO: This needs to be a data RIP relocation once code caching works.
// Current relocation code doesn't support this feature yet.
JITBlockTail->RIP = Entry;
{
// Store the RIP entries.
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
uintptr_t CurrentRIPOffset = 0;
uint64_t CurrentPCOffset = 0;
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
auto &RIPEntry = JITRIPEntries[i];
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
CurrentPCOffset = GuestOpcode.HostEntryOffset;
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
}
}
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
CodeData.Size = getCurr<uint8_t*>() - CodeData.BlockBegin;
@@ -149,6 +149,31 @@ public:
return false;
}
static bool CanHaveSideEffects(FEXCore::X86Tables::X86InstInfo const* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
// If it is marked as having memory access then always say it has a side-effect.
// Not always true but better to be safe.
return true;
}
auto CanHaveSideEffects = false;
auto HasPotentialMemoryAccess = [](X86Tables::DecodedOperand const &Operand) -> bool {
if (Operand.IsNone()) {
return false;
}
// This isn't guaranteed that all of these types will access memory, but be safe.
return Operand.IsGPRDirect() || Operand.IsGPRIndirect() || Operand.IsRIPRelative() || Operand.IsSIB();
};
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Dest);
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[0]);
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[1]);
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[2]);
return CanHaveSideEffects;
}
OpDispatchBuilder(FEXCore::Context::ContextImpl *ctx);
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
@@ -146,10 +146,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
+22
View File
@@ -89,6 +89,28 @@ namespace CPU {
size_t Size;
// RIP that the block's entry comes from.
uint64_t RIP;
// Number of RIP entries for this JIT Code section.
uint32_t NumberOfRIPEntries;
// Offset after this block to the start of the RIP entries.
uint32_t OffsetToRIPEntries;
};
// Entries that live after the JITCodeTail.
// These entries correlate JIT code regions with guest RIP regions.
// Using these entries FEX is able to reconstruct the guest RIP accurately when an instruction cause a signal fault.
// Packed using 16-bit entries to ensure the size isn't too large.
// These smaller sizes means that each entry is relative to each other instead of absolute offset from the start of the JIT block.
// When reconstructing the RIP, each entry must be walked linearly and accumulated with the previous entries.
// This is a trade-off between compression inside the JIT code space and execution time when reconstruction the RIP.
// RIP reconstruction when faulting is less likely so we are requiring the accumulation.
struct JITRIPReconstructEntries {
// The Host PC offset from the previous entry.
uint16_t HostPCOffset;
// How much to offset the RIP from the previous entry.
uint16_t GuestRIPOffset;
};
/**