Merge pull request #5927 from Plagman/plagman/mono_more_caching

DiskCache: also patch branches for more anon hit rate
This commit is contained in:
Ryan Houdek authored and GitHub committed 2026-09-07 14:09:35 -07:00
commit 023d0cdaa6
115 files changed
+261 -122

No files matched your search

@@ -544,6 +544,27 @@ static inline void ApplyPatchableDataRelocation(uint64_t SiteAddress, uint8_t Va
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Value, CPU::Arm64Emitter::PadType::DOPAD);
}
static inline int64_t ReadLiveGuestDisplacement(uint64_t SiteAddress, uint8_t ValueSize) {
uint64_t Raw = 0;
memcpy(&Raw, reinterpret_cast<const void*>(SiteAddress), ValueSize);
// manual sign-extension from guest live bytes
// 1/2 sizes not permitted in DetectDataMasks currently
if (ValueSize == 4) {
return (int32_t)Raw;
} else {
return (int64_t)Raw;
}
}
static inline void ApplyPatchableRIPLiteralRelocation(uint64_t SiteAddress, uint8_t ValueSize, CPU::Arm64Emitter& Emitter) {
Emitter.dc64(SiteAddress + ValueSize + ReadLiveGuestDisplacement(SiteAddress, ValueSize));
}
static inline void ApplyPatchableRIPMoveRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
const uint64_t Target = SiteAddress + ValueSize + ReadLiveGuestDisplacement(SiteAddress, ValueSize);
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Target, CPU::Arm64Emitter::PadType::DOPAD);
}
bool CodeCache::ApplyPackedCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
std::span<const DiskCache::BlobSmallRelocation> SmallRelocs,
std::span<const DiskCache::BlobThunkRelocation> ThunkRelocs) {
@@ -569,6 +590,15 @@ bool CodeCache::ApplyPackedCodeRelocations(uint64_t GuestEntry, std::span<std::b
Reloc.PatchableData.RegisterIndex, Emitter);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
ApplyPatchableRIPLiteralRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize, Emitter);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE: {
ApplyPatchableRIPMoveRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
Reloc.PatchableData.RegisterIndex, Emitter);
break;
}
default: ERROR_AND_DIE_FMT("Unknown packed relocation type {}", ToUnderlying((CPU::RelocationTypes)Reloc.Type));
}
}
+4 -1
View File
@@ -885,7 +885,10 @@ namespace DiskCache {
SmallRelocs[SmallIdx++] = SmallReloc;
break;
}
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE: {
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE:
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE:
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
// same data for all, relative vs. not and register vs. literal will depend on type on apply
BlobSmallRelocation SmallReloc = {};
SmallReloc.Offset = Reloc.Header.Offset;
SmallReloc.Type = uint8_t(Reloc.Header.Type);
+38 -1
View File
@@ -1401,6 +1401,7 @@ void Decoder::DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block) {
}
FEXCore::X86Tables::DecodedOperand* LiteralToPatch = nullptr;
DataMaskType Type;
// mov reg,imm
if (DecodeInst->OP >= 0xB8 && DecodeInst->OP <= 0xBF) {
@@ -1416,12 +1417,20 @@ void Decoder::DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block) {
// if (LiteralToPatch && Value < 0x1000000ULL) {
// LiteralToPatch = nullptr;
// }
Type = DataMaskType::MOV;
}
// jmp/call branches that use a literal rip-relative offset
// some of those may be inlined by multiblock and will be cleaned up at decode end
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_SETS_RIP && DecodeInst->Src[0].IsLiteral()) {
LiteralToPatch = &DecodeInst->Src[0];
Type = DataMaskType::BRANCH;
}
// todo add a bunch more
if (LiteralToPatch) {
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, LastFieldReadSize});
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, Type, LastFieldReadSize});
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
@@ -1429,6 +1438,29 @@ void Decoder::DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block) {
}
}
void Decoder::PruneInlinedBranchDataMasks() {
for (auto& Block : BlockInfo.Blocks) {
if (!Block.DataMasks.size()) {
continue;
}
const auto& LastInst = Block.DecodedInstructions[Block.NumInstructions - 1];
const auto& LastMask = Block.DataMasks.back();
if (LastMask.Type != DataMaskType::BRANCH) {
continue;
}
const uint64_t NextInst = LastInst.PC + LastInst.InstSize;
if (LastMask.FieldAddress < LastInst.PC || LastMask.FieldAddress + LastMask.ValueSize > NextInst) {
continue;
}
if (std::ranges::binary_search(BlockInfo.Blocks, NextInst + LastInst.Src[0].Data.LiteralPatchable.Value, std::less {}, &DecodedBlocks::Entry)) {
Block.DataMasks.pop_back();
}
}
}
void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
// counter-intuitively, the masks are also needed for lookup on anon prefix decodes, not just stores
bool WantsDataMasks = CTX->DiskCache.IsReadingDiskCache() || CTX->DiskCache.IsWritingDiskCache();
@@ -1640,6 +1672,11 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
for (auto& Block : BlockInfo.Blocks) {
Block.IsEntryPoint = BlockInfo.EntryPoints.contains(Block.Entry);
}
// now that multiblock has settled down, remove any branch masks we put down that didn't end the block
if (WantsDataMasks) {
PruneInlinedBranchDataMasks();
}
}
void Decoder::SetupDecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t PC, uint64_t MaxInst) {
+4
View File
@@ -32,8 +32,11 @@ public:
UNIMPLEMENTED_INST,
};
enum class DataMaskType : uint8_t { MOV, BRANCH };
struct DataMask final {
uint64_t FieldAddress;
DataMaskType Type;
uint8_t ValueSize;
};
@@ -110,6 +113,7 @@ private:
void AddBranchTarget(uint64_t Target);
void DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block);
void PruneInlinedBranchDataMasks();
bool CheckRangeExecutable(uint64_t Address, uint64_t Size);
@@ -58,7 +58,8 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit) {
switch (Lit.MoveABI.Header.Type) {
case RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL:
case RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
case RelocationTypes::RELOC_GUEST_RIP_LITERAL:
case RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
Lit.MoveABI.Header.Offset = GetCursorOffset();
break;
}
@@ -102,6 +103,24 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
Relocations.emplace_back(MoveABI);
}
auto Arm64JITCore::InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize) -> NamedSymbolLiteralPair {
return {
.Lit = GuestRIP,
.MoveABI =
{
.GuestPatchableData = {.Header =
{
.Offset = 0, // Set by PlaceNamedSymbolLiteral
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL,
},
.RegisterIndex = 0, // unused
.ValueSize = ValueSize,
// NOTE: Cache serialization will subtract the unit entry address later
.SiteAddress = SiteAddress},
},
};
}
void Arm64JITCore::InsertGuestPatchableDataMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize) {
Relocation MoveABI = Relocation::Default();
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE};
@@ -114,6 +133,18 @@ void Arm64JITCore::InsertGuestPatchableDataMove(ARMEmitter::Register Reg, uint64
Relocations.emplace_back(MoveABI);
}
void Arm64JITCore::InsertGuestPatchableRIPMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize) {
Relocation MoveABI = Relocation::Default();
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE};
MoveABI.GuestPatchableData.RegisterIndex = Reg.Idx();
MoveABI.GuestPatchableData.ValueSize = ValueSize;
MoveABI.GuestPatchableData.SiteAddress = SiteAddress;
// this might get patched on disk cache load
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Value, FEXCore::CPU::Arm64Emitter::PadType::DOPAD);
Relocations.emplace_back(MoveABI);
}
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations(uint64_t GuestBaseAddress) {
// Rebase relocations to library base address
for (auto& Relocation : Relocations) {
@@ -83,7 +83,11 @@ DEF_OP(ExitFunction) {
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
if (Op->PatchSiteAddress) {
InsertGuestPatchableRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP, Op->PatchSiteAddress, Op->PatchSiteSize);
} else {
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
}
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.ExitFunctionEC));
br(TMP2);
} else {
@@ -173,6 +177,7 @@ DEF_OP(ExitFunction) {
ARMEmitter::ForwardLabel TFUnset;
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
// todo do we need to account for cache patching here?
InsertGuestRIPMove(TMP1, NewRIP);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
@@ -180,7 +185,7 @@ DEF_OP(ExitFunction) {
(void)Bind(&TFUnset);
}
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call, Op->PatchSiteAddress, Op->PatchSiteSize);
(void)Bind(&l_CallReturn);
#ifdef ARCHITECTURE_arm64ec
}
+9 -3
View File
@@ -1019,9 +1019,15 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// This is a ExitFunctionLinkData struct
BindOrRestart(&l_ExitLink);
dc64(0); // HostCode
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
dc64(0); // HostCode
if (PendingJumpThunk.PatchSiteAddress) {
// GuestRIP with an extra step
PlaceNamedSymbolLiteral(
InsertGuestPatchableRIPLiteral(PendingJumpThunk.GuestRIP, PendingJumpThunk.PatchSiteAddress, PendingJumpThunk.PatchSiteSize));
} else {
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
}
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
}
BindOrRestart(&l_ExitLink);
+10 -2
View File
@@ -105,6 +105,8 @@ private:
uint64_t CallerAddress;
uint64_t GuestRIP;
ARMEmitter::ForwardLabel Label;
uint64_t PatchSiteAddress = 0;
uint8_t PatchSiteSize = 0;
};
fextl::vector<PendingJumpThunk> PendingJumpThunks;
@@ -345,8 +347,8 @@ private:
uint32_t End;
};
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
void EmitLinkedBranch(uint64_t GuestRIP, bool Call, uint64_t PatchSiteAddress = 0, uint8_t PatchSiteSize = 0) {
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}, PatchSiteAddress, PatchSiteSize});
auto& Thunk = PendingJumpThunks.back();
BindOrRestart(&Thunk.Label);
if (Call) {
@@ -563,6 +565,7 @@ private:
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
void InsertGuestPatchableDataMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize);
void InsertGuestPatchableRIPMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize);
/**
* @brief Inserts a named symbol as a literal in memory
@@ -583,6 +586,11 @@ private:
*/
NamedSymbolLiteralPair InsertGuestRIPLiteral(uint64_t GuestRIP);
/**
* @brief Like InsertGuestRIPLiteral, but with patch information to recompute value from live guest bytes at cache load time
*/
NamedSymbolLiteralPair InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t SiteAddress, uint8_t ValueSize);
/**
* @brief Place the named symbol literal relocation in memory
*
@@ -29,6 +29,14 @@ enum class RelocationTypes : uint32_t {
// The frontend flagged those regions as patchable by the disk cache
// Aligned to struct RelocGuestPatchableData
RELOC_GUEST_PATCHABLE_DATA_MOVE,
// Same as GuestRipLiteral but patchable
// Aligned to struct RelocGuestPatchableData
RELOC_GUEST_PATCHABLE_RIP_LITERAL,
// Like PATCHABLE_RIP_LITERAL but puts it in a register
// Aligned to struct RelocGuestPatchableData
RELOC_GUEST_PATCHABLE_RIP_MOVE,
};
struct FEX_PACKED RelocationHeader final {
@@ -202,9 +202,10 @@ public:
FlushRegisterCache();
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, InvalidNode, InvalidNode);
}
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock,
uint64_t PatchSiteAddress = 0, uint64_t PatchSiteSize = 0) {
FlushRegisterCache();
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, CallReturnAddress, CallReturnBlock);
return _ExitFunction(GetOpSize(NewRIP), NewRIP, Hint, CallReturnAddress, CallReturnBlock, PatchSiteAddress, PatchSiteSize);
}
IRPair<IROp_Break> Break(BreakDefinition Reason) {
FlushRegisterCache();
@@ -1510,12 +1511,18 @@ private:
return _GetRelocatedPC(Op, Offset, false);
}
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */));
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
uint64_t PatchOffset = 0;
uint64_t PatchSize = 0;
if (Op->Src[0].IsLiteralPatchable() && Offset && Offset == (int64_t)Op->Src[0].Literal()) {
PatchOffset = Op->PC + Op->Src[0].Data.LiteralPatchable.FieldOffset;
PatchSize = Op->Src[0].Data.LiteralPatchable.Width;
}
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock, PatchOffset, PatchSize);
}
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock);
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
ExitRelocatedPC(Op, Offset, BranchHint::None, InvalidNode, InvalidNode);
}
[[nodiscard]]
@@ -171,7 +171,7 @@ struct DecodedOperand {
}
uint64_t Literal() const {
LOGMAN_THROW_A_FMT(IsLiteral(), "Precondition: must be a literal");
LOGMAN_THROW_A_FMT(IsLiteral() || IsLiteralPatchable(), "Precondition: must be a literal");
return Data.Literal.Value;
}
+2 -2
View File
@@ -312,8 +312,8 @@
"HasSideEffects": true,
"RAOverride": "2"
},
"ExitFunction OpSize:#Size, GPR:$NewRIP, BranchHint:$Hint, GPR:$CallReturnAddress, SSA:$CallReturnBlock": {
"Desc": ["Exits the current JIT function with a target RIP"
"ExitFunction OpSize:#Size, GPR:$NewRIP, BranchHint:$Hint, GPR:$CallReturnAddress, SSA:$CallReturnBlock, i64:$PatchSiteAddress{0}, i64:$PatchSiteSize{0}": {
"Desc": ["Exits the current JIT function with a target RIP - optionally patchable from guest bytes for caching"
],
"Inline": ["Any"],
"HasSideEffects": true,
+1 -1
View File
@@ -221,7 +221,7 @@ namespace DiskCache {
// TODO: This header is in global installed header path, but uses internal headers.
// Migrate this once that is fixed.
static constexpr uint16_t FormatVersion = 17;
static constexpr uint16_t FormatVersion = 18;
FEX_DEFAULT_VISIBILITY uint16_t GetFormatVersion();
} // namespace DiskCache
+1 -1
View File
@@ -8,7 +8,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"roundss xmm0, xmm1, 00000000b": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
@@ -9,7 +9,7 @@
"DisabledHostFeatures": [
"RPRES"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"cvtsi2ss xmm0, eax": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"cvtsi2sd xmm0, eax": {
@@ -8,7 +8,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
@@ -9,7 +9,7 @@
"SVE256",
"RPRES"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"cvtsi2ss xmm0, eax": {
@@ -8,7 +8,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"cvtsi2sd xmm0, eax": {
@@ -9,7 +9,7 @@
"DisabledHostFeatures": [
"RPRES"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vsqrtss xmm0, xmm1, xmm2": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vroundss xmm0, xmm1, 00000000b": {
@@ -7,7 +7,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vfmaddsubps xmm0, xmm1, xmm2, xmm3": {
@@ -13,7 +13,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vmovups xmm0, xmm0": {
@@ -9,7 +9,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vaddsubpd xmm0, xmm1, xmm2": {
@@ -10,7 +10,7 @@
"FLAGM2",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vmovntps [rax], xmm0": {
@@ -10,7 +10,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vucomiss xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vpshufb xmm0, xmm1, xmm2": {
@@ -8,7 +8,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vfmadd132ss xmm0, xmm1, xmm2": {
@@ -10,7 +10,7 @@
"FLAGM2",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vmovntdqa xmm0, [rax]": {
@@ -11,7 +11,7 @@
"SVE256",
"SVEBITPERM"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vtestps xmm0, xmm1": {
@@ -7,7 +7,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vpermq ymm0, ymm1, 1": {
@@ -8,7 +8,7 @@
"AFP",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vblendvps xmm0, xmm1, xmm2, xmm3": {
@@ -9,7 +9,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vpsrlw xmm0, xmm1, 0": {
+1 -1
View File
@@ -9,7 +9,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"lock add byte [rax], cl": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"sha1nexte xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"sha1nexte xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"pclmulqdq xmm0, xmm1, 00000b": {
+1 -1
View File
@@ -10,7 +10,7 @@
"AFP",
"RPRES"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"These 3DNow! instructions are optimal assuming that FEX doesn't SRA MMX registers",
@@ -8,7 +8,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"Instructions that explicitly push against the limits of ARM's loadstore instructions"
@@ -8,7 +8,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"Instructions that explicitly push against the limits of ARM's loadstore instructions"
@@ -13,7 +13,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -11,7 +11,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -9,7 +9,7 @@
"SVE256",
"RPRES"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -14,7 +14,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -14,7 +14,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -12,7 +12,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [],
"Instructions": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"lock add byte [rax], cl": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Chained add": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"ptest xmm0, xmm1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"The Witcher 3": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Sonic Mania movie player": {
@@ -10,7 +10,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"FMOD scalar loop": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"The Sims 1 hot block": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"add bl, cl": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"add al, 1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"push es": {
@@ -11,7 +11,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"ucomiss xmm0, xmm1": {
@@ -11,7 +11,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"sgdt [rax]": {
@@ -11,7 +11,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"xgetbv": {
@@ -11,7 +11,7 @@
"AFP",
"FCMA"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"ucomisd xmm0, xmm1": {
@@ -12,7 +12,7 @@
"AFP",
"CSSC"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"popcnt ax, bx": {
@@ -12,7 +12,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"popcnt ax, bx": {
@@ -12,7 +12,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vucomiss xmm0, xmm1": {
@@ -10,7 +10,7 @@
"AFP",
"SVEBITPERM"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vtestps xmm0, xmm1": {
@@ -9,7 +9,7 @@
"DisabledHostFeatures": [
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"blsr eax, ebx": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Block1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Block1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Block1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Block1": {
+1 -1
View File
@@ -11,7 +11,7 @@
"CSSC",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"fadd dword [rax]": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"fadd dword [rax]": {
+1 -1
View File
@@ -10,7 +10,7 @@
"FLAGM2",
"CRYPTO"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"pshufb mm0, mm1": {
+1 -1
View File
@@ -8,7 +8,7 @@
"AFP",
"CRYPTO"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"SSE4.2 string instructions are skipped here.",
@@ -8,7 +8,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"dpps xmm0, xmm1, 00000000b": {
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"rep movsb": {
+1 -1
View File
@@ -10,7 +10,7 @@
"FLAGM2",
"MOPS"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"add bl, cl": {
@@ -9,7 +9,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"Instructions in this table that are marked optimal don't have their flag calculation part of this assumption",
@@ -9,7 +9,7 @@
"FlagM",
"FlagM2"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"push es": {
+1 -1
View File
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"pfrcpv mm0, mm1": {
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"rsqrtps xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"rsqrtss xmm0, xmm1": {
@@ -8,7 +8,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"vrsqrtps xmm0, xmm1": {
+1 -1
View File
@@ -6,7 +6,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {}
}
@@ -33,7 +33,7 @@
" - 1b: ECX = MSB",
"[7] - Reserved"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"pcmpestrm xmm0, xmm1, 0_0_00_00_00b": {
+1 -1
View File
@@ -12,7 +12,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Comment": [
"MMX instructions are defined as optimal without SRA being used for these instructions.",
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"sgdt [rax]": {
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"xgetbv": {
@@ -7,7 +7,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"push fs": {
@@ -12,7 +12,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"movupd xmm0, xmm0": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"addsubpd xmm0, xmm1": {
@@ -8,7 +8,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"psrlw xmm0, xmm1": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"pmulhuw xmm0, xmm1": {
@@ -12,7 +12,7 @@
"FRINTTS",
"CSSC"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"movss xmm0, xmm1": {
@@ -10,7 +10,7 @@
"FCMA",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"movsd xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 17
"BinaryCacheVersion": 18
},
"Instructions": {
"addsubps xmm0, xmm1": {
Loaded 100 of 115 files, more files were not shown because too many files have changed in this diff. Show more