FEXCore: Support CpuState relative vector named constants

The motivation towards just having a pointer array in CpuState was that
initialization was fairly cheap and that we have limited space inside
the encoding depending on what we want to do.

Initialization cost is still a concern but doing a memcpy of 128-bytes
isn't that big of a deal.

Limited space in CpuState, while a concern isn't a significant one.
   - Needs to currently be less than 1 page in size
   - Needs to be under the architectural offset limitations of loadstore
     scaled offsets. Which is 65KB for 128-bit vectors

Still keeps the pointer array around for cases when we would need
synthesize an address offset and it's just easier to load the
process-wide table.

The performance improvement here is removing the dependency in the
ldr+ldr chain. In microbenchmarks this has shown to have an improvement
of ~4% by removing this dependency chain on Cortex-X1C.
This commit is contained in:
Ryan Houdek committed 2023-10-04 20:56:29 -07:00
1 parent ee6debe8fd
commit 8a51bb7a61
3 files changed
+30 -10

No files matched your search

@@ -186,6 +186,9 @@ CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t I
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
}
// Copy named vector constants.
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
// Initialize Indexed named vector constants.
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] = reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
@@ -101,28 +101,42 @@ DEF_OP(LoadNamedVectorConstant) {
}
}
// Load the pointer.
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[Op->Constant]));
auto GenerateMemOperand = [this](uint8_t OpSize, uint32_t NamedConstant, FEXCore::ARMEmitter::Register Base) {
const auto ConstantOffset = offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.NamedVectorConstants[NamedConstant]);
if (ConstantOffset <= 255 || // Unscaled 9-bit signed
((ConstantOffset & (OpSize - 1)) == 0 && FEXCore::DividePow2(ConstantOffset, OpSize) <= 4095)) /* 12-bit unsigned scaled */ {
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, ConstantOffset);
}
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[NamedConstant]));
return ARMEmitter::ExtendedMemOperand(TMP1, ARMEmitter::IndexType::OFFSET, 0);
};
if (OpSize == 32) {
// Handle SVE 32-byte variant upfront.
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[Op->Constant]));
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), TMP1, 0);
return;
}
auto MemOperand = GenerateMemOperand(OpSize, Op->Constant, STATE);
switch (OpSize) {
case 1:
ldrb(Dst, TMP1, 0);
ldrb(Dst, MemOperand);
break;
case 2:
ldrh(Dst, TMP1, 0);
ldrh(Dst, MemOperand);
break;
case 4:
ldr(Dst.S(), TMP1, 0);
ldr(Dst.S(), MemOperand);
break;
case 8:
ldr(Dst.D(), TMP1, 0);
ldr(Dst.D(), MemOperand);
break;
case 16:
ldr(Dst.Q(), TMP1, 0);
ldr(Dst.Q(), MemOperand);
break;
case 32: {
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), TMP1, 0);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize);
break;
+3
View File
@@ -259,6 +259,9 @@ namespace FEXCore::Core {
uint64_t L1Pointer{};
uint64_t L2Pointer{};
/** @} */
// Copy of process-wide named vector constants data.
alignas(16) uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2];
} Common;
union {