Merge pull request #3169 from Sonicadvance1/remove_constant_indirection

FEXCore: Support CpuState relative vector named constants
This commit is contained in:
Alyssa Rosenzweig authored and GitHub committed 2023-10-05 08:22:48 -04:00
commit 3413eb3d98
14 files changed
+72 -50

No files matched your search

@@ -186,6 +186,9 @@ CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t I
Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]); Common.NamedVectorConstantPointers[i] = reinterpret_cast<uint64_t>(NamedVectorConstants[i]);
} }
// Copy named vector constants.
memcpy(Common.NamedVectorConstants, NamedVectorConstants, sizeof(NamedVectorConstants));
// Initialize Indexed named vector constants. // Initialize Indexed named vector constants.
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] = reinterpret_cast<uint64_t>(PSHUFLW_LUT.data()); Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] = reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data()); Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
@@ -101,28 +101,42 @@ DEF_OP(LoadNamedVectorConstant) {
} }
} }
// Load the pointer. // Load the pointer.
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[Op->Constant])); auto GenerateMemOperand = [this](uint8_t OpSize, uint32_t NamedConstant, FEXCore::ARMEmitter::Register Base) {
const auto ConstantOffset = offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.NamedVectorConstants[NamedConstant]);
if (ConstantOffset <= 255 || // Unscaled 9-bit signed
((ConstantOffset & (OpSize - 1)) == 0 && FEXCore::DividePow2(ConstantOffset, OpSize) <= 4095)) /* 12-bit unsigned scaled */ {
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, ConstantOffset);
}
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[NamedConstant]));
return ARMEmitter::ExtendedMemOperand(TMP1, ARMEmitter::IndexType::OFFSET, 0);
};
if (OpSize == 32) {
// Handle SVE 32-byte variant upfront.
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.NamedVectorConstantPointers[Op->Constant]));
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), TMP1, 0);
return;
}
auto MemOperand = GenerateMemOperand(OpSize, Op->Constant, STATE);
switch (OpSize) { switch (OpSize) {
case 1: case 1:
ldrb(Dst, TMP1, 0); ldrb(Dst, MemOperand);
break; break;
case 2: case 2:
ldrh(Dst, TMP1, 0); ldrh(Dst, MemOperand);
break; break;
case 4: case 4:
ldr(Dst.S(), TMP1, 0); ldr(Dst.S(), MemOperand);
break; break;
case 8: case 8:
ldr(Dst.D(), TMP1, 0); ldr(Dst.D(), MemOperand);
break; break;
case 16: case 16:
ldr(Dst.Q(), TMP1, 0); ldr(Dst.Q(), MemOperand);
break; break;
case 32: {
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), TMP1, 0);
break;
}
default: default:
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize); LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, OpSize);
break; break;
+3
View File
@@ -259,6 +259,9 @@ namespace FEXCore::Core {
uint64_t L1Pointer{}; uint64_t L1Pointer{};
uint64_t L2Pointer{}; uint64_t L2Pointer{};
/** @} */ /** @} */
// Copy of process-wide named vector constants data.
alignas(16) uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2];
} Common; } Common;
union { union {
@@ -25,4 +25,12 @@ requires(std::is_unsigned_v<T>)
return std::countr_zero(Value); return std::countr_zero(Value);
} }
// Divide a number by a power-of-2 by avoiding integer division.
// Can be a faster implementation than regular integer divide.
// Divisor requires to be power-of-2, is enforced in ilog2 helper.
template<typename T, typename TT>
requires(std::is_unsigned_v<T> && std::is_unsigned_v<TT>)
[[nodiscard]] constexpr T DividePow2(T Dividend, TT Divisor) {
return Dividend >> ilog2(Divisor);
}
} // namespace FEXCore } // namespace FEXCore
+6
View File
@@ -5,3 +5,9 @@ TEST_CASE("ILog2") {
auto i = GENERATE(range(0, 64)); auto i = GENERATE(range(0, 64));
REQUIRE(FEXCore::ilog2(1ull << i) == i); REQUIRE(FEXCore::ilog2(1ull << i) == i);
} }
TEST_CASE("DividePow2") {
auto j = GENERATE(range(0, 64));
auto i = GENERATE(range(0, 64));
REQUIRE(FEXCore::DividePow2(1ull << j, 1ull << i) == ((1ull << j) / (1ull << i)));
}
+2 -3
View File
@@ -674,14 +674,13 @@
] ]
}, },
"phminposuw xmm0, xmm1": { "phminposuw xmm0, xmm1": {
"ExpectedInstructionCount": 7, "ExpectedInstructionCount": 6,
"Optimal": "Yes", "Optimal": "Yes",
"Comment": [ "Comment": [
"0x66 0x0f 0x38 0x41" "0x66 0x0f 0x38 0x41"
], ],
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1544]", "ldr q2, [x28, #1888]",
"ldr q2, [x0]",
"zip1 v3.8h, v2.8h, v17.8h", "zip1 v3.8h, v2.8h, v17.8h",
"zip2 v2.8h, v2.8h, v17.8h", "zip2 v2.8h, v2.8h, v17.8h",
"umin v2.4s, v3.4s, v2.4s", "umin v2.4s, v3.4s, v2.4s",
+4 -6
View File
@@ -1550,14 +1550,13 @@
] ]
}, },
"aeskeygenassist xmm0, xmm1, 0": { "aeskeygenassist xmm0, xmm1, 0": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 5,
"Optimal": "Yes", "Optimal": "Yes",
"Comment": [ "Comment": [
"0x66 0x0f 0x3a 0xdf" "0x66 0x0f 0x3a 0xdf"
], ],
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1600]", "ldr q2, [x28, #2000]",
"ldr q2, [x0]",
"movi v3.2d, #0x0", "movi v3.2d, #0x0",
"mov v16.16b, v17.16b", "mov v16.16b, v17.16b",
"unimplemented (Unimplemented)", "unimplemented (Unimplemented)",
@@ -1565,14 +1564,13 @@
] ]
}, },
"aeskeygenassist xmm0, xmm1, 0xFF": { "aeskeygenassist xmm0, xmm1, 0xFF": {
"ExpectedInstructionCount": 9, "ExpectedInstructionCount": 8,
"Optimal": "No", "Optimal": "No",
"Comment": [ "Comment": [
"0x66 0x0f 0x3a 0xdf" "0x66 0x0f 0x3a 0xdf"
], ],
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1600]", "ldr q2, [x28, #2000]",
"ldr q2, [x0]",
"movi v3.2d, #0x0", "movi v3.2d, #0x0",
"mov v16.16b, v17.16b", "mov v16.16b, v17.16b",
"unimplemented (Unimplemented)", "unimplemented (Unimplemented)",
@@ -3795,7 +3795,7 @@
"mov x0, x6", "mov x0, x6",
"mov x1, x20", "mov x1, x20",
"mov x2, x7", "mov x2, x7",
"ldr x3, [x28, #1912]", "ldr x3, [x28, #2048]",
"str x30, [sp, #-16]!", "str x30, [sp, #-16]!",
"blr x3", "blr x3",
"ldr x30, [sp], #16", "ldr x30, [sp], #16",
@@ -3806,7 +3806,7 @@
"mov x0, x6", "mov x0, x6",
"mov x1, x20", "mov x1, x20",
"mov x2, x7", "mov x2, x7",
"ldr x3, [x28, #1928]", "ldr x3, [x28, #2064]",
"str x30, [sp, #-16]!", "str x30, [sp, #-16]!",
"blr x3", "blr x3",
"ldr x30, [sp], #16", "ldr x30, [sp], #16",
@@ -3870,7 +3870,7 @@
"mov x0, x6", "mov x0, x6",
"mov x1, x20", "mov x1, x20",
"mov x2, x7", "mov x2, x7",
"ldr x3, [x28, #1920]", "ldr x3, [x28, #2056]",
"str x30, [sp, #-16]!", "str x30, [sp, #-16]!",
"blr x3", "blr x3",
"ldr x30, [sp], #16", "ldr x30, [sp], #16",
@@ -3883,7 +3883,7 @@
"mov x0, x6", "mov x0, x6",
"mov x1, x20", "mov x1, x20",
"mov x2, x7", "mov x2, x7",
"ldr x3, [x28, #1936]", "ldr x3, [x28, #2072]",
"str x30, [sp, #-16]!", "str x30, [sp, #-16]!",
"blr x3", "blr x3",
"ldr x30, [sp], #16", "ldr x30, [sp], #16",
+4 -6
View File
@@ -833,26 +833,24 @@
] ]
}, },
"movmskps eax, xmm0": { "movmskps eax, xmm0": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 5,
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0x50", "Comment": "0x0f 0x50",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ushr v2.4s, v16.4s, #31", "ushr v2.4s, v16.4s, #31",
"ldr x0, [x28, #1592]", "ldr q3, [x28, #1984]",
"ldr q3, [x0]",
"ushl v2.4s, v2.4s, v3.4s", "ushl v2.4s, v2.4s, v3.4s",
"addv s2, v2.4s", "addv s2, v2.4s",
"mov w4, v2.s[0]" "mov w4, v2.s[0]"
] ]
}, },
"movmskps rax, xmm0": { "movmskps rax, xmm0": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 5,
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0x50", "Comment": "0x0f 0x50",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ushr v2.4s, v16.4s, #31", "ushr v2.4s, v16.4s, #31",
"ldr x0, [x28, #1592]", "ldr q3, [x28, #1984]",
"ldr q3, [x0]",
"ushl v2.4s, v2.4s, v3.4s", "ushl v2.4s, v2.4s, v3.4s",
"addv s2, v2.4s", "addv s2, v2.4s",
"mov w4, v2.s[0]" "mov w4, v2.s[0]"
@@ -1130,12 +1130,11 @@
] ]
}, },
"addsubpd xmm0, xmm1": { "addsubpd xmm0, xmm1": {
"ExpectedInstructionCount": 4, "ExpectedInstructionCount": 3,
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x66 0x0f 0xd0", "Comment": "0x66 0x0f 0xd0",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1576]", "ldr q2, [x28, #1952]",
"ldr q2, [x0]",
"eor v2.16b, v17.16b, v2.16b", "eor v2.16b, v17.16b, v2.16b",
"fadd v16.2d, v16.2d, v2.2d" "fadd v16.2d, v16.2d, v2.2d"
] ]
@@ -511,12 +511,11 @@
] ]
}, },
"addsubps xmm0, xmm1": { "addsubps xmm0, xmm1": {
"ExpectedInstructionCount": 4, "ExpectedInstructionCount": 3,
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0xf2 0x0f 0xd0", "Comment": "0xf2 0x0f 0xd0",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1560]", "ldr q2, [x28, #1920]",
"ldr q2, [x0]",
"eor v2.16b, v17.16b, v2.16b", "eor v2.16b, v17.16b, v2.16b",
"fadd v16.4s, v16.4s, v2.4s" "fadd v16.4s, v16.4s, v2.4s"
] ]
+4 -6
View File
@@ -4449,14 +4449,13 @@
] ]
}, },
"vaddsubpd xmm0, xmm1, xmm2": { "vaddsubpd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "No",
"Comment": [ "Comment": [
"Map 1 0b01 0xd0 128-bit" "Map 1 0b01 0xd0 128-bit"
], ],
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1576]", "ldr q2, [x28, #1952]",
"ldr q2, [x0]",
"eor v2.16b, v18.16b, v2.16b", "eor v2.16b, v18.16b, v2.16b",
"fadd v2.2d, v17.2d, v2.2d", "fadd v2.2d, v17.2d, v2.2d",
"mov v16.16b, v2.16b" "mov v16.16b, v2.16b"
@@ -4476,14 +4475,13 @@
] ]
}, },
"vaddsubps xmm0, xmm1, xmm2": { "vaddsubps xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "No",
"Comment": [ "Comment": [
"Map 1 0b11 0xd0 128-bit" "Map 1 0b11 0xd0 128-bit"
], ],
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1560]", "ldr q2, [x28, #1920]",
"ldr q2, [x0]",
"eor v2.16b, v18.16b, v2.16b", "eor v2.16b, v18.16b, v2.16b",
"fadd v2.4s, v17.4s, v2.4s", "fadd v2.4s, v17.4s, v2.4s",
"mov v16.16b, v2.16b" "mov v16.16b, v2.16b"
+2 -3
View File
@@ -1579,14 +1579,13 @@
] ]
}, },
"vphminposuw xmm0, xmm1": { "vphminposuw xmm0, xmm1": {
"ExpectedInstructionCount": 8, "ExpectedInstructionCount": 7,
"Optimal": "No", "Optimal": "No",
"Comment": [ "Comment": [
"Map 2 0b01 0x41 256-bit" "Map 2 0b01 0x41 256-bit"
], ],
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1544]", "ldr q2, [x28, #1888]",
"ldr q2, [x0]",
"zip1 v3.8h, v2.8h, v17.8h", "zip1 v3.8h, v2.8h, v17.8h",
"zip2 v2.8h, v2.8h, v17.8h", "zip2 v2.8h, v2.8h, v17.8h",
"umin v2.4s, v3.4s, v2.4s", "umin v2.4s, v3.4s, v2.4s",
+4 -6
View File
@@ -4713,14 +4713,13 @@
] ]
}, },
"vaeskeygenassist xmm0, xmm1, 0": { "vaeskeygenassist xmm0, xmm1, 0": {
"ExpectedInstructionCount": 7, "ExpectedInstructionCount": 6,
"Optimal": "No", "Optimal": "No",
"Comment": [ "Comment": [
"Map 3 0b01 0xdf 128-bit" "Map 3 0b01 0xdf 128-bit"
], ],
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1600]", "ldr q2, [x28, #2000]",
"ldr q2, [x0]",
"movi v3.2d, #0x0", "movi v3.2d, #0x0",
"mov v2.16b, v17.16b", "mov v2.16b, v17.16b",
"unimplemented (Unimplemented)", "unimplemented (Unimplemented)",
@@ -4729,14 +4728,13 @@
] ]
}, },
"vaeskeygenassist xmm0, xmm1, 0xFF": { "vaeskeygenassist xmm0, xmm1, 0xFF": {
"ExpectedInstructionCount": 10, "ExpectedInstructionCount": 9,
"Optimal": "No", "Optimal": "No",
"Comment": [ "Comment": [
"Map 3 0b01 0xdf 128-bit" "Map 3 0b01 0xdf 128-bit"
], ],
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr x0, [x28, #1600]", "ldr q2, [x28, #2000]",
"ldr q2, [x0]",
"movi v3.2d, #0x0", "movi v3.2d, #0x0",
"mov v2.16b, v17.16b", "mov v2.16b, v17.16b",
"unimplemented (Unimplemented)", "unimplemented (Unimplemented)",