Merge pull request #3087 from Sonicadvance1/tbl2_implementation

OpcodeDispatcher: Implement shufps with VTBL2 in worst case
This commit is contained in:
Mai authored and GitHub committed 2023-09-13 22:48:53 -04:00
commit 31d828390f
16 files changed
+255 -54

No files matched your search

@@ -126,6 +126,55 @@ constexpr static auto PSHUFD_LUT {
}()
};
constexpr static auto SHUFPS_LUT {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
// Expectation for this LUT is to simulate SHUFPS with ARM's TBL (two register) instruction.
// SHUFPS behaviour:
// Two 32-bits words from each source are selected from each source in the lower and upper halves of the 128-bit destination.
// Dest[31:0] = Src1[<Word0>]
// Dest[63:32] = Src1[<Word1>]
// Dest[95:64] = Src2[<Word2>]
// Dest[127:96] = Src2[<Word3>]
std::array<LUTType, 256> TotalLUT{};
const uint64_t WordSelectionSrc1[4] = {
0x03'02'01'00,
0x07'06'05'04,
0x0b'0a'09'08,
0x0f'0e'0d'0c,
};
// Src2 needs to offset each byte index by 16-bytes to pull from the second source.
const uint64_t WordSelectionSrc2[4] = {
0x03'02'01'00 + (0x10101010),
0x07'06'05'04 + (0x10101010),
0x0b'0a'09'08 + (0x10101010),
0x0f'0e'0d'0c + (0x10101010),
};
for (size_t i = 0; i < 256; ++i) {
auto &LUT = TotalLUT[i];
const auto Word0 = (i >> 0) & 0b11;
const auto Word1 = (i >> 2) & 0b11;
const auto Word2 = (i >> 4) & 0b11;
const auto Word3 = (i >> 6) & 0b11;
LUT.Val[0] =
(WordSelectionSrc1[Word0] << 0) |
(WordSelectionSrc1[Word1] << 32);
LUT.Val[1] =
(WordSelectionSrc2[Word2] << 0) |
(WordSelectionSrc2[Word3] << 32);
}
return TotalLUT;
}()
};
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {
@@ -140,6 +189,7 @@ CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t I
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] = reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
#ifndef FEX_DISABLE_TELEMETRY
// Fill in telemetry values
@@ -93,6 +93,8 @@ fextl::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::ContextImp
}
CPUBackendFeatures GetInterpreterBackendFeatures() {
return CPUBackendFeatures { };
return CPUBackendFeatures {
.SupportsVTBL2 = true,
};
}
}
@@ -281,6 +281,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
REGISTER_OP(VUABDL, VUABDL);
REGISTER_OP(VUABDL2, VUABDL2);
REGISTER_OP(VTBL1, VTBL1);
REGISTER_OP(VTBL2, VTBL2);
REGISTER_OP(VREV32, VRev32);
REGISTER_OP(VREV64, VRev64);
REGISTER_OP(VPCMPESTRX, VPCMPESTRX);
@@ -309,6 +309,7 @@ namespace FEXCore::CPU {
DEF_OP(VUABDL);
DEF_OP(VUABDL2);
DEF_OP(VTBL1);
DEF_OP(VTBL2);
DEF_OP(VRev32);
DEF_OP(VRev64);
DEF_OP(VPCMPESTRX);
@@ -66,7 +66,7 @@ DEF_OP(LoadNamedVectorIndexedConstant) {
auto Op = IROp->C<IR::IROp_LoadNamedVectorIndexedConstant>();
uint8_t OpSize = IROp->Size;
memcpy(GDP, reinterpret_cast<void*>(Data->State->CurrentFrame->Pointers.Common.NamedVectorConstantPointers[Op->Constant] + Op->Index), OpSize);
memcpy(GDP, reinterpret_cast<void*>(Data->State->CurrentFrame->Pointers.Common.IndexedNamedVectorConstantPointers[Op->Constant] + Op->Index), OpSize);
}
DEF_OP(VMov) {
@@ -2486,6 +2486,31 @@ DEF_OP(VTBL1) {
memcpy(GDP, Tmp.data(), OpSize);
}
DEF_OP(VTBL2) {
const auto Op = IROp->C<IR::IROp_VTBL2>();
const uint8_t OpSize = IROp->Size;
const auto *VectorTable1 = GetSrc<uint8_t*>(Data->SSAData, Op->VectorTable1);
const auto *VectorTable2 = GetSrc<uint8_t*>(Data->SSAData, Op->VectorTable2);
const auto *VectorIndices = GetSrc<uint8_t*>(Data->SSAData, Op->VectorIndices);
TempVectorDataArray Tmp;
for (size_t i = 0; i < OpSize; ++i) {
const uint8_t Index = VectorIndices[i];
if (Index >= (OpSize * 2)) {
Tmp[i] = 0;
}
else if (Index >= OpSize) {
Tmp[i] = VectorTable2[Index - OpSize];
}
else {
Tmp[i] = VectorTable1[Index];
}
}
memcpy(GDP, Tmp.data(), OpSize);
}
DEF_OP(VRev32) {
const auto Op = IROp->C<IR::IROp_VRev32>();
const uint8_t OpSize = IROp->Size;
@@ -1103,6 +1103,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
REGISTER_OP(VUABDL, VUABDL);
REGISTER_OP(VUABDL2, VUABDL2);
REGISTER_OP(VTBL1, VTBL1);
REGISTER_OP(VTBL2, VTBL2);
REGISTER_OP(VREV32, VRev32);
REGISTER_OP(VREV64, VRev64);
REGISTER_OP(VFCADD, VFCADD);
@@ -1231,6 +1232,7 @@ CPUBackendFeatures GetArm64JITBackendFeatures() {
.SupportsShiftedBitwise = true,
.SupportsFlags = true,
.SupportsSaturatingRoundingShifts = true,
.SupportsVTBL2 = true,
};
}
@@ -469,6 +469,7 @@ private:
DEF_OP(VUABDL);
DEF_OP(VUABDL2);
DEF_OP(VTBL1);
DEF_OP(VTBL2);
DEF_OP(VRev32);
DEF_OP(VRev64);
DEF_OP(VFCADD);
@@ -3845,6 +3845,52 @@ DEF_OP(VTBL1) {
}
}
DEF_OP(VTBL2) {
const auto Op = IROp->C<IR::IROp_VTBL2>();
const auto OpSize = IROp->Size;
const auto Dst = GetVReg(Node);
const auto VectorIndices = GetVReg(Op->VectorIndices.ID());
auto VectorTable1 = GetVReg(Op->VectorTable1.ID());
auto VectorTable2 = GetVReg(Op->VectorTable2.ID());
if (!AreVectorsSequential(VectorTable1, VectorTable2)) {
// Vector registers aren't sequential, need to move to temporaries.
if (OpSize == 32) {
mov(VTMP1.Z(), VectorTable1.Z());
mov(VTMP2.Z(), VectorTable2.Z());
}
else {
mov(VTMP1.Q(), VectorTable1.Q());
mov(VTMP2.Q(), VectorTable2.Q());
}
VectorTable1 = VTMP1;
VectorTable2 = VTMP2;
}
switch (OpSize) {
case 8: {
tbl(Dst.D(), VectorTable1.Q(), VectorTable2.Q(), VectorIndices.D());
break;
}
case 16: {
tbl(Dst.Q(), VectorTable1.Q(), VectorTable2.Q(), VectorIndices.Q());
break;
}
case 32: {
LOGMAN_THROW_AA_FMT(HostSupportsSVE256,
"Host does not support SVE. Cannot perform 256-bit table lookup");
tbl(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), VectorTable1.Z(), VectorTable2.Z(), VectorIndices.Z());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown OpSize: {}", OpSize);
break;
}
}
DEF_OP(VRev32) {
const auto Op = IROp->C<IR::IROp_VRev32>();
const auto OpSize = IROp->Size;
@@ -1384,9 +1384,11 @@ OrderedNode* OpDispatchBuilder::SHUFOpImpl(OpcodeArgs, size_t ElementSize,
return _VZip(DstSize, 8, DupSrc1, DupSrc2);
}
default:
// Intentional doing nothing.
// Uncomment if you want to see remaining fallthrough in the output logs.
//LogMan::Msg::DFmt("shufps: 0x{:x}", Shuffle);
// Use a TBL2 operation to handle this implementation. If the backend supports it.
if (CTX->BackendFeatures.SupportsVTBL2) {
auto LookupIndexes = LoadAndCacheIndexedNamedVectorConstant(DstSize, FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS, Shuffle * 16);
return _VTBL2(DstSize, Src1Node, Src2Node, LookupIndexes);
}
break;
}
}
+10 -1
View File
@@ -1726,7 +1726,16 @@
],
"DestSize": "RegisterSize"
},
"FPR = VTBL2 u8:#RegisterSize, FPR:$VectorTable1, FPR:$VectorTable2, FPR:$VectorIndices": {
"Desc": ["Does a vector table lookup from two registers in to the destination",
"Lookup is byte sized per byte element.",
"Any index larger than what the registers provide will result in zero for that element",
"Table is always treated as a two 128bit registers",
"Indices matches destination size. Either 64bit or 128bit",
"Careful about not using sequential table registers, will result in some moves if they aren't sequential."
],
"DestSize": "RegisterSize"
},
"FPR = VBSL u8:#RegisterSize, FPR:$VectorMask, FPR:$VectorTrue, FPR:$VectorFalse": {
"Desc": ["Does a vector bitwise select.",
"If the bit in the field is 1 then the corresponding bit is pulled from VectorTrue",
@@ -38,6 +38,7 @@ namespace CPU {
bool SupportsShiftedBitwise = false;
bool SupportsFlags = false;
bool SupportsSaturatingRoundingShifts = false;
bool SupportsVTBL2 = false;
};
class CPUBackend {
+1
View File
@@ -539,6 +539,7 @@ enum IndexNamedVectorConstant : uint8_t {
INDEXED_NAMED_VECTOR_PSHUFLW = 0,
INDEXED_NAMED_VECTOR_PSHUFHW,
INDEXED_NAMED_VECTOR_PSHUFD,
INDEXED_NAMED_VECTOR_SHUFPS,
INDEXED_NAMED_VECTOR_MAX,
};
@@ -0,0 +1,71 @@
%ifdef CONFIG
{
"RegData": {
"XMM0": ["0xc2f837b5c2f837b5", "0x7aa25d8d7aa25d8d"],
"XMM1": ["0x4044836d86ec17ec", "0x4054c664c2f837b5"],
"XMM2": ["0x4035fe425aee6320", "0x4054c664c2f837b5"],
"XMM3": ["0x402359003eea209b", "0x4054c664c2f837b5"],
"XMM4": ["0x4050a018bd66277c", "0x402359003eea209b"],
"XMM5": ["0x4ea4a8c17ebaf102", "0x3eea209bbd66277c"],
"XMM6": ["0x40497b13404439b5", "0x3eea209b4ea4a8c1"],
"XMM7": ["0x4040528b4040528b", "0x3eea209b4ea4a8c1"],
"XMM8": ["0x403839b8403839b8", "0x3eea209b4ea4a8c1"],
"XMM9": ["0x4056cde54056cde5", "0x403839b8403839b8"],
"XMM10": ["0x4056b34aa10e0221", "0x4056cde54056cde5"],
"XMM11": ["0x4052997f0ed3d85a", "0xa10e0221a10e0221"],
"XMM12": ["0x40419d2240395a6b", "0xa10e0221a10e0221"],
"XMM13": ["0x40177e2840568cc5", "0x40419d2240395a6b"],
"XMM14": ["0x9f16b11c40408402", "0x40419d2240395a6b"],
"XMM15": ["0x5feda661404d3159", "0x9f16b11c40408402"]
}
}
%endif
movaps xmm0, [rel .data + 16 * 0]
movaps xmm1, [rel .data + 16 * 1]
movaps xmm2, [rel .data + 16 * 2]
movaps xmm3, [rel .data + 16 * 3]
movaps xmm4, [rel .data + 16 * 4]
movaps xmm5, [rel .data + 16 * 5]
movaps xmm6, [rel .data + 16 * 6]
movaps xmm7, [rel .data + 16 * 7]
movaps xmm8, [rel .data + 16 * 8]
movaps xmm9, [rel .data + 16 * 9]
movaps xmm10, [rel .data + 16 * 10]
movaps xmm11, [rel .data + 16 * 11]
movaps xmm12, [rel .data + 16 * 12]
movaps xmm13, [rel .data + 16 * 13]
movaps xmm14, [rel .data + 16 * 14]
movaps xmm15, [rel .data + 16 * 15]
; Test inverted sources from shufps_optimization.asm
shufps xmm1, xmm0, 01000100b
shufps xmm0, [rel .data + 16 * 16], 0
shufps xmm2, xmm1, 11101110b
shufps xmm3, xmm2, 11100100b
shufps xmm4, xmm3, 01001110b
shufps xmm5, xmm4, 10001000b
shufps xmm6, xmm5, 11011101b
shufps xmm7, xmm6, 11100101b
shufps xmm8, xmm7, 11101111b
shufps xmm9, xmm8, 01001111b
shufps xmm10, xmm9, 00000100b
shufps xmm11, xmm10, 00001110b
shufps xmm12, xmm11, 11100111b
shufps xmm13, xmm12, 01000111b
shufps xmm14, xmm13, 11100001b
shufps xmm15, xmm14, 01000001b
hlt
align 16
; 512bytes of random data
.data:
dq 83.0999,69.50512,41.02678,13.05881,5.35242,21.9932,9.67383,5.32372,29.02872,66.50151,19.30764,91.3633,40.45086,50.96153,32.64489,23.97574,90.64316,24.22547,98.9394,91.21715,90.80143,99.48407,64.97245,74.39838,35.22761,25.35321,5.8732,90.19956,33.03133,52.02952,58.38554,10.17531,47.84703,84.04831,90.02965,65.81329,96.27991,6.64479,25.58971,95.00694,88.1929,37.16964,49.52602,10.27223,77.70605,20.21439,9.8056,41.29389,15.4071,57.54286,9.61117,55.54302,52.90745,4.88086,72.52882,3.0201,56.55091,71.22749,61.84736,88.74295,47.72641,24.17404,33.70564,96.71303
@@ -3232,7 +3232,7 @@
"mov x0, x6",
"mov x1, x20",
"mov x2, x7",
"ldr x3, [x28, #1912]",
"ldr x3, [x28, #1920]",
"str x30, [sp, #-16]!",
"blr x3",
"ldr x30, [sp], #16",
@@ -3243,7 +3243,7 @@
"mov x0, x6",
"mov x1, x20",
"mov x2, x7",
"ldr x3, [x28, #1928]",
"ldr x3, [x28, #1936]",
"str x30, [sp, #-16]!",
"blr x3",
"ldr x30, [sp], #16",
@@ -3307,7 +3307,7 @@
"mov x0, x6",
"mov x1, x20",
"mov x2, x7",
"ldr x3, [x28, #1920]",
"ldr x3, [x28, #1928]",
"str x30, [sp, #-16]!",
"blr x3",
"ldr x30, [sp], #16",
@@ -3320,7 +3320,7 @@
"mov x0, x6",
"mov x1, x20",
"mov x2, x7",
"ldr x3, [x28, #1936]",
"ldr x3, [x28, #1944]",
"str x30, [sp, #-16]!",
"blr x3",
"ldr x30, [sp], #16",
+21 -17
View File
@@ -4131,34 +4131,38 @@
]
},
"shufps xmm0, xmm1, 1": {
"ExpectedInstructionCount": 8,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": "0x0f 0xc6",
"ExpectedArm64ASM": [
"mov v0.16b, v16.16b",
"mov v0.s[0], v16.s[1]",
"mov v2.16b, v0.16b",
"mov v2.s[1], v16.s[0]",
"mov v2.s[2], v17.s[0]",
"mov v0.16b, v2.16b",
"mov v0.s[3], v17.s[0]",
"mov v16.16b, v0.16b"
"ldr x0, [x28, #1640]",
"ldr q2, [x0, #16]",
"tbl v16.16b, {v16.16b, v17.16b}, v2.16b"
]
},
"shufps xmm1, xmm0, 1": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Comment": "0x0f 0xc6",
"ExpectedArm64ASM": [
"ldr x0, [x28, #1640]",
"ldr q2, [x0, #16]",
"mov v0.16b, v17.16b",
"mov v1.16b, v16.16b",
"tbl v17.16b, {v0.16b, v1.16b}, v2.16b"
]
},
"shufps xmm0, [rax], 1": {
"ExpectedInstructionCount": 9,
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "0x0f 0xc6",
"ExpectedArm64ASM": [
"ldr q2, [x4]",
"ldr x0, [x28, #1640]",
"ldr q3, [x0, #16]",
"mov v0.16b, v16.16b",
"mov v0.s[0], v16.s[1]",
"mov v3.16b, v0.16b",
"mov v3.s[1], v16.s[0]",
"mov v3.s[2], v2.s[0]",
"mov v0.16b, v3.16b",
"mov v0.s[3], v2.s[0]",
"mov v16.16b, v0.16b"
"mov v1.16b, v2.16b",
"tbl v16.16b, {v0.16b, v1.16b}, v3.16b"
]
},
"shufps xmm0, [rax], 0xFF": {
+12 -27
View File
@@ -3207,7 +3207,7 @@
]
},
"vshufps xmm0, xmm1, xmm2, 01b": {
"ExpectedInstructionCount": 12,
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": [
"Map 1 0b00 0xC6 128-bit"
@@ -3215,14 +3215,9 @@
"ExpectedArm64ASM": [
"mov z2.d, p7/m, z17.d",
"mov z3.d, p7/m, z18.d",
"mov v0.16b, v2.16b",
"mov v0.s[0], v2.s[1]",
"mov v4.16b, v0.16b",
"mov v0.16b, v4.16b",
"mov v0.s[1], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v2.s[2], v3.s[0]",
"mov v2.s[3], v3.s[0]",
"ldr x0, [x28, #1640]",
"ldr q4, [x0, #16]",
"tbl v2.16b, {v2.16b, v3.16b}, v4.16b",
"mov v2.16b, v2.16b",
"mov z16.d, p7/m, z2.d"
]
@@ -3274,7 +3269,7 @@
]
},
"vshufps xmm0, xmm1, xmm2, 10b": {
"ExpectedInstructionCount": 12,
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": [
"Map 1 0b00 0xC6 128-bit"
@@ -3282,14 +3277,9 @@
"ExpectedArm64ASM": [
"mov z2.d, p7/m, z17.d",
"mov z3.d, p7/m, z18.d",
"mov v0.16b, v2.16b",
"mov v0.s[0], v2.s[2]",
"mov v4.16b, v0.16b",
"mov v0.16b, v4.16b",
"mov v0.s[1], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v2.s[2], v3.s[0]",
"mov v2.s[3], v3.s[0]",
"ldr x0, [x28, #1640]",
"ldr q4, [x0, #32]",
"tbl v2.16b, {v2.16b, v3.16b}, v4.16b",
"mov v2.16b, v2.16b",
"mov z16.d, p7/m, z2.d"
]
@@ -3341,7 +3331,7 @@
]
},
"vshufps xmm0, xmm1, xmm2, 11b": {
"ExpectedInstructionCount": 12,
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": [
"Map 1 0b00 0xC6 128-bit"
@@ -3349,14 +3339,9 @@
"ExpectedArm64ASM": [
"mov z2.d, p7/m, z17.d",
"mov z3.d, p7/m, z18.d",
"mov v0.16b, v2.16b",
"mov v0.s[0], v2.s[3]",
"mov v4.16b, v0.16b",
"mov v0.16b, v4.16b",
"mov v0.s[1], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v2.s[2], v3.s[0]",
"mov v2.s[3], v3.s[0]",
"ldr x0, [x28, #1640]",
"ldr q4, [x0, #48]",
"tbl v2.16b, {v2.16b, v3.16b}, v4.16b",
"mov v2.16b, v2.16b",
"mov z16.d, p7/m, z2.d"
]