mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 09:00:17 +02:00
Merge pull request #3087 from Sonicadvance1/tbl2_implementation
OpcodeDispatcher: Implement shufps with VTBL2 in worst case
This commit is contained in:
16 files changed
+255
-54
No files matched your search
@@ -126,6 +126,55 @@ constexpr static auto PSHUFD_LUT {
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto SHUFPS_LUT {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
// 32-bit words in [127:96], [95:64], [63:32], [31:0] are selected using the 8-bit Index.
|
||||
// Expectation for this LUT is to simulate SHUFPS with ARM's TBL (two register) instruction.
|
||||
// SHUFPS behaviour:
|
||||
// Two 32-bits words from each source are selected from each source in the lower and upper halves of the 128-bit destination.
|
||||
// Dest[31:0] = Src1[<Word0>]
|
||||
// Dest[63:32] = Src1[<Word1>]
|
||||
// Dest[95:64] = Src2[<Word2>]
|
||||
// Dest[127:96] = Src2[<Word3>]
|
||||
|
||||
std::array<LUTType, 256> TotalLUT{};
|
||||
const uint64_t WordSelectionSrc1[4] = {
|
||||
0x03'02'01'00,
|
||||
0x07'06'05'04,
|
||||
0x0b'0a'09'08,
|
||||
0x0f'0e'0d'0c,
|
||||
};
|
||||
|
||||
// Src2 needs to offset each byte index by 16-bytes to pull from the second source.
|
||||
const uint64_t WordSelectionSrc2[4] = {
|
||||
0x03'02'01'00 + (0x10101010),
|
||||
0x07'06'05'04 + (0x10101010),
|
||||
0x0b'0a'09'08 + (0x10101010),
|
||||
0x0f'0e'0d'0c + (0x10101010),
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
const auto Word0 = (i >> 0) & 0b11;
|
||||
const auto Word1 = (i >> 2) & 0b11;
|
||||
const auto Word2 = (i >> 4) & 0b11;
|
||||
const auto Word3 = (i >> 6) & 0b11;
|
||||
|
||||
LUT.Val[0] =
|
||||
(WordSelectionSrc1[Word0] << 0) |
|
||||
(WordSelectionSrc1[Word1] << 32);
|
||||
|
||||
LUT.Val[1] =
|
||||
(WordSelectionSrc2[Word2] << 0) |
|
||||
(WordSelectionSrc2[Word3] << 32);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {
|
||||
|
||||
@@ -140,6 +189,7 @@ CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t I
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW] = reinterpret_cast<uint64_t>(PSHUFLW_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
|
||||
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
// Fill in telemetry values
|
||||
|
||||
@@ -93,6 +93,8 @@ fextl::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::ContextImp
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
return CPUBackendFeatures {
|
||||
.SupportsVTBL2 = true,
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -281,6 +281,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VTBL2, VTBL2);
|
||||
REGISTER_OP(VREV32, VRev32);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VPCMPESTRX, VPCMPESTRX);
|
||||
|
||||
@@ -309,6 +309,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VTBL2);
|
||||
DEF_OP(VRev32);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VPCMPESTRX);
|
||||
|
||||
@@ -66,7 +66,7 @@ DEF_OP(LoadNamedVectorIndexedConstant) {
|
||||
auto Op = IROp->C<IR::IROp_LoadNamedVectorIndexedConstant>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
memcpy(GDP, reinterpret_cast<void*>(Data->State->CurrentFrame->Pointers.Common.NamedVectorConstantPointers[Op->Constant] + Op->Index), OpSize);
|
||||
memcpy(GDP, reinterpret_cast<void*>(Data->State->CurrentFrame->Pointers.Common.IndexedNamedVectorConstantPointers[Op->Constant] + Op->Index), OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VMov) {
|
||||
@@ -2486,6 +2486,31 @@ DEF_OP(VTBL1) {
|
||||
memcpy(GDP, Tmp.data(), OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VTBL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto *VectorTable1 = GetSrc<uint8_t*>(Data->SSAData, Op->VectorTable1);
|
||||
const auto *VectorTable2 = GetSrc<uint8_t*>(Data->SSAData, Op->VectorTable2);
|
||||
const auto *VectorIndices = GetSrc<uint8_t*>(Data->SSAData, Op->VectorIndices);
|
||||
|
||||
TempVectorDataArray Tmp;
|
||||
|
||||
for (size_t i = 0; i < OpSize; ++i) {
|
||||
const uint8_t Index = VectorIndices[i];
|
||||
if (Index >= (OpSize * 2)) {
|
||||
Tmp[i] = 0;
|
||||
}
|
||||
else if (Index >= OpSize) {
|
||||
Tmp[i] = VectorTable2[Index - OpSize];
|
||||
}
|
||||
else {
|
||||
Tmp[i] = VectorTable1[Index];
|
||||
}
|
||||
}
|
||||
memcpy(GDP, Tmp.data(), OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VRev32) {
|
||||
const auto Op = IROp->C<IR::IROp_VRev32>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
@@ -1103,6 +1103,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VTBL2, VTBL2);
|
||||
REGISTER_OP(VREV32, VRev32);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VFCADD, VFCADD);
|
||||
@@ -1231,6 +1232,7 @@ CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
.SupportsShiftedBitwise = true,
|
||||
.SupportsFlags = true,
|
||||
.SupportsSaturatingRoundingShifts = true,
|
||||
.SupportsVTBL2 = true,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -469,6 +469,7 @@ private:
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VTBL2);
|
||||
DEF_OP(VRev32);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VFCADD);
|
||||
|
||||
@@ -3845,6 +3845,52 @@ DEF_OP(VTBL1) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VTBL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL2>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices.ID());
|
||||
auto VectorTable1 = GetVReg(Op->VectorTable1.ID());
|
||||
auto VectorTable2 = GetVReg(Op->VectorTable2.ID());
|
||||
|
||||
if (!AreVectorsSequential(VectorTable1, VectorTable2)) {
|
||||
// Vector registers aren't sequential, need to move to temporaries.
|
||||
if (OpSize == 32) {
|
||||
mov(VTMP1.Z(), VectorTable1.Z());
|
||||
mov(VTMP2.Z(), VectorTable2.Z());
|
||||
}
|
||||
else {
|
||||
mov(VTMP1.Q(), VectorTable1.Q());
|
||||
mov(VTMP2.Q(), VectorTable2.Q());
|
||||
}
|
||||
|
||||
VectorTable1 = VTMP1;
|
||||
VectorTable2 = VTMP2;
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
case 8: {
|
||||
tbl(Dst.D(), VectorTable1.Q(), VectorTable2.Q(), VectorIndices.D());
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
tbl(Dst.Q(), VectorTable1.Q(), VectorTable2.Q(), VectorIndices.Q());
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE256,
|
||||
"Host does not support SVE. Cannot perform 256-bit table lookup");
|
||||
|
||||
tbl(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), VectorTable1.Z(), VectorTable2.Z(), VectorIndices.Z());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown OpSize: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VRev32) {
|
||||
const auto Op = IROp->C<IR::IROp_VRev32>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -1384,9 +1384,11 @@ OrderedNode* OpDispatchBuilder::SHUFOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
return _VZip(DstSize, 8, DupSrc1, DupSrc2);
|
||||
}
|
||||
default:
|
||||
// Intentional doing nothing.
|
||||
// Uncomment if you want to see remaining fallthrough in the output logs.
|
||||
//LogMan::Msg::DFmt("shufps: 0x{:x}", Shuffle);
|
||||
// Use a TBL2 operation to handle this implementation. If the backend supports it.
|
||||
if (CTX->BackendFeatures.SupportsVTBL2) {
|
||||
auto LookupIndexes = LoadAndCacheIndexedNamedVectorConstant(DstSize, FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS, Shuffle * 16);
|
||||
return _VTBL2(DstSize, Src1Node, Src2Node, LookupIndexes);
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1726,7 +1726,16 @@
|
||||
],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"FPR = VTBL2 u8:#RegisterSize, FPR:$VectorTable1, FPR:$VectorTable2, FPR:$VectorIndices": {
|
||||
"Desc": ["Does a vector table lookup from two registers in to the destination",
|
||||
"Lookup is byte sized per byte element.",
|
||||
"Any index larger than what the registers provide will result in zero for that element",
|
||||
"Table is always treated as a two 128bit registers",
|
||||
"Indices matches destination size. Either 64bit or 128bit",
|
||||
"Careful about not using sequential table registers, will result in some moves if they aren't sequential."
|
||||
],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VBSL u8:#RegisterSize, FPR:$VectorMask, FPR:$VectorTrue, FPR:$VectorFalse": {
|
||||
"Desc": ["Does a vector bitwise select.",
|
||||
"If the bit in the field is 1 then the corresponding bit is pulled from VectorTrue",
|
||||
|
||||
@@ -38,6 +38,7 @@ namespace CPU {
|
||||
bool SupportsShiftedBitwise = false;
|
||||
bool SupportsFlags = false;
|
||||
bool SupportsSaturatingRoundingShifts = false;
|
||||
bool SupportsVTBL2 = false;
|
||||
};
|
||||
|
||||
class CPUBackend {
|
||||
|
||||
@@ -539,6 +539,7 @@ enum IndexNamedVectorConstant : uint8_t {
|
||||
INDEXED_NAMED_VECTOR_PSHUFLW = 0,
|
||||
INDEXED_NAMED_VECTOR_PSHUFHW,
|
||||
INDEXED_NAMED_VECTOR_PSHUFD,
|
||||
INDEXED_NAMED_VECTOR_SHUFPS,
|
||||
INDEXED_NAMED_VECTOR_MAX,
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"XMM0": ["0xc2f837b5c2f837b5", "0x7aa25d8d7aa25d8d"],
|
||||
"XMM1": ["0x4044836d86ec17ec", "0x4054c664c2f837b5"],
|
||||
"XMM2": ["0x4035fe425aee6320", "0x4054c664c2f837b5"],
|
||||
"XMM3": ["0x402359003eea209b", "0x4054c664c2f837b5"],
|
||||
"XMM4": ["0x4050a018bd66277c", "0x402359003eea209b"],
|
||||
"XMM5": ["0x4ea4a8c17ebaf102", "0x3eea209bbd66277c"],
|
||||
"XMM6": ["0x40497b13404439b5", "0x3eea209b4ea4a8c1"],
|
||||
"XMM7": ["0x4040528b4040528b", "0x3eea209b4ea4a8c1"],
|
||||
"XMM8": ["0x403839b8403839b8", "0x3eea209b4ea4a8c1"],
|
||||
"XMM9": ["0x4056cde54056cde5", "0x403839b8403839b8"],
|
||||
"XMM10": ["0x4056b34aa10e0221", "0x4056cde54056cde5"],
|
||||
"XMM11": ["0x4052997f0ed3d85a", "0xa10e0221a10e0221"],
|
||||
"XMM12": ["0x40419d2240395a6b", "0xa10e0221a10e0221"],
|
||||
"XMM13": ["0x40177e2840568cc5", "0x40419d2240395a6b"],
|
||||
"XMM14": ["0x9f16b11c40408402", "0x40419d2240395a6b"],
|
||||
"XMM15": ["0x5feda661404d3159", "0x9f16b11c40408402"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
movaps xmm0, [rel .data + 16 * 0]
|
||||
movaps xmm1, [rel .data + 16 * 1]
|
||||
|
||||
movaps xmm2, [rel .data + 16 * 2]
|
||||
movaps xmm3, [rel .data + 16 * 3]
|
||||
|
||||
movaps xmm4, [rel .data + 16 * 4]
|
||||
movaps xmm5, [rel .data + 16 * 5]
|
||||
|
||||
movaps xmm6, [rel .data + 16 * 6]
|
||||
movaps xmm7, [rel .data + 16 * 7]
|
||||
|
||||
movaps xmm8, [rel .data + 16 * 8]
|
||||
movaps xmm9, [rel .data + 16 * 9]
|
||||
|
||||
movaps xmm10, [rel .data + 16 * 10]
|
||||
movaps xmm11, [rel .data + 16 * 11]
|
||||
|
||||
movaps xmm12, [rel .data + 16 * 12]
|
||||
movaps xmm13, [rel .data + 16 * 13]
|
||||
|
||||
movaps xmm14, [rel .data + 16 * 14]
|
||||
movaps xmm15, [rel .data + 16 * 15]
|
||||
|
||||
; Test inverted sources from shufps_optimization.asm
|
||||
shufps xmm1, xmm0, 01000100b
|
||||
shufps xmm0, [rel .data + 16 * 16], 0
|
||||
shufps xmm2, xmm1, 11101110b
|
||||
shufps xmm3, xmm2, 11100100b
|
||||
shufps xmm4, xmm3, 01001110b
|
||||
shufps xmm5, xmm4, 10001000b
|
||||
shufps xmm6, xmm5, 11011101b
|
||||
shufps xmm7, xmm6, 11100101b
|
||||
shufps xmm8, xmm7, 11101111b
|
||||
shufps xmm9, xmm8, 01001111b
|
||||
shufps xmm10, xmm9, 00000100b
|
||||
shufps xmm11, xmm10, 00001110b
|
||||
shufps xmm12, xmm11, 11100111b
|
||||
shufps xmm13, xmm12, 01000111b
|
||||
shufps xmm14, xmm13, 11100001b
|
||||
shufps xmm15, xmm14, 01000001b
|
||||
|
||||
hlt
|
||||
|
||||
align 16
|
||||
; 512bytes of random data
|
||||
.data:
|
||||
dq 83.0999,69.50512,41.02678,13.05881,5.35242,21.9932,9.67383,5.32372,29.02872,66.50151,19.30764,91.3633,40.45086,50.96153,32.64489,23.97574,90.64316,24.22547,98.9394,91.21715,90.80143,99.48407,64.97245,74.39838,35.22761,25.35321,5.8732,90.19956,33.03133,52.02952,58.38554,10.17531,47.84703,84.04831,90.02965,65.81329,96.27991,6.64479,25.58971,95.00694,88.1929,37.16964,49.52602,10.27223,77.70605,20.21439,9.8056,41.29389,15.4071,57.54286,9.61117,55.54302,52.90745,4.88086,72.52882,3.0201,56.55091,71.22749,61.84736,88.74295,47.72641,24.17404,33.70564,96.71303
|
||||
@@ -3232,7 +3232,7 @@
|
||||
"mov x0, x6",
|
||||
"mov x1, x20",
|
||||
"mov x2, x7",
|
||||
"ldr x3, [x28, #1912]",
|
||||
"ldr x3, [x28, #1920]",
|
||||
"str x30, [sp, #-16]!",
|
||||
"blr x3",
|
||||
"ldr x30, [sp], #16",
|
||||
@@ -3243,7 +3243,7 @@
|
||||
"mov x0, x6",
|
||||
"mov x1, x20",
|
||||
"mov x2, x7",
|
||||
"ldr x3, [x28, #1928]",
|
||||
"ldr x3, [x28, #1936]",
|
||||
"str x30, [sp, #-16]!",
|
||||
"blr x3",
|
||||
"ldr x30, [sp], #16",
|
||||
@@ -3307,7 +3307,7 @@
|
||||
"mov x0, x6",
|
||||
"mov x1, x20",
|
||||
"mov x2, x7",
|
||||
"ldr x3, [x28, #1920]",
|
||||
"ldr x3, [x28, #1928]",
|
||||
"str x30, [sp, #-16]!",
|
||||
"blr x3",
|
||||
"ldr x30, [sp], #16",
|
||||
@@ -3320,7 +3320,7 @@
|
||||
"mov x0, x6",
|
||||
"mov x1, x20",
|
||||
"mov x2, x7",
|
||||
"ldr x3, [x28, #1936]",
|
||||
"ldr x3, [x28, #1944]",
|
||||
"str x30, [sp, #-16]!",
|
||||
"blr x3",
|
||||
"ldr x30, [sp], #16",
|
||||
|
||||
@@ -4131,34 +4131,38 @@
|
||||
]
|
||||
},
|
||||
"shufps xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 8,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x0f 0xc6",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.s[0], v16.s[1]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v2.s[1], v16.s[0]",
|
||||
"mov v2.s[2], v17.s[0]",
|
||||
"mov v0.16b, v2.16b",
|
||||
"mov v0.s[3], v17.s[0]",
|
||||
"mov v16.16b, v0.16b"
|
||||
"ldr x0, [x28, #1640]",
|
||||
"ldr q2, [x0, #16]",
|
||||
"tbl v16.16b, {v16.16b, v17.16b}, v2.16b"
|
||||
]
|
||||
},
|
||||
"shufps xmm1, xmm0, 1": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x0f 0xc6",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr x0, [x28, #1640]",
|
||||
"ldr q2, [x0, #16]",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v1.16b, v16.16b",
|
||||
"tbl v17.16b, {v0.16b, v1.16b}, v2.16b"
|
||||
]
|
||||
},
|
||||
"shufps xmm0, [rax], 1": {
|
||||
"ExpectedInstructionCount": 9,
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x0f 0xc6",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x4]",
|
||||
"ldr x0, [x28, #1640]",
|
||||
"ldr q3, [x0, #16]",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.s[0], v16.s[1]",
|
||||
"mov v3.16b, v0.16b",
|
||||
"mov v3.s[1], v16.s[0]",
|
||||
"mov v3.s[2], v2.s[0]",
|
||||
"mov v0.16b, v3.16b",
|
||||
"mov v0.s[3], v2.s[0]",
|
||||
"mov v16.16b, v0.16b"
|
||||
"mov v1.16b, v2.16b",
|
||||
"tbl v16.16b, {v0.16b, v1.16b}, v3.16b"
|
||||
]
|
||||
},
|
||||
"shufps xmm0, [rax], 0xFF": {
|
||||
|
||||
@@ -3207,7 +3207,7 @@
|
||||
]
|
||||
},
|
||||
"vshufps xmm0, xmm1, xmm2, 01b": {
|
||||
"ExpectedInstructionCount": 12,
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b00 0xC6 128-bit"
|
||||
@@ -3215,14 +3215,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov z2.d, p7/m, z17.d",
|
||||
"mov z3.d, p7/m, z18.d",
|
||||
"mov v0.16b, v2.16b",
|
||||
"mov v0.s[0], v2.s[1]",
|
||||
"mov v4.16b, v0.16b",
|
||||
"mov v0.16b, v4.16b",
|
||||
"mov v0.s[1], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v2.s[2], v3.s[0]",
|
||||
"mov v2.s[3], v3.s[0]",
|
||||
"ldr x0, [x28, #1640]",
|
||||
"ldr q4, [x0, #16]",
|
||||
"tbl v2.16b, {v2.16b, v3.16b}, v4.16b",
|
||||
"mov v2.16b, v2.16b",
|
||||
"mov z16.d, p7/m, z2.d"
|
||||
]
|
||||
@@ -3274,7 +3269,7 @@
|
||||
]
|
||||
},
|
||||
"vshufps xmm0, xmm1, xmm2, 10b": {
|
||||
"ExpectedInstructionCount": 12,
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b00 0xC6 128-bit"
|
||||
@@ -3282,14 +3277,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov z2.d, p7/m, z17.d",
|
||||
"mov z3.d, p7/m, z18.d",
|
||||
"mov v0.16b, v2.16b",
|
||||
"mov v0.s[0], v2.s[2]",
|
||||
"mov v4.16b, v0.16b",
|
||||
"mov v0.16b, v4.16b",
|
||||
"mov v0.s[1], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v2.s[2], v3.s[0]",
|
||||
"mov v2.s[3], v3.s[0]",
|
||||
"ldr x0, [x28, #1640]",
|
||||
"ldr q4, [x0, #32]",
|
||||
"tbl v2.16b, {v2.16b, v3.16b}, v4.16b",
|
||||
"mov v2.16b, v2.16b",
|
||||
"mov z16.d, p7/m, z2.d"
|
||||
]
|
||||
@@ -3341,7 +3331,7 @@
|
||||
]
|
||||
},
|
||||
"vshufps xmm0, xmm1, xmm2, 11b": {
|
||||
"ExpectedInstructionCount": 12,
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b00 0xC6 128-bit"
|
||||
@@ -3349,14 +3339,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov z2.d, p7/m, z17.d",
|
||||
"mov z3.d, p7/m, z18.d",
|
||||
"mov v0.16b, v2.16b",
|
||||
"mov v0.s[0], v2.s[3]",
|
||||
"mov v4.16b, v0.16b",
|
||||
"mov v0.16b, v4.16b",
|
||||
"mov v0.s[1], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v2.s[2], v3.s[0]",
|
||||
"mov v2.s[3], v3.s[0]",
|
||||
"ldr x0, [x28, #1640]",
|
||||
"ldr q4, [x0, #48]",
|
||||
"tbl v2.16b, {v2.16b, v3.16b}, v4.16b",
|
||||
"mov v2.16b, v2.16b",
|
||||
"mov z16.d, p7/m, z2.d"
|
||||
]
|
||||
|
||||
Reference in new issue
Block a user