Merge pull request #3212 from Sonicadvance1/dpp_opt

OpcodeDispatcher: Optimize 128-bit DPPS and DPPD
This commit is contained in:
Mai authored and GitHub committed 2023-11-01 04:01:05 +01:00
commit 77d92872bc
20 files changed
+990 -117

No files matched your search

@@ -176,6 +176,54 @@ constexpr static auto SHUFPS_LUT {
}()
};
constexpr static auto DPPS_MASK {
[]() consteval {
struct LUTType {
uint32_t Val[4];
};
std::array<LUTType, 16> TotalLUT{};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto &LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1U;
}
return 0U;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
LUT.Val[2] = GetLUT(i, 2);
LUT.Val[3] = GetLUT(i, 3);
}
return TotalLUT;
}()
};
constexpr static auto DPPD_MASK {
[]() consteval {
struct LUTType {
uint64_t Val[2];
};
std::array<LUTType, 4> TotalLUT{};
for (size_t i = 0; i < TotalLUT.size(); ++i) {
auto &LUT = TotalLUT[i];
constexpr auto GetLUT = [](size_t i, size_t Index) {
if (i & (1U << Index)) {
return -1ULL;
}
return 0ULL;
};
LUT.Val[0] = GetLUT(i, 0);
LUT.Val[1] = GetLUT(i, 1);
}
return TotalLUT;
}()
};
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {
@@ -194,6 +242,8 @@ CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t I
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] = reinterpret_cast<uint64_t>(DPPS_MASK.data());
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] = reinterpret_cast<uint64_t>(DPPD_MASK.data());
#ifndef FEX_DISABLE_TELEMETRY
// Fill in telemetry values
@@ -1269,6 +1269,41 @@ DEF_OP(VAddP) {
}
}
DEF_OP(VFAddV) {
const auto Op = IROp->C<IR::IROp_VAddV>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto ElementSize = Op->Header.ElementSize;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE || OpSize == Core::CPUState::XMM_AVX_REG_SIZE, "Only AVX and SSE size supported");
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit : ARMEmitter::SubRegSize::i64Bit);
if (HostSupportsSVE256 && Is256Bit) {
const auto Pred = PRED_TMP_32B.Merging();
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
}
if (HostSupportsSVE128) {
const auto Pred = PRED_TMP_16B.Merging();
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
} else {
// ASIMD doesn't support faddv, need to use multiple faddp to match behaviour.
if (ElementSize == 4) {
faddp(SubRegSize.Vector, Dst.Q(), Vector.Q(), Vector.Q());
faddp(SubRegSize.Scalar, Dst, Vector);
}
else {
faddp(SubRegSize.Scalar, Dst, Vector);
}
}
}
DEF_OP(VAddV) {
const auto Op = IROp->C<IR::IROp_VAddV>();
const auto OpSize = IROp->Size;
@@ -3750,6 +3785,7 @@ DEF_OP(VDupElement) {
const auto Index = Op->Index;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE;
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
@@ -3764,7 +3800,12 @@ DEF_OP(VDupElement) {
if (HostSupportsSVE256 && Is256Bit) {
dup(SubRegSize, Dst.Z(), Vector.Z(), Index);
} else {
dup(SubRegSize, Dst.Q(), Vector.Q(), Index);
if (Is128Bit) {
dup(SubRegSize, Dst.Q(), Vector.Q(), Index);
}
else {
dup(SubRegSize, Dst.D(), Vector.D(), Index);
}
}
}
@@ -979,6 +979,10 @@ private:
const X86Tables::DecodedOperand& Src2,
const X86Tables::DecodedOperand& Imm, size_t ElementSize);
OrderedNode* VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
const X86Tables::DecodedOperand& Src2,
const X86Tables::DecodedOperand& Imm);
OrderedNode* ExtendVectorElementsImpl(OpcodeArgs, size_t ElementSize,
size_t DstElementSize, bool Signed);
@@ -4670,6 +4670,205 @@ OrderedNode* OpDispatchBuilder::DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOp
const X86Tables::DecodedOperand& Imm, size_t ElementSize) {
LOGMAN_THROW_A_FMT(Imm.IsLiteral(), "Imm needs to be literal here");
const uint8_t Mask = Imm.Data.Literal.Value;
const auto SizeMask = [ElementSize]() {
if (ElementSize == 4) {
return 0b1111;
}
return 0b11;
}();
const uint8_t SrcMask = (Mask >> 4) & SizeMask;
const uint8_t DstMask = Mask & SizeMask;
const auto NamedIndexMask = [ElementSize]() {
if (ElementSize == 4) {
return FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK;
}
return FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK;
}();
const auto DstSize = GetDstSize(Op);
OrderedNode *ZeroVec = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
if (SrcMask == 0 || DstMask == 0) {
// What are you even doing here? Go away.
return ZeroVec;
}
OrderedNode *Src1V = LoadSource(FPRClass, Op, Src1, Op->Flags);
OrderedNode *Src2V = LoadSource(FPRClass, Op, Src2, Op->Flags);
// First step is to do an FMUL
OrderedNode *Temp = _VFMul(DstSize, ElementSize, Src1V, Src2V);
// Now mask results based on IndexMask.
if (SrcMask != SizeMask) {
auto InputMask = LoadAndCacheIndexedNamedVectorConstant(DstSize, NamedIndexMask, SrcMask * 16);
Temp = _VAnd(DstSize, ElementSize, Temp, InputMask);
}
// Now due a float reduction
Temp = _VFAddV(DstSize, ElementSize, Temp);
// Now using the destination mask we choose where the result ends up
// It can duplicate and zero results
if (ElementSize == 8) {
switch (DstMask) {
case 0b01:
// Dest[63:0] = Result
// Dest[127:64] = Zero
return _VZip(DstSize, ElementSize, Temp, ZeroVec);
case 0b10:
// Dest[63:0] = Zero
// Dest[127:64] = Result
return _VZip(DstSize, ElementSize, ZeroVec, Temp);
case 0b11:
// Broadcast
// Dest[63:0] = Result
// Dest[127:64] = Result
return _VDupElement(DstSize, ElementSize, Temp, 0);
case 0:
default:
LOGMAN_MSG_A_FMT("Unsupported");
}
}
else {
auto BadPath = [&]() {
OrderedNode *Result = ZeroVec;
for (size_t i = 0; i < (DstSize / ElementSize); ++i) {
const auto Bit = 1U << (i % 4);
if ((DstMask & Bit) != 0) {
Result = _VInsElement(DstSize, ElementSize, i, 0, Result, Temp);
}
}
return Result;
};
switch (DstMask) {
case 0b0001:
// Dest[31:0] = Result
// Dest[63:32] = Zero
// Dest[95:64] = Zero
// Dest[127:96] = Zero
return _VZip(DstSize, ElementSize, Temp, ZeroVec);
case 0b0010:
// Dest[31:0] = Zero
// Dest[63:32] = Result
// Dest[95:64] = Zero
// Dest[127:96] = Zero
return _VZip(DstSize / 2, ElementSize, ZeroVec, Temp);
case 0b0011:
// Dest[31:0] = Result
// Dest[63:32] = Result
// Dest[95:64] = Zero
// Dest[127:96] = Zero
return _VDupElement(DstSize / 2, ElementSize, Temp, 0);
case 0b0100:
// Dest[31:0] = Zero
// Dest[63:32] = Zero
// Dest[95:64] = Result
// Dest[127:96] = Zero
return _VZip(DstSize, 8, ZeroVec, Temp);
case 0b0101:
// Dest[31:0] = Result
// Dest[63:32] = Zero
// Dest[95:64] = Result
// Dest[127:96] = Zero
return _VZip(DstSize, 8, Temp, Temp);
case 0b0110:
// Dest[31:0] = Zero
// Dest[63:32] = Result
// Dest[95:64] = Result
// Dest[127:96] = Zero
return BadPath();
case 0b0111:
// Dest[31:0] = Result
// Dest[63:32] = Result
// Dest[95:64] = Result
// Dest[127:96] = Zero
Temp = _VDupElement(DstSize, ElementSize, Temp, 0);
return _VInsElement(DstSize, ElementSize, 3, 0, Temp, ZeroVec);
case 0b1000:
// Dest[31:0] = Zero
// Dest[63:32] = Zero
// Dest[95:64] = Zero
// Dest[127:96] = Result
return _VExtr(DstSize, 1, Temp, ZeroVec, 4);
case 0b1001:
// Dest[31:0] = Result
// Dest[63:32] = Zero
// Dest[95:64] = Zero
// Dest[127:96] = Result
return BadPath();
case 0b1010:
// Dest[31:0] = Zero
// Dest[63:32] = Result
// Dest[95:64] = Zero
// Dest[127:96] = Result
Temp = _VDupElement(DstSize, ElementSize, Temp, 0);
return _VZip(DstSize, 4, ZeroVec, Temp);
case 0b1011:
// Dest[31:0] = Result
// Dest[63:32] = Result
// Dest[95:64] = Zero
// Dest[127:96] = Result
Temp = _VDupElement(DstSize, ElementSize, Temp, 0);
return _VInsElement(DstSize, ElementSize, 2, 0, Temp, ZeroVec);
case 0b1100:
// Dest[31:0] = Zero
// Dest[63:32] = Zero
// Dest[95:64] = Result
// Dest[127:96] = Result
Temp = _VDupElement(DstSize, ElementSize, Temp, 0);
return _VZip(DstSize, 8, ZeroVec, Temp);
case 0b1101:
// Dest[31:0] = Result
// Dest[63:32] = Zero
// Dest[95:64] = Result
// Dest[127:96] = Result
Temp = _VDupElement(DstSize, ElementSize, Temp, 0);
return _VInsElement(DstSize, ElementSize, 1, 0, Temp, ZeroVec);
case 0b1110:
// Dest[31:0] = Zero
// Dest[63:32] = Result
// Dest[95:64] = Result
// Dest[127:96] = Result
Temp = _VDupElement(DstSize, ElementSize, Temp, 0);
return _VInsElement(DstSize, ElementSize, 0, 0, Temp, ZeroVec);
case 0b1111:
// Broadcast
// Dest[31:0] = Result
// Dest[63:32] = Zero
// Dest[95:64] = Zero
// Dest[127:96] = Zero
return _VDupElement(DstSize, ElementSize, Temp, 0);
case 0:
default:
LOGMAN_MSG_A_FMT("Unsupported");
}
}
FEX_UNREACHABLE;
}
template<size_t ElementSize>
void OpDispatchBuilder::DPPOp(OpcodeArgs) {
OrderedNode *Result = DPPOpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1], ElementSize);
StoreResult(FPRClass, Op, Result, -1);
}
template
void OpDispatchBuilder::DPPOp<4>(OpcodeArgs);
template
void OpDispatchBuilder::DPPOp<8>(OpcodeArgs);
OrderedNode* OpDispatchBuilder::VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
const X86Tables::DecodedOperand& Src2,
const X86Tables::DecodedOperand& Imm) {
LOGMAN_THROW_A_FMT(Imm.IsLiteral(), "Imm needs to be literal here");
constexpr size_t ElementSize = 4;
const uint8_t Mask = Imm.Data.Literal.Value;
const uint8_t SrcMask = Mask >> 4;
const uint8_t DstMask = Mask & 0xF;
@@ -4716,20 +4915,18 @@ OrderedNode* OpDispatchBuilder::DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOp
return Result;
}
template<size_t ElementSize>
void OpDispatchBuilder::DPPOp(OpcodeArgs) {
OrderedNode *Result = DPPOpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1], ElementSize);
StoreResult(FPRClass, Op, Result, -1);
}
template
void OpDispatchBuilder::DPPOp<4>(OpcodeArgs);
template
void OpDispatchBuilder::DPPOp<8>(OpcodeArgs);
template <size_t ElementSize>
void OpDispatchBuilder::VDPPOp(OpcodeArgs) {
OrderedNode *Result = DPPOpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2], ElementSize);
const auto DstSize = GetDstSize(Op);
OrderedNode *Result{};
if (ElementSize == 4 && DstSize == Core::CPUState::XMM_AVX_REG_SIZE) {
// 256-bit DPPS isn't handled by the 128-bit solution.
Result = VDPPSOpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2]);
}
else {
Result = DPPOpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2], ElementSize);
}
// We don't need to emit a _VMov to clear the upper lane, since DPPOpImpl uses a zero vector
// to construct the results, so the upper lane will always be cleared for the 128-bit version.
+7
View File
@@ -1825,6 +1825,13 @@
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFAddV u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
"Desc": ["Does a horizontal float vector add of elements across the source vector",
"Result is a zero extended scalar"
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFSub u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
+2
View File
@@ -541,6 +541,8 @@ enum IndexNamedVectorConstant : uint8_t {
INDEXED_NAMED_VECTOR_PSHUFHW,
INDEXED_NAMED_VECTOR_PSHUFD,
INDEXED_NAMED_VECTOR_SHUFPS,
INDEXED_NAMED_VECTOR_DPPS_MASK,
INDEXED_NAMED_VECTOR_DPPD_MASK,
INDEXED_NAMED_VECTOR_MAX,
};
+63
View File
@@ -0,0 +1,63 @@
%ifdef CONFIG
{
"RegData": {
"XMM0": ["0", "0"],
"XMM1": ["0x00000000c197f874", "0"],
"XMM2": ["0xff80000000000000", "0x0000000000000000"],
"XMM3": ["0x7a147e317a147e31", "0x0000000000000000"],
"XMM4": ["0x0000000000000000", "0x000000006cd0f887"],
"XMM5": ["0x000000007f800000", "0x000000007f800000"],
"XMM6": ["0xff80000000000000", "0x00000000ff800000"],
"XMM7": ["0xfc944256fc944256", "0x00000000fc944256"],
"XMM8": ["0x0000000000000000", "0xc3ac072e00000000"],
"XMM9": ["0x000000005c5c09a3", "0x5c5c09a300000000"],
"XMM10": ["0xdc34227c00000000", "0xdc34227c00000000"],
"XMM11": ["0xda1627d2da1627d2", "0xda1627d200000000"],
"XMM12": ["0x0000000000000000", "0x7f8000007f800000"],
"XMM13": ["0x000000005f30e9d3", "0x5f30e9d35f30e9d3"],
"XMM14": ["0xda3f264a00000000", "0xda3f264ada3f264a"],
"XMM15": ["0x7f8000007f800000", "0x7f8000007f800000"]
}
}
%endif
movaps xmm0, [rel .data + 16 * 0]
movaps xmm1, [rel .data + 16 * 1]
movaps xmm2, [rel .data + 16 * 2]
movaps xmm3, [rel .data + 16 * 3]
movaps xmm4, [rel .data + 16 * 4]
movaps xmm5, [rel .data + 16 * 5]
movaps xmm6, [rel .data + 16 * 6]
movaps xmm7, [rel .data + 16 * 7]
movaps xmm8, [rel .data + 16 * 8]
movaps xmm9, [rel .data + 16 * 9]
movaps xmm10, [rel .data + 16 * 10]
movaps xmm11, [rel .data + 16 * 11]
movaps xmm12, [rel .data + 16 * 12]
movaps xmm13, [rel .data + 16 * 13]
movaps xmm14, [rel .data + 16 * 14]
movaps xmm15, [rel .data + 16 * 15]
; Full source mask but different broadcast tests
dpps xmm0, [rel .data + 16 * 16], 1111_0000b
dpps xmm1, [rel .data + 16 * 16], 1111_0001b
dpps xmm2, [rel .data + 16 * 16], 1111_0010b
dpps xmm3, [rel .data + 16 * 16], 1111_0011b
dpps xmm4, [rel .data + 16 * 16], 1111_0100b
dpps xmm5, [rel .data + 16 * 16], 1111_0101b
dpps xmm6, [rel .data + 16 * 16], 1111_0110b
dpps xmm7, [rel .data + 16 * 16], 1111_0111b
dpps xmm8, [rel .data + 16 * 16], 1111_1000b
dpps xmm9, [rel .data + 16 * 16], 1111_1001b
dpps xmm10, [rel .data + 16 * 16], 1111_1010b
dpps xmm11, [rel .data + 16 * 16], 1111_1011b
dpps xmm12, [rel .data + 16 * 16], 1111_1100b
dpps xmm13, [rel .data + 16 * 16], 1111_1101b
dpps xmm14, [rel .data + 16 * 16], 1111_1110b
dpps xmm15, [rel .data + 16 * 16], 1111_1111b
hlt
align 16
; 512bytes of random data
.data:
dq 83.0999,69.50512,41.02678,13.05881,5.35242,21.9932,9.67383,5.32372,29.02872,66.50151,19.30764,91.3633,40.45086,50.96153,32.64489,23.97574,90.64316,24.22547,98.9394,91.21715,90.80143,99.48407,64.97245,74.39838,35.22761,25.35321,5.8732,90.19956,33.03133,52.02952,58.38554,10.17531,47.84703,84.04831,90.02965,65.81329,96.27991,6.64479,25.58971,95.00694,88.1929,37.16964,49.52602,10.27223,77.70605,20.21439,9.8056,41.29389,15.4071,57.54286,9.61117,55.54302,52.90745,4.88086,72.52882,3.0201,56.55091,71.22749,61.84736,88.74295,47.72641,24.17404,33.70564,96.71303
+63
View File
@@ -0,0 +1,63 @@
%ifdef CONFIG
{
"RegData": {
"XMM0": ["0", "0"],
"XMM1": ["0x40a7e92935462e9e", "0"],
"XMM2": ["0", "0x40a0712d6903205c"],
"XMM3": ["0x408c728276ca7656", "0x408c728276ca7656"],
"XMM4": ["0", "0"],
"XMM5": ["0x40c0cd5f41a95ce2", "0"],
"XMM6": ["0", "0x40b84aaf198a4022"],
"XMM7": ["0x40abf229b504629d", "0x40abf229b504629d"],
"XMM8": ["0", "0"],
"XMM9": ["0x40c8384d475e602a", "0"],
"XMM10": ["0", "0x40c8d105fa49a70e"],
"XMM11": ["0x40c248e5ffd69239", "0x40c248e5ffd69239"],
"XMM12": ["0", "0"],
"XMM13": ["0x40beb622c0fe35c7", "0"],
"XMM14": ["0", "0x40b74171bb41b9ba"],
"XMM15": ["0x40ac8195a7735fbe", "0x40ac8195a7735fbe"]
}
}
%endif
movaps xmm0, [rel .data + 16 * 0]
movaps xmm1, [rel .data + 16 * 1]
movaps xmm2, [rel .data + 16 * 2]
movaps xmm3, [rel .data + 16 * 3]
movaps xmm4, [rel .data + 16 * 4]
movaps xmm5, [rel .data + 16 * 5]
movaps xmm6, [rel .data + 16 * 6]
movaps xmm7, [rel .data + 16 * 7]
movaps xmm8, [rel .data + 16 * 8]
movaps xmm9, [rel .data + 16 * 9]
movaps xmm10, [rel .data + 16 * 10]
movaps xmm11, [rel .data + 16 * 11]
movaps xmm12, [rel .data + 16 * 12]
movaps xmm13, [rel .data + 16 * 13]
movaps xmm14, [rel .data + 16 * 14]
movaps xmm15, [rel .data + 16 * 15]
; Full source mask but different broadcast tests
dppd xmm0, [rel .data + 16 * 16], 1111_0000b
dppd xmm1, [rel .data + 16 * 16], 1111_0001b
dppd xmm2, [rel .data + 16 * 16], 1111_0010b
dppd xmm3, [rel .data + 16 * 16], 1111_0011b
dppd xmm4, [rel .data + 16 * 16], 1111_0100b
dppd xmm5, [rel .data + 16 * 16], 1111_0101b
dppd xmm6, [rel .data + 16 * 16], 1111_0110b
dppd xmm7, [rel .data + 16 * 16], 1111_0111b
dppd xmm8, [rel .data + 16 * 16], 1111_1000b
dppd xmm9, [rel .data + 16 * 16], 1111_1001b
dppd xmm10, [rel .data + 16 * 16], 1111_1010b
dppd xmm11, [rel .data + 16 * 16], 1111_1011b
dppd xmm12, [rel .data + 16 * 16], 1111_1100b
dppd xmm13, [rel .data + 16 * 16], 1111_1101b
dppd xmm14, [rel .data + 16 * 16], 1111_1110b
dppd xmm15, [rel .data + 16 * 16], 1111_1111b
hlt
align 16
; 512bytes of random data
.data:
dq 83.0999,69.50512,41.02678,13.05881,5.35242,21.9932,9.67383,5.32372,29.02872,66.50151,19.30764,91.3633,40.45086,50.96153,32.64489,23.97574,90.64316,24.22547,98.9394,91.21715,90.80143,99.48407,64.97245,74.39838,35.22761,25.35321,5.8732,90.19956,33.03133,52.02952,58.38554,10.17531,47.84703,84.04831,90.02965,65.81329,96.27991,6.64479,25.58971,95.00694,88.1929,37.16964,49.52602,10.27223,77.70605,20.21439,9.8056,41.29389,15.4071,57.54286,9.61117,55.54302,52.90745,4.88086,72.52882,3.0201,56.55091,71.22749,61.84736,88.74295,47.72641,24.17404,33.70564,96.71303
+3 -3
View File
@@ -113,7 +113,7 @@
"ExpectedArm64ASM": [
"ldr d2, [x28, #752]",
"ldr d3, [x28, #768]",
"dup v4.4s, v2.s[1]",
"dup v4.2s, v2.s[1]",
"fsub s2, s2, s4",
"faddp v3.4s, v3.4s, v3.4s",
"mov v2.s[1], v3.s[0]",
@@ -153,7 +153,7 @@
"ldr d2, [x28, #768]",
"fmov s0, #0x70 (1.0000)",
"fdiv s2, s0, s2",
"dup v2.4s, v2.s[0]",
"dup v2.2s, v2.s[0]",
"str d2, [x28, #752]"
]
},
@@ -168,7 +168,7 @@
"fmov s0, #0x70 (1.0000)",
"fsqrt s1, s2",
"fdiv s2, s0, s1",
"dup v2.4s, v2.s[0]",
"dup v2.2s, v2.s[0]",
"str d2, [x28, #752]"
]
},
+1 -1
View File
@@ -682,7 +682,7 @@
"0x66 0x0f 0x38 0x41"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #1888]",
"ldr q2, [x28, #1904]",
"zip1 v3.8h, v2.8h, v17.8h",
"zip2 v2.8h, v2.8h, v17.8h",
"umin v2.4s, v3.4s, v2.4s",
+222 -41
View File
@@ -951,25 +951,13 @@
]
},
"dpps xmm0, xmm1, 00001111b": {
"ExpectedInstructionCount": 13,
"Optimal": "No",
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"mov v3.s[0], v2.s[0]",
"mov v3.s[1], v2.s[0]",
"mov v3.s[2], v2.s[0]",
"mov v3.s[3], v2.s[0]",
"faddp v3.4s, v3.4s, v2.4s",
"faddp v3.4s, v3.4s, v2.4s",
"mov v2.s[0], v3.s[0]",
"mov v2.s[1], v3.s[0]",
"mov v2.s[2], v3.s[0]",
"mov v16.16b, v2.16b",
"mov v16.s[3], v3.s[0]"
"movi v16.2d, #0x0"
]
},
"dpps xmm0, xmm1, 11110000b": {
@@ -982,8 +970,76 @@
"movi v16.2d, #0x0"
]
},
"dpps xmm0, xmm1, 11111111b": {
"ExpectedInstructionCount": 9,
"dpps xmm0, xmm1, 11110001b": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"zip1 v16.4s, v3.4s, v2.4s"
]
},
"dpps xmm0, xmm1, 11110010b": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"zip1 v16.2s, v2.2s, v3.2s"
]
},
"dpps xmm0, xmm1, 11110011b": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"fmul v2.4s, v16.4s, v17.4s",
"faddp v2.4s, v2.4s, v2.4s",
"faddp s2, v2.2s",
"dup v16.2s, v2.s[0]"
]
},
"dpps xmm0, xmm1, 11110100b": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"zip1 v16.2d, v2.2d, v3.2d"
]
},
"dpps xmm0, xmm1, 11110101b": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"fmul v2.4s, v16.4s, v17.4s",
"faddp v2.4s, v2.4s, v2.4s",
"faddp s2, v2.2s",
"zip1 v16.2d, v2.2d, v2.2d"
]
},
"dpps xmm0, xmm1, 11110110b": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
@@ -991,15 +1047,150 @@
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v2.4s",
"faddp v3.4s, v3.4s, v2.4s",
"mov v2.s[0], v3.s[0]",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"mov v2.s[1], v3.s[0]",
"mov v2.s[2], v3.s[0]",
"mov v16.16b, v2.16b",
"mov v16.s[2], v3.s[0]"
]
},
"dpps xmm0, xmm1, 11110111b": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"dup v3.4s, v3.s[0]",
"mov v16.16b, v3.16b",
"mov v16.s[3], v2.s[0]"
]
},
"dpps xmm0, xmm1, 11111000b": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"ext v16.16b, v2.16b, v3.16b, #4"
]
},
"dpps xmm0, xmm1, 11111001b": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"mov v2.s[0], v3.s[0]",
"mov v16.16b, v2.16b",
"mov v16.s[3], v3.s[0]"
]
},
"dpps xmm0, xmm1, 11111010b": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"dup v3.4s, v3.s[0]",
"zip1 v16.4s, v2.4s, v3.4s"
]
},
"dpps xmm0, xmm1, 11111011b": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"dup v3.4s, v3.s[0]",
"mov v16.16b, v3.16b",
"mov v16.s[2], v2.s[0]"
]
},
"dpps xmm0, xmm1, 11111100b": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"dup v3.4s, v3.s[0]",
"zip1 v16.2d, v2.2d, v3.2d"
]
},
"dpps xmm0, xmm1, 11111101b": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"dup v3.4s, v3.s[0]",
"mov v16.16b, v3.16b",
"mov v16.s[1], v2.s[0]"
]
},
"dpps xmm0, xmm1, 11111110b": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddp v3.4s, v3.4s, v3.4s",
"faddp s3, v3.2s",
"dup v3.4s, v3.s[0]",
"mov v16.16b, v3.16b",
"mov v16.s[0], v2.s[0]"
]
},
"dpps xmm0, xmm1, 11111111b": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"fmul v2.4s, v16.4s, v17.4s",
"faddp v2.4s, v2.4s, v2.4s",
"faddp s2, v2.2s",
"dup v16.4s, v2.s[0]"
]
},
"dppd xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
@@ -1011,20 +1202,13 @@
]
},
"dppd xmm0, xmm1, 00001111b": {
"ExpectedInstructionCount": 8,
"Optimal": "No",
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x41"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.2d, v16.2d, v17.2d",
"mov v3.d[0], v2.d[0]",
"mov v3.d[1], v2.d[0]",
"faddp v3.2d, v3.2d, v2.2d",
"mov v2.d[0], v3.d[0]",
"mov v16.16b, v2.16b",
"mov v16.d[1], v3.d[0]"
"movi v16.2d, #0x0"
]
},
"dppd xmm0, xmm1, 11110000b": {
@@ -1038,18 +1222,15 @@
]
},
"dppd xmm0, xmm1, 11111111b": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x41"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.2d, v16.2d, v17.2d",
"faddp v3.2d, v3.2d, v2.2d",
"mov v2.d[0], v3.d[0]",
"mov v16.16b, v2.16b",
"mov v16.d[1], v3.d[0]"
"fmul v2.2d, v16.2d, v17.2d",
"faddp d2, v2.2d",
"dup v16.2d, v2.d[0]"
]
},
"mpsadbw xmm0, xmm1, 000b": {
@@ -1557,7 +1738,7 @@
"0x66 0x0f 0x3a 0xdf"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #2000]",
"ldr q2, [x28, #2016]",
"movi v3.2d, #0x0",
"mov v16.16b, v17.16b",
"unimplemented (Unimplemented)",
@@ -1571,7 +1752,7 @@
"0x66 0x0f 0x3a 0xdf"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #2000]",
"ldr q2, [x28, #2016]",
"movi v3.2d, #0x0",
"mov v16.16b, v17.16b",
"unimplemented (Unimplemented)",
@@ -0,0 +1,292 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"SVE128"
],
"DisabledHostFeatures": [
"SVE256",
"AFP"
]
},
"Instructions": {
"dpps xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v16.2d, #0x0"
]
},
"dpps xmm0, xmm1, 00001111b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v16.2d, #0x0"
]
},
"dpps xmm0, xmm1, 11110000b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v16.2d, #0x0"
]
},
"dpps xmm0, xmm1, 11110001b": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"zip1 v16.4s, v3.4s, v2.4s"
]
},
"dpps xmm0, xmm1, 11110010b": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"zip1 v16.2s, v2.2s, v3.2s"
]
},
"dpps xmm0, xmm1, 11110011b": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"fmul v2.4s, v16.4s, v17.4s",
"faddv s2, p6, z2.s",
"dup v16.2s, v2.s[0]"
]
},
"dpps xmm0, xmm1, 11110100b": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"zip1 v16.2d, v2.2d, v3.2d"
]
},
"dpps xmm0, xmm1, 11110101b": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"fmul v2.4s, v16.4s, v17.4s",
"faddv s2, p6, z2.s",
"zip1 v16.2d, v2.2d, v2.2d"
]
},
"dpps xmm0, xmm1, 11110110b": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"mov v2.s[1], v3.s[0]",
"mov v16.16b, v2.16b",
"mov v16.s[2], v3.s[0]"
]
},
"dpps xmm0, xmm1, 11110111b": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"dup v3.4s, v3.s[0]",
"mov v16.16b, v3.16b",
"mov v16.s[3], v2.s[0]"
]
},
"dpps xmm0, xmm1, 11111000b": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"ext v16.16b, v2.16b, v3.16b, #4"
]
},
"dpps xmm0, xmm1, 11111001b": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"mov v2.s[0], v3.s[0]",
"mov v16.16b, v2.16b",
"mov v16.s[3], v3.s[0]"
]
},
"dpps xmm0, xmm1, 11111010b": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"dup v3.4s, v3.s[0]",
"zip1 v16.4s, v2.4s, v3.4s"
]
},
"dpps xmm0, xmm1, 11111011b": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"dup v3.4s, v3.s[0]",
"mov v16.16b, v3.16b",
"mov v16.s[2], v2.s[0]"
]
},
"dpps xmm0, xmm1, 11111100b": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"dup v3.4s, v3.s[0]",
"zip1 v16.2d, v2.2d, v3.2d"
]
},
"dpps xmm0, xmm1, 11111101b": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"dup v3.4s, v3.s[0]",
"mov v16.16b, v3.16b",
"mov v16.s[1], v2.s[0]"
]
},
"dpps xmm0, xmm1, 11111110b": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v16.4s, v17.4s",
"faddv s3, p6, z3.s",
"dup v3.4s, v3.s[0]",
"mov v16.16b, v3.16b",
"mov v16.s[0], v2.s[0]"
]
},
"dpps xmm0, xmm1, 11111111b": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x40"
],
"ExpectedArm64ASM": [
"fmul v2.4s, v16.4s, v17.4s",
"faddv s2, p6, z2.s",
"dup v16.4s, v2.s[0]"
]
},
"dppd xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x41"
],
"ExpectedArm64ASM": [
"movi v16.2d, #0x0"
]
},
"dppd xmm0, xmm1, 00001111b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x41"
],
"ExpectedArm64ASM": [
"movi v16.2d, #0x0"
]
},
"dppd xmm0, xmm1, 11110000b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x41"
],
"ExpectedArm64ASM": [
"movi v16.2d, #0x0"
]
},
"dppd xmm0, xmm1, 11111111b": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x41"
],
"ExpectedArm64ASM": [
"fmul v2.2d, v16.2d, v17.2d",
"faddv d2, p6, z2.d",
"dup v16.2d, v2.d[0]"
]
}
}
}
@@ -3790,7 +3790,7 @@
"mov x0, x6",
"mov x1, x20",
"mov x2, x7",
"ldr x3, [x28, #2048]",
"ldr x3, [x28, #2064]",
"str x30, [sp, #-16]!",
"blr x3",
"ldr x30, [sp], #16",
@@ -3801,7 +3801,7 @@
"mov x0, x6",
"mov x1, x20",
"mov x2, x7",
"ldr x3, [x28, #2064]",
"ldr x3, [x28, #2080]",
"str x30, [sp, #-16]!",
"blr x3",
"ldr x30, [sp], #16",
@@ -3865,7 +3865,7 @@
"mov x0, x6",
"mov x1, x20",
"mov x2, x7",
"ldr x3, [x28, #2056]",
"ldr x3, [x28, #2072]",
"str x30, [sp, #-16]!",
"blr x3",
"ldr x30, [sp], #16",
@@ -3878,7 +3878,7 @@
"mov x0, x6",
"mov x1, x20",
"mov x2, x7",
"ldr x3, [x28, #2072]",
"ldr x3, [x28, #2088]",
"str x30, [sp, #-16]!",
"blr x3",
"ldr x30, [sp], #16",
+2 -2
View File
@@ -43,7 +43,7 @@
"ExpectedArm64ASM": [
"ldr d2, [x28, #768]",
"frecpe s2, s2",
"dup v2.4s, v2.s[0]",
"dup v2.2s, v2.s[0]",
"str d2, [x28, #752]"
]
},
@@ -56,7 +56,7 @@
"ExpectedArm64ASM": [
"ldr d2, [x28, #768]",
"frsqrte s2, s2",
"dup v2.4s, v2.s[0]",
"dup v2.2s, v2.s[0]",
"str d2, [x28, #752]"
]
}
+6 -6
View File
@@ -868,7 +868,7 @@
"Comment": "0x0f 0x50",
"ExpectedArm64ASM": [
"ushr v2.4s, v16.4s, #31",
"ldr q3, [x28, #1984]",
"ldr q3, [x28, #2000]",
"ushl v2.4s, v2.4s, v3.4s",
"addv s2, v2.4s",
"mov w4, v2.s[0]"
@@ -880,7 +880,7 @@
"Comment": "0x0f 0x50",
"ExpectedArm64ASM": [
"ushr v2.4s, v16.4s, #31",
"ldr q3, [x28, #1984]",
"ldr q3, [x28, #2000]",
"ushl v2.4s, v2.4s, v3.4s",
"addv s2, v2.4s",
"mov w4, v2.s[0]"
@@ -1287,7 +1287,7 @@
"Comment": "0x0f 0x70",
"ExpectedArm64ASM": [
"ldr d2, [x28, #768]",
"dup v2.8h, v2.h[0]",
"dup v2.4h, v2.h[0]",
"str d2, [x28, #752]"
]
},
@@ -1297,7 +1297,7 @@
"Comment": "0x0f 0x70",
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"dup v2.8h, v2.h[0]",
"dup v2.4h, v2.h[0]",
"str d2, [x28, #752]"
]
},
@@ -1331,7 +1331,7 @@
"Comment": "0x0f 0x70",
"ExpectedArm64ASM": [
"ldr d2, [x28, #768]",
"dup v2.8h, v2.h[3]",
"dup v2.4h, v2.h[3]",
"str d2, [x28, #752]"
]
},
@@ -1341,7 +1341,7 @@
"Comment": "0x0f 0x70",
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"dup v2.8h, v2.h[3]",
"dup v2.4h, v2.h[3]",
"str d2, [x28, #752]"
]
},
@@ -1164,7 +1164,7 @@
"Optimal": "Yes",
"Comment": "0x66 0x0f 0xd0",
"ExpectedArm64ASM": [
"ldr q2, [x28, #1952]",
"ldr q2, [x28, #1968]",
"eor v2.16b, v17.16b, v2.16b",
"fadd v16.2d, v16.2d, v2.2d"
]
@@ -495,7 +495,7 @@
"Optimal": "Yes",
"Comment": "0xf2 0x0f 0xd0",
"ExpectedArm64ASM": [
"ldr q2, [x28, #1920]",
"ldr q2, [x28, #1936]",
"eor v2.16b, v17.16b, v2.16b",
"fadd v16.4s, v16.4s, v2.4s"
]
+2 -2
View File
@@ -4668,7 +4668,7 @@
"Map 1 0b01 0xd0 128-bit"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #1952]",
"ldr q2, [x28, #1968]",
"eor v2.16b, v18.16b, v2.16b",
"fadd v16.2d, v17.2d, v2.2d"
]
@@ -4693,7 +4693,7 @@
"Map 1 0b11 0xd0 128-bit"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #1920]",
"ldr q2, [x28, #1936]",
"eor v2.16b, v18.16b, v2.16b",
"fadd v16.4s, v17.4s, v2.4s"
]
+1 -1
View File
@@ -1742,7 +1742,7 @@
"Map 2 0b01 0x41 256-bit"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #1888]",
"ldr q2, [x28, #1904]",
"zip1 v3.8h, v2.8h, v17.8h",
"zip2 v2.8h, v2.8h, v17.8h",
"umin v2.4s, v3.4s, v2.4s",
+15 -42
View File
@@ -3520,25 +3520,13 @@
]
},
"vdpps xmm0, xmm1, xmm2, 00001111b": {
"ExpectedInstructionCount": 13,
"ExpectedInstructionCount": 1,
"Optimal": "No",
"Comment": [
"Map 3 0b01 0x40 128-bit"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v17.4s, v18.4s",
"mov v3.s[0], v2.s[0]",
"mov v3.s[1], v2.s[0]",
"mov v3.s[2], v2.s[0]",
"mov v3.s[3], v2.s[0]",
"faddp v3.4s, v3.4s, v2.4s",
"faddp v3.4s, v3.4s, v2.4s",
"mov v2.s[0], v3.s[0]",
"mov v2.s[1], v3.s[0]",
"mov v2.s[2], v3.s[0]",
"mov v16.16b, v2.16b",
"mov v16.s[3], v3.s[0]"
"movi v16.2d, #0x0"
]
},
"vdpps xmm0, xmm1, xmm2, 11110000b": {
@@ -3552,21 +3540,16 @@
]
},
"vdpps xmm0, xmm1, xmm2, 11111111b": {
"ExpectedInstructionCount": 9,
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": [
"Map 3 0b01 0x40 128-bit"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.4s, v17.4s, v18.4s",
"faddp v3.4s, v3.4s, v2.4s",
"faddp v3.4s, v3.4s, v2.4s",
"mov v2.s[0], v3.s[0]",
"mov v2.s[1], v3.s[0]",
"mov v2.s[2], v3.s[0]",
"mov v16.16b, v2.16b",
"mov v16.s[3], v3.s[0]"
"fmul v2.4s, v17.4s, v18.4s",
"faddp v2.4s, v2.4s, v2.4s",
"faddp s2, v2.2s",
"dup v16.4s, v2.s[0]"
]
},
"vdpps ymm0, ymm1, ymm2, 00000000b": {
@@ -3788,20 +3771,13 @@
]
},
"vdppd xmm0, xmm1, xmm2, 00001111b": {
"ExpectedInstructionCount": 8,
"ExpectedInstructionCount": 1,
"Optimal": "No",
"Comment": [
"Map 3 0b01 0x41 128-bit"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.2d, v17.2d, v18.2d",
"mov v3.d[0], v2.d[0]",
"mov v3.d[1], v2.d[0]",
"faddp v3.2d, v3.2d, v2.2d",
"mov v2.d[0], v3.d[0]",
"mov v16.16b, v2.16b",
"mov v16.d[1], v3.d[0]"
"movi v16.2d, #0x0"
]
},
"vdppd xmm0, xmm1, xmm2, 11110000b": {
@@ -3815,18 +3791,15 @@
]
},
"vdppd xmm0, xmm1, xmm2, 11111111b": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"Map 3 0b01 0x41 128-bit"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"fmul v3.2d, v17.2d, v18.2d",
"faddp v3.2d, v3.2d, v2.2d",
"mov v2.d[0], v3.d[0]",
"mov v16.16b, v2.16b",
"mov v16.d[1], v3.d[0]"
"fmul v2.2d, v17.2d, v18.2d",
"faddp d2, v2.2d",
"dup v16.2d, v2.d[0]"
]
},
"vmpsadbw xmm0, xmm1, xmm2, 000b": {
@@ -5130,7 +5103,7 @@
"Map 3 0b01 0xdf 128-bit"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #2000]",
"ldr q2, [x28, #2016]",
"movi v3.2d, #0x0",
"mov v16.16b, v17.16b",
"unimplemented (Unimplemented)",
@@ -5144,7 +5117,7 @@
"Map 3 0b01 0xdf 128-bit"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #2000]",
"ldr q2, [x28, #2016]",
"movi v3.2d, #0x0",
"mov v16.16b, v17.16b",
"unimplemented (Unimplemented)",