Merge pull request #4715 from Sonicadvance1/i_dislike_tuple_4

FEXCore: Remove reference SHA implementation
This commit is contained in:
LC authored and GitHub committed 2025-07-25 21:34:29 -04:00
commit bffb81241a
4 files changed
+59 -726

No files matched your search

@@ -11,10 +11,7 @@ $end_info$
#include <FEXCore/Utils/LogManager.h>
#include "Interface/Core/OpcodeDispatcher.h"
#include <array>
#include <cstdint>
#include <tuple>
#include <utility>
namespace FEXCore::IR {
class OrderedNode;
@@ -25,21 +22,13 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref RotatedNode {};
if (CTX->HostFeatures.SupportsSHA) {
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
} else {
// SHA1 extension missing, manually rotate.
// Emulate rotate.
auto ShiftLeft = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Dest, 30);
RotatedNode = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeft, Dest, 2);
}
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
auto RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
@@ -62,153 +51,49 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result;
if (CTX->HostFeatures.SupportsSHA) {
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
auto Src1 = SHADataShuffle(Dest);
auto Src2 = SHADataShuffle(Src);
// The result is swizzled differently than expected
Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
} else {
// Shift the incoming source left by a 32-bit element, inserting Zeros.
// This could be slightly improved to use a VInsGPR with the zero register.
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
// Emulate rotate.
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
// Element0 didn't get XOR'd with anything, so do it now.
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
// Emulate rotate.
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
}
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
auto Src1 = SHADataShuffle(Dest);
auto Src2 = SHADataShuffle(Src);
// The result is swizzled differently than expected
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
using FnType = Ref (*)(OpDispatchBuilder&, Ref, Ref, Ref);
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1c?
return Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._Andn(OpSize::i32Bit, D, B));
};
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p with different key
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
};
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1m
return Self.BitwiseAtLeastTwo(B, C, D);
};
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
};
constexpr std::array<uint32_t, 4> k_array {
0x5A827999U,
0x6ED9EBA1U,
0x8F1BBCDCU,
0xCA62C1D6U,
};
constexpr std::array<FnType, 4> fn_array {
f0,
f1,
f2,
f3,
};
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result {};
if (CTX->HostFeatures.SupportsSHA) {
Ref ConstantVector {};
switch (Imm8) {
case 0:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0);
break;
case 1:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1);
break;
case 2:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2);
break;
case 3:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3);
break;
}
Ref ConstantVector {};
switch (Imm8) {
case 0:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0);
break;
case 1:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1);
break;
case 2:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2);
break;
case 3:
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3);
break;
}
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
Ref Src1 = SHADataShuffle(Dest);
Ref Src2 = SHADataShuffle(Src);
Src2 = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src2, ConstantVector);
Ref Src1 = SHADataShuffle(Dest);
Ref Src2 = SHADataShuffle(Src);
Src2 = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src2, ConstantVector);
switch (Imm8) {
case 0: Result = SHADataShuffle(_VSha1C(Src1, ZeroRegister, Src2)); break;
case 2: Result = SHADataShuffle(_VSha1M(Src1, ZeroRegister, Src2)); break;
case 1:
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
}
} else {
const FnType Fn = fn_array[Imm8];
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
const auto Round0 = [&]() -> RoundResult {
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
auto A1 =
_Add(OpSize::i32Bit,
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
auto B1 = A;
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
auto D1 = C;
auto E1 = D;
return {A1, B1, C1, D1, E1};
};
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
// Kill W and E at the beginning
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
auto Q = _Add(OpSize::i32Bit, W, E);
auto ANext =
_Add(OpSize::i32Bit,
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
auto BNext = A;
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
auto DNext = C;
auto ENext = D;
return {ANext, BNext, CNext, DNext, ENext};
};
auto [A1, B1, C1, D1, E1] = Round0();
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
switch (Imm8) {
case 0: Result = SHADataShuffle(_VSha1C(Src1, ZeroRegister, Src2)); break;
case 2: Result = SHADataShuffle(_VSha1M(Src1, ZeroRegister, Src2)); break;
case 1:
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
}
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
@@ -218,69 +103,20 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result {};
if (CTX->HostFeatures.SupportsSHA) {
Result = _VSha256U0(Dest, Src);
} else {
const auto Sigma0 = [this](Ref W) -> Ref {
return _Xor(
OpSize::i32Bit,
_Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 7)), _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 18))),
_Lshr(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 3)));
};
auto W4 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
auto W3 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto W2 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
auto W1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
auto W0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, Sig3);
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, Sig2);
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, Sig1);
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, Sig0);
}
auto Result = _VSha256U0(Dest, Src);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
const auto Sigma1 = [this](Ref W) -> Ref {
return _Xor(
OpSize::i32Bit,
_Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 17)), _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 19))),
_Lshr(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 10)));
};
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
Ref Result;
if (CTX->HostFeatures.SupportsSHA) {
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Src2 = _VZip2(OpSize::i128Bit, OpSize::i64Bit, DupDst, Src);
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Src2 = _VZip2(OpSize::i128Bit, OpSize::i64Bit, DupDst, Src);
Result = _VSha256U1(Src1, Src2);
} else {
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
}
auto Result = _VSha256U1(Src1, Src2);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
@@ -301,81 +137,27 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
// Hardcoded to XMM0
auto XMM0 = LoadXMMRegister(0);
Ref Result;
if (CTX->HostFeatures.SupportsSHA) {
auto shuffle_abcd = [this](Ref Src1, Ref Src2) -> Ref {
// Generates a suitable SHA256 `abcd` configuration from x86 format.
auto Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
};
auto shuffle_abcd = [this](Ref Src1, Ref Src2) -> Ref {
// Generates a suitable SHA256 `abcd` configuration from x86 format.
auto Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
};
auto shuffle_efgh = [this](Ref Src1, Ref Src2) -> Ref {
// Generates a suitable SHA256 `efgh` configuration from x86 format.
auto Tmp = _VZip(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
};
auto shuffle_efgh = [this](Ref Src1, Ref Src2) -> Ref {
// Generates a suitable SHA256 `efgh` configuration from x86 format.
auto Tmp = _VZip(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
};
auto ABCD = shuffle_abcd(Dest, Src);
auto EFGH = shuffle_efgh(Dest, Src);
auto ABCD = shuffle_abcd(Dest, Src);
auto EFGH = shuffle_efgh(Dest, Src);
// x86 uses only the bottom 64-bits of the key, so duplicate to match ARM64 semantics.
auto Key = _VDupElement(OpSize::i128Bit, OpSize::i64Bit, XMM0, 0);
// x86 uses only the bottom 64-bits of the key, so duplicate to match ARM64 semantics.
auto Key = _VDupElement(OpSize::i128Bit, OpSize::i64Bit, XMM0, 0);
auto A = _VSha256H(ABCD, EFGH, Key);
auto B = _VSha256H2(EFGH, ABCD, Key);
Result = shuffle_abcd(A, B);
} else {
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
};
const auto Sigma0 = [this](Ref A) -> Ref {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
A, ShiftType::ROR, 22);
};
const auto Sigma1 = [this](Ref E) -> Ref {
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
E, ShiftType::ROR, 25);
};
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
Q0 = _Add(OpSize::i32Bit, Q0, H0);
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
Q1 = _Add(OpSize::i32Bit, Q1, G0);
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
// Rematerialize C0. As with G0.
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
}
auto A = _VSha256H(ABCD, EFGH, Key);
auto B = _VSha256H2(EFGH, ABCD, Key);
auto Result = shuffle_abcd(A, B);
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
}
+1 -1
View File
@@ -577,7 +577,7 @@ FEXCore::HostFeatures FetchHostFeatures(FEX::CPUFeatures& Features, bool Support
HostFeatures.SupportsRCPC = true;
HostFeatures.SupportsTSOImm9 = true;
HostFeatures.SupportsAVX = true;
HostFeatures.SupportsSHA = Feature.Feat_rand;
HostFeatures.SupportsSHA = Feature.Feat_sha;
HostFeatures.SupportsPMULL_128Bit = Feature.Feat_pclmulqdq;
HostFeatures.SupportsAES256 = Feature.Feat_aes;
HostFeatures.SupportsCLZERO = Feature.Feat_clzero;
-193
View File
@@ -693,199 +693,6 @@
"rev32 v16.8h, v2.8h"
]
},
"sha1nexte xmm0, xmm1": {
"ExpectedInstructionCount": 5,
"Comment": [
"0x66 0x0f 0x38 0xc8"
],
"ExpectedArm64ASM": [
"shl v2.4s, v16.4s, #30",
"usra v2.4s, v16.4s, #2",
"add v2.4s, v17.4s, v2.4s",
"mov v16.16b, v17.16b",
"mov v16.s[3], v2.s[3]"
]
},
"sha1msg1 xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Comment": [
"0x66 0x0f 0x38 0xc9"
],
"ExpectedArm64ASM": [
"ext v2.16b, v17.16b, v16.16b, #8",
"eor v16.16b, v16.16b, v2.16b"
]
},
"sha1msg2 xmm0, xmm1": {
"ExpectedInstructionCount": 11,
"Comment": [
"0x66 0x0f 0x38 0xca"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"ext v2.16b, v2.16b, v17.16b, #12",
"eor v2.16b, v16.16b, v2.16b",
"shl v3.4s, v2.4s, #1",
"usra v3.4s, v2.4s, #31",
"dup v2.4s, v3.s[3]",
"eor v2.16b, v16.16b, v2.16b",
"shl v4.4s, v2.4s, #1",
"usra v4.4s, v2.4s, #31",
"mov v16.16b, v3.16b",
"mov v16.s[0], v4.s[0]"
]
},
"sha256rnds2 xmm0, xmm1": {
"ExpectedInstructionCount": 56,
"Comment": [
"0x66 0x0f 0x38 0xcb"
],
"ExpectedArm64ASM": [
"mov w20, v17.s[1]",
"mov w21, v17.s[0]",
"mov w22, v16.s[1]",
"and w23, w20, w21",
"bic w22, w22, w20",
"eor w22, w23, w22",
"ror w23, w20, #6",
"eor w23, w23, w20, ror #11",
"eor w23, w23, w20, ror #25",
"add w22, w22, w23",
"mov w23, v16.s[0]",
"add w22, w22, w23",
"mov w23, v16.s[0]",
"add w22, w22, w23",
"mov w23, v17.s[3]",
"mov w24, v17.s[2]",
"mov w30, v16.s[3]",
"and w18, w24, w30",
"orr w30, w24, w30",
"and w30, w23, w30",
"orr w30, w30, w18",
"add w30, w22, w30",
"ror w18, w23, #2",
"eor w18, w18, w23, ror #13",
"eor w18, w18, w23, ror #22",
"add w30, w30, w18",
"mov w18, v16.s[2]",
"add w22, w22, w18",
"and w20, w22, w20",
"bic w21, w21, w22",
"eor w20, w20, w21",
"ror w21, w22, #6",
"eor w21, w21, w22, ror #11",
"eor w21, w21, w22, ror #25",
"add w20, w20, w21",
"mov w21, v16.s[1]",
"add w20, w20, w21",
"mov w21, v16.s[1]",
"add w20, w20, w21",
"and w21, w23, w24",
"orr w23, w23, w24",
"and w23, w30, w23",
"orr w21, w23, w21",
"add w21, w20, w21",
"ror w23, w30, #2",
"eor w23, w23, w30, ror #13",
"eor w23, w23, w30, ror #22",
"add w21, w21, w23",
"mov w23, v16.s[3]",
"add w20, w20, w23",
"mov v2.16b, v16.16b",
"mov v2.s[3], w21",
"mov v2.s[2], w30",
"mov v2.s[1], w20",
"mov v16.16b, v2.16b",
"mov v16.s[0], w22"
]
},
"sha256msg1 xmm0, xmm1": {
"ExpectedInstructionCount": 35,
"Comment": [
"0x66 0x0f 0x38 0xcc"
],
"ExpectedArm64ASM": [
"mov w20, v17.s[0]",
"mov w21, v16.s[3]",
"mov w22, v16.s[2]",
"mov w23, v16.s[1]",
"mov w24, v16.s[0]",
"ror w30, w20, #7",
"ror w18, w20, #18",
"eor w30, w30, w18",
"lsr w20, w20, #3",
"eor w20, w30, w20",
"add w20, w21, w20",
"ror w30, w21, #7",
"ror w18, w21, #18",
"eor w30, w30, w18",
"lsr w21, w21, #3",
"eor w21, w30, w21",
"add w21, w22, w21",
"ror w30, w22, #7",
"ror w18, w22, #18",
"eor w30, w30, w18",
"lsr w22, w22, #3",
"eor w22, w30, w22",
"add w22, w23, w22",
"ror w30, w23, #7",
"ror w18, w23, #18",
"eor w30, w30, w18",
"lsr w23, w23, #3",
"eor w23, w30, w23",
"add w23, w24, w23",
"mov v2.16b, v16.16b",
"mov v2.s[3], w20",
"mov v2.s[2], w21",
"mov v2.s[1], w22",
"mov v16.16b, v2.16b",
"mov v16.s[0], w23"
]
},
"sha256msg2 xmm0, xmm1": {
"ExpectedInstructionCount": 36,
"Comment": [
"0x66 0x0f 0x38 0xcd"
],
"ExpectedArm64ASM": [
"mov w20, v17.s[2]",
"mov w21, v17.s[3]",
"mov w22, v16.s[0]",
"ror w23, w20, #17",
"ror w24, w20, #19",
"eor w23, w23, w24",
"lsr w20, w20, #10",
"eor w20, w23, w20",
"add w20, w22, w20",
"mov w22, v16.s[1]",
"ror w23, w21, #17",
"ror w24, w21, #19",
"eor w23, w23, w24",
"lsr w21, w21, #10",
"eor w21, w23, w21",
"add w21, w22, w21",
"mov w22, v16.s[2]",
"ror w23, w20, #17",
"ror w24, w20, #19",
"eor w23, w23, w24",
"lsr w24, w20, #10",
"eor w23, w23, w24",
"add w22, w22, w23",
"mov w23, v16.s[3]",
"ror w24, w21, #17",
"ror w30, w21, #19",
"eor w24, w24, w30",
"lsr w30, w21, #10",
"eor w24, w24, w30",
"add w23, w23, w24",
"mov v2.16b, v16.16b",
"mov v2.s[3], w23",
"mov v2.s[2], w22",
"mov v2.s[1], w21",
"mov v16.16b, v2.16b",
"mov v16.s[0], w20"
]
},
"movbe ax, word [rbx]": {
"ExpectedInstructionCount": 3,
"Comment": [
-256
View File
@@ -1348,262 +1348,6 @@
"trn2 v2.4s, v3.4s, v2.4s",
"addp v16.8h, v4.8h, v2.8h"
]
},
"sha1rnds4 xmm0, xmm1, 00b": {
"ExpectedInstructionCount": 57,
"Comment": [
"0x66 0x0f 0x3a 0xcc"
],
"ExpectedArm64ASM": [
"mov w20, #0x7999",
"movk w20, #0x5a82, lsl #16",
"mov w21, v17.s[3]",
"mov w22, v16.s[3]",
"mov w23, v16.s[2]",
"mov w24, v16.s[1]",
"mov w30, v16.s[0]",
"and w18, w23, w24",
"bic w20, w30, w23",
"eor w20, w18, w20",
"ror w18, w22, #27",
"add w20, w20, w18",
"add w20, w20, w21",
"mov w21, #0x7999",
"movk w21, #0x5a82, lsl #16",
"add w20, w20, w21",
"ror w23, w23, #2",
"mov w18, v17.s[2]",
"add w30, w18, w30",
"and w18, w22, w23",
"bic w21, w24, w22",
"eor w21, w18, w21",
"ror w18, w20, #27",
"add w21, w21, w18",
"add w21, w21, w30",
"mov w30, #0x7999",
"movk w30, #0x5a82, lsl #16",
"add w21, w21, w30",
"ror w22, w22, #2",
"mov w18, v17.s[1]",
"add w24, w18, w24",
"and w18, w20, w22",
"bic w30, w23, w20",
"eor w30, w18, w30",
"ror w18, w21, #27",
"add w30, w30, w18",
"add w24, w30, w24",
"mov w30, #0x7999",
"movk w30, #0x5a82, lsl #16",
"add w24, w24, w30",
"ror w20, w20, #2",
"mov w18, v17.s[0]",
"add w23, w18, w23",
"and w18, w21, w20",
"bic w22, w22, w21",
"eor w22, w18, w22",
"ror w18, w24, #27",
"add w22, w22, w18",
"add w22, w22, w23",
"add w22, w22, w30",
"ror w21, w21, #2",
"mov v2.16b, v16.16b",
"mov v2.s[3], w22",
"mov v2.s[2], w24",
"mov v2.s[1], w21",
"mov v16.16b, v2.16b",
"mov v16.s[0], w20"
]
},
"sha1rnds4 xmm0, xmm1, 01b": {
"ExpectedInstructionCount": 53,
"Comment": [
"0x66 0x0f 0x3a 0xcc"
],
"ExpectedArm64ASM": [
"mov w20, #0xeba1",
"movk w20, #0x6ed9, lsl #16",
"mov w21, v17.s[3]",
"mov w22, v16.s[3]",
"mov w23, v16.s[2]",
"mov w24, v16.s[1]",
"mov w30, v16.s[0]",
"eor w18, w23, w24",
"eor w18, w18, w30",
"ror w20, w22, #27",
"add w20, w18, w20",
"add w20, w20, w21",
"mov w21, #0xeba1",
"movk w21, #0x6ed9, lsl #16",
"add w20, w20, w21",
"ror w23, w23, #2",
"mov w18, v17.s[2]",
"add w30, w18, w30",
"eor w18, w22, w23",
"eor w18, w18, w24",
"ror w21, w20, #27",
"add w21, w18, w21",
"add w21, w21, w30",
"mov w30, #0xeba1",
"movk w30, #0x6ed9, lsl #16",
"add w21, w21, w30",
"ror w22, w22, #2",
"mov w18, v17.s[1]",
"add w24, w18, w24",
"eor w18, w20, w22",
"eor w18, w18, w23",
"ror w30, w21, #27",
"add w30, w18, w30",
"add w24, w30, w24",
"mov w30, #0xeba1",
"movk w30, #0x6ed9, lsl #16",
"add w24, w24, w30",
"ror w20, w20, #2",
"mov w18, v17.s[0]",
"add w23, w18, w23",
"eor w18, w21, w20",
"eor w22, w18, w22",
"ror w18, w24, #27",
"add w22, w22, w18",
"add w22, w22, w23",
"add w22, w22, w30",
"ror w21, w21, #2",
"mov v2.16b, v16.16b",
"mov v2.s[3], w22",
"mov v2.s[2], w24",
"mov v2.s[1], w21",
"mov v16.16b, v2.16b",
"mov v16.s[0], w20"
]
},
"sha1rnds4 xmm0, xmm1, 10b": {
"ExpectedInstructionCount": 61,
"Comment": [
"0x66 0x0f 0x3a 0xcc"
],
"ExpectedArm64ASM": [
"mov w20, #0xbcdc",
"movk w20, #0x8f1b, lsl #16",
"mov w21, v17.s[3]",
"mov w22, v16.s[3]",
"mov w23, v16.s[2]",
"mov w24, v16.s[1]",
"mov w30, v16.s[0]",
"and w18, w24, w30",
"orr w20, w24, w30",
"and w20, w23, w20",
"orr w20, w20, w18",
"ror w18, w22, #27",
"add w20, w20, w18",
"add w20, w20, w21",
"mov w21, #0xbcdc",
"movk w21, #0x8f1b, lsl #16",
"add w20, w20, w21",
"ror w23, w23, #2",
"mov w18, v17.s[2]",
"add w30, w18, w30",
"and w18, w23, w24",
"orr w21, w23, w24",
"and w21, w22, w21",
"orr w21, w21, w18",
"ror w18, w20, #27",
"add w21, w21, w18",
"add w21, w21, w30",
"mov w30, #0xbcdc",
"movk w30, #0x8f1b, lsl #16",
"add w21, w21, w30",
"ror w22, w22, #2",
"mov w18, v17.s[1]",
"add w24, w18, w24",
"and w18, w22, w23",
"orr w30, w22, w23",
"and w30, w20, w30",
"orr w30, w30, w18",
"ror w18, w21, #27",
"add w30, w30, w18",
"add w24, w30, w24",
"mov w30, #0xbcdc",
"movk w30, #0x8f1b, lsl #16",
"add w24, w24, w30",
"ror w20, w20, #2",
"mov w18, v17.s[0]",
"add w23, w18, w23",
"and w18, w20, w22",
"orr w22, w20, w22",
"and w22, w21, w22",
"orr w22, w22, w18",
"ror w18, w24, #27",
"add w22, w22, w18",
"add w22, w22, w23",
"add w22, w22, w30",
"ror w21, w21, #2",
"mov v2.16b, v16.16b",
"mov v2.s[3], w22",
"mov v2.s[2], w24",
"mov v2.s[1], w21",
"mov v16.16b, v2.16b",
"mov v16.s[0], w20"
]
},
"sha1rnds4 xmm0, xmm1, 11b": {
"ExpectedInstructionCount": 53,
"Comment": [
"0x66 0x0f 0x3a 0xcc"
],
"ExpectedArm64ASM": [
"mov w20, #0xc1d6",
"movk w20, #0xca62, lsl #16",
"mov w21, v17.s[3]",
"mov w22, v16.s[3]",
"mov w23, v16.s[2]",
"mov w24, v16.s[1]",
"mov w30, v16.s[0]",
"eor w18, w23, w24",
"eor w18, w18, w30",
"ror w20, w22, #27",
"add w20, w18, w20",
"add w20, w20, w21",
"mov w21, #0xc1d6",
"movk w21, #0xca62, lsl #16",
"add w20, w20, w21",
"ror w23, w23, #2",
"mov w18, v17.s[2]",
"add w30, w18, w30",
"eor w18, w22, w23",
"eor w18, w18, w24",
"ror w21, w20, #27",
"add w21, w18, w21",
"add w21, w21, w30",
"mov w30, #0xc1d6",
"movk w30, #0xca62, lsl #16",
"add w21, w21, w30",
"ror w22, w22, #2",
"mov w18, v17.s[1]",
"add w24, w18, w24",
"eor w18, w20, w22",
"eor w18, w18, w23",
"ror w30, w21, #27",
"add w30, w18, w30",
"add w24, w30, w24",
"mov w30, #0xc1d6",
"movk w30, #0xca62, lsl #16",
"add w24, w24, w30",
"ror w20, w20, #2",
"mov w18, v17.s[0]",
"add w23, w18, w23",
"eor w18, w21, w20",
"eor w22, w18, w22",
"ror w18, w24, #27",
"add w22, w22, w18",
"add w22, w22, w23",
"add w22, w22, w30",
"ror w21, w21, #2",
"mov v2.16b, v16.16b",
"mov v2.s[3], w22",
"mov v2.s[2], w24",
"mov v2.s[1], w21",
"mov v16.16b, v2.16b",
"mov v16.s[0], w20"
]
}
}
}