Merge pull request #5904 from javelina-pkwy/opt/pmulhrsw

Faster translation for *PMULHRSW
This commit is contained in:
LC authored and GitHub committed 2026-09-02 19:54:24 -04:00
commit 83469208c9
107 files changed
+234 -194

No files matched your search

@@ -3335,6 +3335,55 @@ DEF_OP(VUShrNI2) {
}
}
DEF_OP(VRSHRN) {
const auto Op = IROp->C<IR::IROp_VRSHRN>();
const auto OpSize = IROp->Size;
const auto BitShift = Op->BitShift;
const auto SubRegSize = ConvertSubRegSize4(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
rshrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
} else {
rshrn(SubRegSize, Dst.D(), Vector.D(), BitShift);
}
}
DEF_OP(VRSHRNPair) {
const auto Op = IROp->C<IR::IROp_VRSHRNPair>();
const auto OpSize = IROp->Size;
const auto BitShift = Op->BitShift;
const auto SubRegSize = ConvertSubRegSize4(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto VectorLower = GetVReg(Op->VectorLower);
auto VectorUpper = GetVReg(Op->VectorUpper);
if (HostSupportsSVE256 && Is256Bit) {
rshrnb(SubRegSize, VTMP1.Z(), VectorLower.Z(), BitShift);
rshrnb(SubRegSize, VTMP2.Z(), VectorUpper.Z(), BitShift);
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP2.Z());
} else {
if (Dst == VectorUpper) {
// RSHRN writes the lower half and would destroy the upper input.
mov(VTMP1.Q(), VectorUpper.Q());
VectorUpper = VTMP1;
}
rshrn(SubRegSize, Dst.D(), VectorLower.D(), BitShift);
rshrn2(SubRegSize, Dst.Q(), VectorUpper.Q(), BitShift);
}
}
DEF_OP(VSXTL) {
const auto Op = IROp->C<IR::IROp_VSXTL>();
const auto OpSize = IROp->Size;
@@ -3786,32 +3786,14 @@ void OpDispatchBuilder::VPMULHWOp(OpcodeArgs, bool Signed) {
}
Ref OpDispatchBuilder::PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2) {
Ref Res {};
if (Size == OpSize::i64Bit) {
// Implementation is more efficient for 8byte registers
Res = _VSMull(Size << 1, OpSize::i16Bit, Src1, Src2);
Res = _VSShrI(Size << 1, OpSize::i32Bit, Res, 14);
auto OneVector = _VectorImm(Size << 1, OpSize::i32Bit, 1);
Res = _VAdd(Size << 1, OpSize::i32Bit, Res, OneVector);
return _VUShrNI(Size << 1, OpSize::i32Bit, Res, 1);
Ref Res = _VSMull(Size << 1, OpSize::i16Bit, Src1, Src2);
return _VRSHRN(Size << 1, OpSize::i32Bit, Res, 15);
} else {
// 128-bit and 256-bit are less efficient
Ref ResultLow;
Ref ResultHigh;
ResultLow = _VSMull(Size, OpSize::i16Bit, Src1, Src2);
ResultHigh = _VSMull2(Size, OpSize::i16Bit, Src1, Src2);
ResultLow = _VSShrI(Size, OpSize::i32Bit, ResultLow, 14);
ResultHigh = _VSShrI(Size, OpSize::i32Bit, ResultHigh, 14);
auto OneVector = _VectorImm(Size, OpSize::i32Bit, 1);
ResultLow = _VAdd(Size, OpSize::i32Bit, ResultLow, OneVector);
ResultHigh = _VAdd(Size, OpSize::i32Bit, ResultHigh, OneVector);
// Combine the results
Res = _VUShrNI(Size, OpSize::i32Bit, ResultLow, 1);
return _VUShrNI2(Size, OpSize::i32Bit, Res, ResultHigh, 1);
Ref ResultLow = _VSMull(Size, OpSize::i16Bit, Src1, Src2);
Ref ResultHigh = _VSMull2(Size, OpSize::i16Bit, Src1, Src2);
return _VRSHRNPair(Size, OpSize::i32Bit, ResultLow, ResultHigh, 15);
}
}
+24
View File
@@ -2036,6 +2036,30 @@
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
]
},
"FPR = VRSHRN OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
"TiedSource": 0,
"Desc": ["Rounding shift right each element and then narrows to the next lower element size",
"Writes result to the bottom half of the destination register, upper half is zeroed"
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize >> 1",
"EmitValidation": [
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
]
},
"FPR = VRSHRNPair OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper, u8:$BitShift": {
"Desc": ["Rounding shift right and narrow a pair of vectors into one result"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize >> 1",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit",
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
]
},
"FPR = VSXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": ["Sign extends elements from the source element size to the next size up"],
"DestSize": "RegisterSize",
+1 -1
View File
@@ -207,7 +207,7 @@ namespace DiskCache {
// TODO: This header is in global installed header path, but uses internal headers.
// Migrate this once that is fixed.
static constexpr uint16_t FormatVersion = 7;
static constexpr uint16_t FormatVersion = 8;
FEX_DEFAULT_VISIBILITY uint16_t GetFormatVersion();
} // namespace DiskCache
+31 -1
View File
@@ -5,7 +5,9 @@
"XMM2": ["0x31A6343B36E09E7A", "0x48134B294E4F5186", "0x0000000000000000", "0x0000000000000000"],
"XMM3": ["0x31A6343B36E09E7A", "0x48134B294E4F5186", "0x0000000000000000", "0x0000000000000000"],
"XMM4": ["0x31A6343B36E09E7A", "0x48134B294E4F5186", "0x31A6343B36E09E7A", "0x48134B294E4F5186"],
"XMM5": ["0x31A6343B36E09E7A", "0x48134B294E4F5186", "0x31A6343B36E09E7A", "0x48134B294E4F5186"]
"XMM5": ["0x31A6343B36E09E7A", "0x48134B294E4F5186", "0x31A6343B36E09E7A", "0x48134B294E4F5186"],
"XMM6": ["0x7FFE000180018000", "0xDD7BDD7BFFFFFFFF", "0x0000000000000000", "0x0000000000000000"],
"XMM7": ["0x7FFE000180018000", "0xDD7BDD7BFFFFFFFF", "0x8000FFFF7FFF7FFE", "0x00017FFD20000001"]
}
}
%endif
@@ -21,6 +23,12 @@ vpmulhrsw xmm3, xmm0, [rdx + 32]
vpmulhrsw ymm4, ymm0, ymm1
vpmulhrsw ymm5, ymm0, [rdx + 32]
vmovdqa xmm6, [rel .edge_a]
vpmulhrsw xmm6, xmm6, [rel .edge_b]
vmovdqa ymm7, [rel .edge_a256]
vpmulhrsw ymm7, ymm7, [rel .edge_b256]
hlt
align 32
@@ -34,3 +42,25 @@ dq 0x6162636465666768
dq 0x7172737475767778
dq 0x6162636465666768
dq 0x7172737475767778
align 16
.edge_a:
dq 0x7FFF800080008000
dq 0x3039CFC70001FFFF
.edge_b:
dq 0x7FFFFFFF7FFF8000
dq 0xA4605BA080007FFF
align 32
.edge_a256:
dq 0x7FFF800080008000
dq 0x3039CFC70001FFFF
dq 0x8000000180007FFF
dq 0x00027FFFC0004000
.edge_b256:
dq 0x7FFFFFFF7FFF8000
dq 0xA4605BA080007FFF
dq 0x8000800080017FFF
dq 0x40007FFEC0000002
+1 -1
View File
@@ -8,7 +8,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"roundss xmm0, xmm1, 00000000b": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
@@ -9,7 +9,7 @@
"DisabledHostFeatures": [
"RPRES"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"cvtsi2ss xmm0, eax": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"cvtsi2sd xmm0, eax": {
@@ -8,7 +8,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
@@ -9,7 +9,7 @@
"SVE256",
"RPRES"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"cvtsi2ss xmm0, eax": {
@@ -8,7 +8,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"cvtsi2sd xmm0, eax": {
@@ -9,7 +9,7 @@
"DisabledHostFeatures": [
"RPRES"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vsqrtss xmm0, xmm1, xmm2": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vroundss xmm0, xmm1, 00000000b": {
@@ -7,7 +7,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vfmaddsubps xmm0, xmm1, xmm2, xmm3": {
@@ -13,7 +13,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vmovups xmm0, xmm0": {
@@ -9,7 +9,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vaddsubpd xmm0, xmm1, xmm2": {
@@ -10,7 +10,7 @@
"FLAGM2",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vmovntps [rax], xmm0": {
@@ -10,7 +10,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vucomiss xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vpshufb xmm0, xmm1, xmm2": {
@@ -337,27 +337,20 @@
]
},
"vpmulhrsw xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 12,
"ExpectedInstructionCount": 5,
"Comment": [
"Map 2 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"sshr v2.4s, v2.4s, #14",
"sshr v3.4s, v3.4s, #14",
"movi v4.4s, #0x1",
"add v2.4s, v2.4s, v4.4s",
"add v3.4s, v3.4s, v4.4s",
"shrn v2.4h, v2.4s, #1",
"mov v0.16b, v2.16b",
"shrn2 v0.8h, v3.4s, #1",
"mov v16.16b, v0.16b",
"rshrn v16.4h, v2.4s, #15",
"rshrn2 v16.8h, v3.4s, #15",
"stp xzr, xzr, [x28, #192]"
]
},
"vpmulhrsw ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 12,
"Comment": [
"Map 2 0b01 0x0b 256-bit"
],
@@ -366,25 +359,14 @@
"ldr q3, [x28, #224]",
"smull v4.4s, v17.4h, v18.4h",
"smull2 v5.4s, v17.8h, v18.8h",
"sshr v4.4s, v4.4s, #14",
"sshr v5.4s, v5.4s, #14",
"movi v6.4s, #0x1",
"add v4.4s, v4.4s, v6.4s",
"add v5.4s, v5.4s, v6.4s",
"shrn v4.4h, v4.4s, #1",
"mov v0.16b, v4.16b",
"shrn2 v0.8h, v5.4s, #1",
"mov v16.16b, v0.16b",
"rshrn v16.4h, v4.4s, #15",
"rshrn2 v16.8h, v5.4s, #15",
"smull v4.4s, v2.4h, v3.4h",
"smull2 v2.4s, v2.8h, v3.8h",
"sshr v4.4s, v4.4s, #14",
"sshr v2.4s, v2.4s, #14",
"movi v3.4s, #0x1",
"add v4.4s, v4.4s, v3.4s",
"add v2.4s, v2.4s, v3.4s",
"shrn v4.4h, v4.4s, #1",
"shrn2 v4.8h, v2.4s, #1",
"str q4, [x28, #192]"
"mov v0.16b, v2.16b",
"rshrn v2.4h, v4.4s, #15",
"rshrn2 v2.8h, v0.4s, #15",
"str q2, [x28, #192]"
]
},
"vpermilps xmm0, xmm1, xmm2": {
@@ -8,7 +8,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vfmadd132ss xmm0, xmm1, xmm2": {
@@ -10,7 +10,7 @@
"FLAGM2",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vmovntdqa xmm0, [rax]": {
@@ -11,7 +11,7 @@
"SVE256",
"SVEBITPERM"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vtestps xmm0, xmm1": {
@@ -7,7 +7,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vpermq ymm0, ymm1, 1": {
@@ -8,7 +8,7 @@
"AFP",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vblendvps xmm0, xmm1, xmm2, xmm3": {
@@ -9,7 +9,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vpsrlw xmm0, xmm1, 0": {
+1 -1
View File
@@ -9,7 +9,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"lock add byte [rax], cl": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"sha1nexte xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"sha1nexte xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"pclmulqdq xmm0, xmm1, 00000b": {
+1 -1
View File
@@ -10,7 +10,7 @@
"AFP",
"RPRES"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"These 3DNow! instructions are optimal assuming that FEX doesn't SRA MMX registers",
@@ -8,7 +8,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"Instructions that explicitly push against the limits of ARM's loadstore instructions"
@@ -8,7 +8,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"Instructions that explicitly push against the limits of ARM's loadstore instructions"
@@ -13,7 +13,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -11,7 +11,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -9,7 +9,7 @@
"SVE256",
"RPRES"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -14,7 +14,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -14,7 +14,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -12,7 +12,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [],
"Instructions": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"lock add byte [rax], cl": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Chained add": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"ptest xmm0, xmm1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"The Witcher 3": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Sonic Mania movie player": {
@@ -10,7 +10,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"FMOD scalar loop": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"The Sims 1 hot block": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"add bl, cl": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"add al, 1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"push es": {
@@ -11,7 +11,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"ucomiss xmm0, xmm1": {
@@ -11,7 +11,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"sgdt [rax]": {
@@ -11,7 +11,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"xgetbv": {
@@ -11,7 +11,7 @@
"AFP",
"FCMA"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"ucomisd xmm0, xmm1": {
@@ -12,7 +12,7 @@
"AFP",
"CSSC"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"popcnt ax, bx": {
@@ -12,7 +12,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"popcnt ax, bx": {
@@ -12,7 +12,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vucomiss xmm0, xmm1": {
@@ -10,7 +10,7 @@
"AFP",
"SVEBITPERM"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vtestps xmm0, xmm1": {
@@ -9,7 +9,7 @@
"DisabledHostFeatures": [
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"blsr eax, ebx": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Block1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Block1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Block1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Block1": {
+1 -1
View File
@@ -11,7 +11,7 @@
"CSSC",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"fadd dword [rax]": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"fadd dword [rax]": {
+6 -18
View File
@@ -10,7 +10,7 @@
"FLAGM2",
"CRYPTO"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"pshufb mm0, mm1": {
@@ -338,9 +338,8 @@
]
},
"pmulhrsw mm0, mm1": {
"ExpectedInstructionCount": 12,
"ExpectedInstructionCount": 9,
"Comment": [
"Might be able to use sqdmulh",
"NP 0x0f 0x38 0x0b"
],
"ExpectedArm64ASM": [
@@ -350,32 +349,21 @@
"ldr d2, [x28, #1056]",
"ldr d3, [x28, #1072]",
"smull v2.4s, v2.4h, v3.4h",
"sshr v2.4s, v2.4s, #14",
"movi v3.4s, #0x1",
"add v2.4s, v2.4s, v3.4s",
"shrn v2.4h, v2.4s, #1",
"rshrn v2.4h, v2.4s, #15",
"str d2, [x28, #1056]",
"strh w20, [x28, #1064]"
]
},
"pmulhrsw xmm0, xmm1": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 4,
"Comment": [
"Might be able to use sqdmulh",
"0x66 0x0f 0x38 0x0b"
],
"ExpectedArm64ASM": [
"smull v2.4s, v16.4h, v17.4h",
"smull2 v3.4s, v16.8h, v17.8h",
"sshr v2.4s, v2.4s, #14",
"sshr v3.4s, v3.4s, #14",
"movi v4.4s, #0x1",
"add v2.4s, v2.4s, v4.4s",
"add v3.4s, v3.4s, v4.4s",
"shrn v2.4h, v2.4s, #1",
"mov v0.16b, v2.16b",
"shrn2 v0.8h, v3.4s, #1",
"mov v16.16b, v0.16b"
"rshrn v16.4h, v2.4s, #15",
"rshrn2 v16.8h, v3.4s, #15"
]
},
"pblendvb xmm0, xmm1": {
+1 -1
View File
@@ -8,7 +8,7 @@
"AFP",
"CRYPTO"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"SSE4.2 string instructions are skipped here.",
@@ -8,7 +8,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"dpps xmm0, xmm1, 00000000b": {
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"rep movsb": {
+1 -1
View File
@@ -10,7 +10,7 @@
"FLAGM2",
"MOPS"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"add bl, cl": {
@@ -9,7 +9,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"Instructions in this table that are marked optimal don't have their flag calculation part of this assumption",
@@ -9,7 +9,7 @@
"FlagM",
"FlagM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"push es": {
+1 -1
View File
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"pfrcpv mm0, mm1": {
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"rsqrtps xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"rsqrtss xmm0, xmm1": {
@@ -8,7 +8,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vrsqrtps xmm0, xmm1": {
+1 -1
View File
@@ -6,7 +6,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {}
}
@@ -33,7 +33,7 @@
" - 1b: ECX = MSB",
"[7] - Reserved"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"pcmpestrm xmm0, xmm1, 0_0_00_00_00b": {
+1 -1
View File
@@ -12,7 +12,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Comment": [
"MMX instructions are defined as optimal without SRA being used for these instructions.",
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"sgdt [rax]": {
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"xgetbv": {
@@ -7,7 +7,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"push fs": {
@@ -12,7 +12,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"movupd xmm0, xmm0": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"addsubpd xmm0, xmm1": {
@@ -8,7 +8,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"psrlw xmm0, xmm1": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"pmulhuw xmm0, xmm1": {
@@ -12,7 +12,7 @@
"FRINTTS",
"CSSC"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"movss xmm0, xmm1": {
@@ -10,7 +10,7 @@
"FCMA",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"movsd xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"addsubps xmm0, xmm1": {
@@ -10,7 +10,7 @@
"FCMA",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"cvtpd2dq xmm0, xmm1": {
@@ -12,7 +12,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"cvttss2si eax, xmm0": {
@@ -8,7 +8,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"movmskps eax, xmm0": {
+1 -1
View File
@@ -13,7 +13,7 @@
"FLAGM2",
"FRINTTS"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vmovups xmm0, xmm0": {
@@ -8,7 +8,7 @@
"FCMA"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vaddsubpd xmm0, xmm1, xmm2": {
@@ -13,7 +13,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vcvttss2si eax, xmm0": {
+8 -23
View File
@@ -11,7 +11,7 @@
"FLAGM2",
"SVEBITPERM"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"vpshufb xmm0, xmm1, xmm2": {
@@ -296,26 +296,19 @@
]
},
"vpmulhrsw xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 4,
"Comment": [
"Map 2 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"sshr v2.4s, v2.4s, #14",
"sshr v3.4s, v3.4s, #14",
"movi v4.4s, #0x1",
"add v2.4s, v2.4s, v4.4s",
"add v3.4s, v3.4s, v4.4s",
"shrn v2.4h, v2.4s, #1",
"mov v0.16b, v2.16b",
"shrn2 v0.8h, v3.4s, #1",
"mov v16.16b, v0.16b"
"rshrn v16.4h, v2.4s, #15",
"rshrn2 v16.8h, v3.4s, #15"
]
},
"vpmulhrsw ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 17,
"ExpectedInstructionCount": 9,
"Comment": [
"Map 2 0b01 0x0b 256-bit"
],
@@ -326,17 +319,9 @@
"smullb z0.s, z17.h, z18.h",
"smullt z1.s, z17.h, z18.h",
"zip2 z3.s, z0.s, z1.s",
"asr z2.s, z2.s, #14",
"asr z3.s, z3.s, #14",
"mov z4.s, #1",
"add z2.s, z2.s, z4.s",
"add z3.s, z3.s, z4.s",
"shrnb z2.h, z2.s, #1",
"uzp1 z2.h, z2.h, z2.h",
"shrnb z1.h, z3.s, #1",
"uzp1 z1.h, z1.h, z1.h",
"movprfx z16, z2",
"splice z16.h, p6, z16.h, z1.h"
"rshrnb z0.h, z2.s, #15",
"rshrnb z1.h, z3.s, #15",
"uzp1 z16.h, z0.h, z1.h"
]
},
"vpermilps xmm0, xmm1, xmm2": {
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 7
"BinaryCacheVersion": 8
},
"Instructions": {
"pext eax, ebx, ecx": {
Loaded 100 of 107 files, more files were not shown because too many files have changed in this diff. Show more