Merge pull request #3095 from Sonicadvance1/bswap

OpcodeDispatcher: Optimize 16-bit MOVBE
This commit is contained in:
Mai authored and GitHub committed 2023-09-15 05:18:26 -04:00
commit f84a264b0e
11 files changed
+100 -6

No files matched your search

@@ -856,6 +856,18 @@ DEF_OP(Bfi) {
GD = Res;
}
DEF_OP(Bfxil) {
auto Op = IROp->C<IR::IROp_Bfxil>();
uint64_t SourceMask = (1ULL << Op->Width) - 1;
if (Op->Width == 64) {
SourceMask = ~0ULL;
}
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Dest);
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
const uint64_t Res = (Src1 & ~SourceMask) | ((Src2 >> Op->lsb) & SourceMask);
GD = Res;
}
DEF_OP(Bfe) {
auto Op = IROp->C<IR::IROp_Bfe>();
@@ -84,6 +84,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes);
REGISTER_OP(REV, Rev);
REGISTER_OP(BFI, Bfi);
REGISTER_OP(BFXIL, Bfxil);
REGISTER_OP(BFE, Bfe);
REGISTER_OP(SBFE, Sbfe);
REGISTER_OP(SELECT, Select);
@@ -121,6 +121,7 @@ namespace FEXCore::CPU {
DEF_OP(CountLeadingZeroes);
DEF_OP(Rev);
DEF_OP(Bfi);
DEF_OP(Bfxil);
DEF_OP(Bfe);
DEF_OP(Sbfe);
DEF_OP(Select);
@@ -1177,6 +1177,29 @@ DEF_OP(Bfi) {
}
}
DEF_OP(Bfxil) {
auto Op = IROp->C<IR::IROp_Bfxil>();
const uint8_t OpSize = IROp->Size;
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto Dst = GetReg(Node);
const auto SrcDst = GetReg(Op->Dest.ID());
const auto Src = GetReg(Op->Src.ID());
if (Dst == SrcDst) {
// If Dst and SrcDst match then this turns in to a single instruction.
bfxil(EmitSize, Dst, Src, Op->lsb, Op->Width);
}
else {
// Destination didn't match the dst source register.
// TODO: Inefficient until FEX can have RA constraints here.
mov(EmitSize, TMP1, SrcDst);
bfxil(EmitSize, TMP1, Src, Op->lsb, Op->Width);
mov(EmitSize, Dst, TMP1.R());
}
}
DEF_OP(Bfe) {
auto Op = IROp->C<IR::IROp_Bfe>();
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
@@ -894,6 +894,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes);
REGISTER_OP(REV, Rev);
REGISTER_OP(BFI, Bfi);
REGISTER_OP(BFXIL, Bfxil);
REGISTER_OP(BFE, Bfe);
REGISTER_OP(SBFE, Sbfe);
REGISTER_OP(SELECT, Select);
@@ -272,6 +272,7 @@ private:
DEF_OP(CountLeadingZeroes);
DEF_OP(Rev);
DEF_OP(Bfi);
DEF_OP(Bfxil);
DEF_OP(Bfe);
DEF_OP(Sbfe);
DEF_OP(Select);
@@ -1056,6 +1056,34 @@ DEF_OP(Bfi) {
}
}
DEF_OP(Bfxil) {
auto Op = IROp->C<IR::IROp_Bfxil>();
const uint8_t OpSize = IROp->Size;
auto Dst = GetDst<RA_64>(Node);
uint64_t SourceMask = (1ULL << Op->Width) - 1;
if (Op->Width == 64) {
SourceMask = ~0ULL;
}
const uint64_t DestMask = ~SourceMask;
mov(TMP1, GetSrc<RA_64>(Op->Src.ID()));
shr(TMP1, Op->lsb);
mov(Dst, GetSrc<RA_64>(Op->Dest.ID()));
mov(TMP2, DestMask);
and_(Dst, TMP2);
mov(TMP2, SourceMask);
and_(TMP1, TMP2);
or_(Dst, TMP1);
if (OpSize != 8) {
mov(rcx, uint64_t((1ULL << (OpSize * 8)) - 1));
and_(Dst, rcx);
}
}
DEF_OP(Bfe) {
auto Op = IROp->C<IR::IROp_Bfe>();
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
@@ -1375,6 +1403,7 @@ void X86JITCore::RegisterALUHandlers() {
REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes);
REGISTER_OP(REV, Rev);
REGISTER_OP(BFI, Bfi);
REGISTER_OP(BFXIL, Bfxil);
REGISTER_OP(BFE, Bfe);
REGISTER_OP(SBFE, Sbfe);
REGISTER_OP(SELECT, Select);
@@ -280,6 +280,7 @@ private:
DEF_OP(CountLeadingZeroes);
DEF_OP(Rev);
DEF_OP(Bfi);
DEF_OP(Bfxil);
DEF_OP(Bfe);
DEF_OP(Sbfe);
DEF_OP(Select);
@@ -5647,9 +5647,24 @@ void OpDispatchBuilder::LZCNT(OpcodeArgs) {
}
void OpDispatchBuilder::MOVBEOp(OpcodeArgs) {
const uint8_t GPRSize = CTX->GetGPRSize();
const auto SrcSize = GetSrcSize(Op);
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, 1);
Src = _Rev(OpSizeFromSrc(Op), Src);
StoreResult(GPRClass, Op, Src, 1);
Src = _Rev(IR::SizeToOpSize(std::max<uint8_t>(4u, SrcSize)), Src);
if (SrcSize == 2) {
// 16-bit does an insert.
// Rev of 16-bit value as 32-bit replaces the result in the upper 16-bits of the result.
// bfxil the 16-bit result in to the GPR.
OrderedNode *Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags, -1);
auto Result = _Bfxil(IR::SizeToOpSize(GPRSize), 16, 16, Dest, Src);
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, -1);
}
else {
// 32-bit does regular zext
StoreResult(GPRClass, Op, Op->Dest, Src, -1);
}
}
void OpDispatchBuilder::CLWB(OpcodeArgs) {
+11
View File
@@ -1073,6 +1073,17 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Bfxil OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Dest, GPR:$Src": {
"Desc": ["Copies a bitfield from one GPR to another",
"Inserting in to the low bits of the destination",
"The source bitfield is from Src[(Width + lsb):lsb]",
"The bitfield is copied in to Dest[Width:0]"
],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Bfe OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Src": {
"Desc": ["Extracts a bitfield from one GPR with zext",
"The source bitfield is from Src[Width:0]",
+3 -4
View File
@@ -1012,16 +1012,15 @@
]
},
"movbe ax, word [rbx]": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x38 0xf0"
],
"ExpectedArm64ASM": [
"ldrh w20, [x7]",
"rev w20, w20",
"lsr w20, w20, #16",
"bfxil x4, x20, #0, #16"
"bfxil x4, x20, #16, #16"
]
},
"movbe eax, dword [rbx]": {