IR: Add VBroadcastFromMem opcode

Allows the implementations of the vbroadcast instructions to perform the
load and broadcast in one operation as opposed to doing the load and then
broadcast separately.

Notably, the broadcasting loads can also be used on systems that have SVE 128-bit
support as well, not only 256-bit.

On non-SVE systems, we use the equivalent AdvSIMD instructions.
This commit is contained in:
Lioncache committed 2023-08-17 12:17:24 -04:00
1 parent 6d562f8b3b
commit 879bc5176e
11 files changed
+195 -36

No files matched your search

@@ -153,6 +153,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
REGISTER_OP(STOREMEMTSO, StoreMem);
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
REGISTER_OP(VBROADCASTFROMMEM, VBroadcastFromMem);
REGISTER_OP(MEMSET, MemSet);
REGISTER_OP(MEMCPY, MemCpy);
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
@@ -187,6 +187,7 @@ namespace FEXCore::CPU {
DEF_OP(StoreMem);
DEF_OP(VLoadVectorMasked);
DEF_OP(VStoreVectorMasked);
DEF_OP(VBroadcastFromMem);
DEF_OP(MemSet);
DEF_OP(MemCpy);
DEF_OP(CacheLineClear);
@@ -418,6 +418,42 @@ DEF_OP(VStoreVectorMasked) {
}
}
DEF_OP(VBroadcastFromMem) {
const auto Op = IROp->C<IR::IROp_VBroadcastFromMem>();
const auto OpSize = IROp->Size;
const auto ElementSize = IROp->ElementSize;
const auto NumElements = OpSize / ElementSize;
const auto *MemData = *GetSrc<const uint8_t**>(Data->SSAData, Op->Address);
const auto BroadcastElement = [NumElements]<typename T>(void* Dst, const T* MemPtr) {
auto* DstU8 = static_cast<uint8_t*>(Dst);
for (size_t i = 0; i < NumElements; i++) {
std::memcpy(DstU8 + (i * sizeof(T)), MemPtr, sizeof(T));
}
};
switch (ElementSize) {
case 1:
BroadcastElement(GDP, MemData);
break;
case 2:
BroadcastElement(GDP, reinterpret_cast<const uint16_t*>(MemData));
break;
case 4:
BroadcastElement(GDP, reinterpret_cast<const uint32_t*>(MemData));
break;
case 8:
BroadcastElement(GDP, reinterpret_cast<const uint64_t*>(MemData));
break;
default:
LOGMAN_MSG_A_FMT("Unhandled VBroadcastFromMem element size: {}", ElementSize);
break;
}
}
DEF_OP(MemSet) {
const auto Op = IROp->C<IR::IROp_MemSet>();
const int32_t Size = Op->Size;
+1 -1
View File
@@ -964,7 +964,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
REGISTER_OP_RT(STOREMEMTSO, StoreMemTSO);
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
REGISTER_OP(VBROADCASTFROMMEM, VBroadcastFromMem);
REGISTER_OP(MEMSET, MemSet);
REGISTER_OP(MEMCPY, MemCpy);
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
@@ -342,6 +342,7 @@ private:
DEF_OP(StoreMemTSO);
DEF_OP(VLoadVectorMasked);
DEF_OP(VStoreVectorMasked);
DEF_OP(VBroadcastFromMem);
DEF_OP(MemSet);
DEF_OP(MemCpy);
DEF_OP(ParanoidLoadMemTSO);
@@ -1429,6 +1429,68 @@ DEF_OP(VStoreVectorMasked) {
}
}
DEF_OP(VBroadcastFromMem) {
const auto Op = IROp->C<IR::IROp_VBroadcastFromMem>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto ElementSize = IROp->ElementSize;
const auto Dst = GetVReg(Node);
const auto MemReg = GetReg(Op->Address.ID());
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8,
"Invalid element size");
if (HostSupportsSVE128 || HostSupportsSVE256) {
if (Is256Bit) {
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use SVE 256-bit broadcast");
}
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B.Zeroing()
: PRED_TMP_16B.Zeroing();
switch (ElementSize) {
case 1:
ld1rb(ARMEmitter::SubRegSize::i8Bit, Dst.Z(),
GoverningPredicate, MemReg);
break;
case 2:
ld1rh(ARMEmitter::SubRegSize::i16Bit, Dst.Z(),
GoverningPredicate, MemReg);
break;
case 4:
ld1rw(ARMEmitter::SubRegSize::i32Bit, Dst.Z(),
GoverningPredicate, MemReg);
break;
case 8:
ld1rd(Dst.Z(), GoverningPredicate, MemReg);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled VBroadcastFromMem size: {}", ElementSize);
return;
}
} else {
switch (ElementSize) {
case 1:
ld1r<ARMEmitter::SubRegSize::i8Bit>(Dst.Q(), MemReg);
break;
case 2:
ld1r<ARMEmitter::SubRegSize::i16Bit>(Dst.Q(), MemReg);
break;
case 4:
ld1r<ARMEmitter::SubRegSize::i32Bit>(Dst.Q(), MemReg);
break;
case 8:
ld1r<ARMEmitter::SubRegSize::i64Bit>(Dst.Q(), MemReg);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled VBroadcastFromMem size: {}", ElementSize);
return;
}
}
}
DEF_OP(StoreMem) {
const auto Op = IROp->C<IR::IROp_StoreMem>();
const auto OpSize = IROp->Size;
@@ -348,6 +348,7 @@ private:
DEF_OP(StoreMem);
DEF_OP(VLoadVectorMasked);
DEF_OP(VStoreVectorMasked);
DEF_OP(VBroadcastFromMem);
DEF_OP(MemSet);
DEF_OP(MemCpy);
DEF_OP(CacheLineClear);
@@ -844,6 +844,58 @@ DEF_OP(VStoreVectorMasked) {
}
}
DEF_OP(VBroadcastFromMem) {
const auto Op = IROp->C<IR::IROp_VBroadcastFromMem>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto ElementSize = IROp->ElementSize;
const auto Dst = GetDst(Node);
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Address.ID());
if (Is256Bit) {
const auto DstYMM = ToYMM(Dst);
switch (ElementSize) {
case 1:
vpbroadcastb(DstYMM, byte [MemReg]);
break;
case 2:
vpbroadcastw(DstYMM, word [MemReg]);
break;
case 4:
vpbroadcastd(DstYMM, dword [MemReg]);
break;
case 8:
vpbroadcastq(DstYMM, qword [MemReg]);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled VBroadcastFromMem element size: {}", ElementSize);
return;
}
} else {
switch (ElementSize) {
case 1:
vpbroadcastb(Dst, byte [MemReg]);
break;
case 2:
vpbroadcastw(Dst, word [MemReg]);
break;
case 4:
vpbroadcastd(Dst, dword [MemReg]);
break;
case 8:
vpbroadcastq(Dst, qword [MemReg]);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled VBroadcastFromMem element size: {}", ElementSize);
return;
}
}
}
DEF_OP(MemSet) {
const auto Op = IROp->C<IR::IROp_MemSet>();
@@ -1125,6 +1177,7 @@ void X86JITCore::RegisterMemoryHandlers() {
REGISTER_OP(STOREMEMTSO, StoreMem);
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
REGISTER_OP(VBROADCASTFROMMEM, VBroadcastFromMem);
REGISTER_OP(MEMSET, MemSet);
REGISTER_OP(MEMCPY, MemCpy);
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
@@ -1265,8 +1265,17 @@ void OpDispatchBuilder::VBROADCASTOp(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
Result = _VDupElement(DstSize, ElementSize, Src, 0);
} else {
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ElementSize, Op->Flags, -1);
Result = _VDupElement(DstSize, ElementSize, Src, 0);
if constexpr (ElementSize == 16) {
// TODO: Make 128-bit loads go through the broadcast path.
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ElementSize, Op->Flags, -1);
Result = _VDupElement(DstSize, ElementSize, Src, 0);
} else {
// Get the address to broadcast from into a GPR.
OrderedNode *Address = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], CTX->GetGPRSize(), Op->Flags, -1, false);
Address = AppendSegmentOffset(Address, Op->Flags);
Result = _VBroadcastFromMem(DstSize, ElementSize, Address);
}
}
if (Is128Bit) {
+6
View File
@@ -499,6 +499,12 @@
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VBroadcastFromMem u8:#RegisterSize, u8:#ElementSize, GPR:$Address": {
"Desc": ["Broadcasts an ElementSize value from memory into each element of a vector."],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"GPR = MemSet i1:$IsAtomic, u8:$Size, GPR:$Prefix, GPR:$Addr, GPR:$Value, GPR:$Length, GPR:$Direction": {
"Desc": ["Duplicates behaviour of x86 STOS repeat",
"Returns the final address that gets generated without the prefix appended."
+22 -33
View File
@@ -1063,40 +1063,37 @@
]
},
"vbroadcastss xmm0, [rax]": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x18 128-bit"
],
"ExpectedArm64ASM": [
"ldr s4, [x4]",
"dup v4.4s, v4.s[0]",
"ld1rw {z4.s}, p6/z, [x4]",
"mov v4.16b, v4.16b",
"mov v4.16b, v4.16b",
"mov z16.d, p7/m, z4.d"
]
},
"vbroadcastss ymm0, [rax]": {
"ExpectedInstructionCount": 3,
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x18 256-bit"
],
"ExpectedArm64ASM": [
"ldr s4, [x4]",
"mov z4.s, s4",
"ld1rw {z4.s}, p7/z, [x4]",
"mov z16.d, p7/m, z4.d"
]
},
"vbroadcastsd ymm0, [rax]": {
"ExpectedInstructionCount": 3,
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x19 256-bit"
],
"ExpectedArm64ASM": [
"ldr d4, [x4]",
"mov z4.d, d4",
"ld1rd {z4.d}, p7/z, [x4]",
"mov z16.d, p7/m, z4.d"
]
},
@@ -2357,14 +2354,13 @@
]
},
"vpbroadcastd xmm0, [rax]": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x58 128-bit"
],
"ExpectedArm64ASM": [
"ldr s4, [x4]",
"dup v4.4s, v4.s[0]",
"ld1rw {z4.s}, p6/z, [x4]",
"mov v4.16b, v4.16b",
"mov v4.16b, v4.16b",
"mov z16.d, p7/m, z4.d"
@@ -2383,14 +2379,13 @@
]
},
"vpbroadcastd ymm0, [rax]": {
"ExpectedInstructionCount": 3,
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x58 256-bit"
],
"ExpectedArm64ASM": [
"ldr s4, [x4]",
"mov z4.s, s4",
"ld1rw {z4.s}, p7/z, [x4]",
"mov z16.d, p7/m, z4.d"
]
},
@@ -2409,14 +2404,13 @@
]
},
"vpbroadcastq xmm0, [rax]": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x59 128-bit"
],
"ExpectedArm64ASM": [
"ldr d4, [x4]",
"dup v4.2d, v4.d[0]",
"ld1rd {z4.d}, p6/z, [x4]",
"mov v4.16b, v4.16b",
"mov v4.16b, v4.16b",
"mov z16.d, p7/m, z4.d"
@@ -2440,10 +2434,9 @@
"Comment": [
"Map 2 0b01 0x59 256-bit"
],
"ExpectedInstructionCount": 3,
"ExpectedInstructionCount": 2,
"ExpectedArm64ASM": [
"ldr d4, [x4]",
"mov z4.d, d4",
"ld1rd {z4.d}, p7/z, [x4]",
"mov z16.d, p7/m, z4.d"
]
},
@@ -2474,14 +2467,13 @@
]
},
"vpbroadcastb xmm0, [rax]": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x78 128-bit"
],
"ExpectedArm64ASM": [
"ldr b4, [x4]",
"dup v4.16b, v4.b[0]",
"ld1rb {z4.b}, p6/z, [x4]",
"mov v4.16b, v4.16b",
"mov v4.16b, v4.16b",
"mov z16.d, p7/m, z4.d"
@@ -2505,10 +2497,9 @@
"Comment": [
"Map 2 0b01 0x78 256-bit"
],
"ExpectedInstructionCount": 3,
"ExpectedInstructionCount": 2,
"ExpectedArm64ASM": [
"ldr b4, [x4]",
"mov z4.b, b4",
"ld1rb {z4.b}, p7/z, [x4]",
"mov z16.d, p7/m, z4.d"
]
},
@@ -2527,14 +2518,13 @@
]
},
"vpbroadcastw xmm0, [rax]": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x79 128-bit"
],
"ExpectedArm64ASM": [
"ldr h4, [x4]",
"dup v4.8h, v4.h[0]",
"ld1rh {z4.h}, p6/z, [x4]",
"mov v4.16b, v4.16b",
"mov v4.16b, v4.16b",
"mov z16.d, p7/m, z4.d"
@@ -2558,10 +2548,9 @@
"Comment": [
"Map 2 0b01 0x79 256-bit"
],
"ExpectedInstructionCount": 3,
"ExpectedInstructionCount": 2,
"ExpectedArm64ASM": [
"ldr h4, [x4]",
"mov z4.h, h4",
"ld1rh {z4.h}, p7/z, [x4]",
"mov z16.d, p7/m, z4.d"
]
},