mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 22:00:16 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
20caf69951 | ||
|
|
731e4d6271 | ||
|
|
48beb18f29 | ||
|
|
41c8731443 | ||
|
|
efb276f489 | ||
|
|
9febddefa3 | ||
|
|
ce9a860335 | ||
|
|
0123946ed1 | ||
|
|
c7098d0da1 | ||
|
|
65a162bdf9 | ||
|
|
2de485d02a | ||
|
|
bf64facaf6 | ||
|
|
2e7fc60dbf | ||
|
|
baddfe00b1 | ||
|
|
e7e59204d3 | ||
|
|
95d5b14f99 | ||
|
|
802eaee9c8 | ||
|
|
e89f48f237 | ||
|
|
e771e25632 | ||
|
|
25c202575e | ||
|
|
f59fc0f747 | ||
|
|
f7a076e00c | ||
|
|
56fadecdaf | ||
|
|
9f681f9e41 | ||
|
|
1b11f2f184 | ||
|
|
b2e61c37be | ||
|
|
7c0cf51f09 | ||
|
|
649a49488b | ||
|
|
aa2180d494 | ||
|
|
969cae581c | ||
|
|
1bf7e2544a | ||
|
|
fad22144a2 | ||
|
|
b440e176fb | ||
|
|
56c6b0d2cb | ||
|
|
0596a963e1 | ||
|
|
357cc04940 | ||
|
|
7c6e836865 | ||
|
|
54a7317312 | ||
|
|
1fb20710e6 | ||
|
|
811ea093b5 | ||
|
|
71fe9aee21 | ||
|
|
38c834e731 | ||
|
|
740ff60a71 | ||
|
|
ee69b9f650 | ||
|
|
a4565ce783 | ||
|
|
3131ee4de1 | ||
|
|
3ecc66fbcf | ||
|
|
e322e84785 | ||
|
|
f0fa7a5b6a | ||
|
|
718221be71 | ||
|
|
ee592ba03c | ||
|
|
56947f3a94 | ||
|
|
f41b9bc514 | ||
|
|
0463512c6c | ||
|
|
47369d058e | ||
|
|
b6f34fa209 | ||
|
|
03e0ca9833 | ||
|
|
d19473160d | ||
|
|
aeb2c98cbf | ||
|
|
2350ae5a07 | ||
|
|
6076d1747e | ||
|
|
f4ce6fb621 | ||
|
|
f0d25d413b | ||
|
|
c84503c271 | ||
|
|
1465df874b | ||
|
|
60c52e3826 | ||
|
|
13b806130b | ||
|
|
22058c06a1 | ||
|
|
d5c96555f1 | ||
|
|
c3e8cd8d30 | ||
|
|
d761fc44f4 | ||
|
|
a1aa2547ce | ||
|
|
44213c3968 | ||
|
|
830bd347c5 | ||
|
|
bcfdf39d63 | ||
|
|
bfed21870f | ||
|
|
4278c48791 | ||
|
|
c1db7a78b1 | ||
|
|
1bf06f8946 | ||
|
|
09bfe58827 | ||
|
|
474c780399 | ||
|
|
4a67893f1d | ||
|
|
73ffaa1e18 | ||
|
|
e675f4241a | ||
|
|
7a61d9d2b4 | ||
|
|
06b950a9cd | ||
|
|
de70651406 | ||
|
|
5ad7fdb2f3 | ||
|
|
9b6cc8f7e0 | ||
|
|
5c6de4ed14 | ||
|
|
82f936cb6d | ||
|
|
493b952e3f | ||
|
|
460a21625e | ||
|
|
00ab3f8440 | ||
|
|
034b62292b | ||
|
|
7b615a07d0 | ||
|
|
704841f004 | ||
|
|
c122f3faf9 | ||
|
|
f74f276d64 | ||
|
|
4b10cbdafd | ||
|
|
04c701e912 | ||
|
|
0c29f8faad | ||
|
|
5ed82fa0f6 | ||
|
|
6cca007817 | ||
|
|
c0a9463700 | ||
|
|
55b3d67eb4 | ||
|
|
65ddae1b71 | ||
|
|
063f524084 | ||
|
|
2dd0a82059 | ||
|
|
b810070e9f | ||
|
|
4c7ac17f7d | ||
|
|
84767c8b20 | ||
|
|
eccfb53bd5 | ||
|
|
f4e930262f | ||
|
|
d26d9e7e03 | ||
|
|
51c1998d70 | ||
|
|
87f818249d | ||
|
|
d81f92f5e2 | ||
|
|
29ffe02afe | ||
|
|
dfe4076fe4 | ||
|
|
44f9df062e | ||
|
|
6f98ef8cbb | ||
|
|
d54888a4c6 | ||
|
|
38c58706da | ||
|
|
8ad9286bd4 | ||
|
|
03bc962564 | ||
|
|
947b7ae6fe | ||
|
|
30317ac979 | ||
|
|
4544e7c1af | ||
|
|
34050431ff | ||
|
|
65439956bf | ||
|
|
a6cbce4fd7 | ||
|
|
17b851d4f3 | ||
|
|
362b5728be | ||
|
|
bd5159c7d5 | ||
|
|
660dfcd1f9 | ||
|
|
4e21177988 | ||
|
|
5a83a65905 | ||
|
|
061fc44923 | ||
|
|
0c7afa0672 | ||
|
|
30bf0d5767 | ||
|
|
ee8e3127d2 | ||
|
|
e8f64f2976 | ||
|
|
0a4b21da87 | ||
|
|
074777bc75 | ||
|
|
47c403998a | ||
|
|
0fa095cec5 | ||
|
|
cfa4e7f165 | ||
|
|
75b8226e7d | ||
|
|
9e3c50ca2c | ||
|
|
365d8b9508 | ||
|
|
02ebe06496 | ||
|
|
34d5e70e6b | ||
|
|
2d8bd7b59d | ||
|
|
3a6f5e638b | ||
|
|
c06274066f | ||
|
|
820b0be9f2 | ||
|
|
3e74817dd9 | ||
|
|
3b9a2d2141 | ||
|
|
9ba431a51d | ||
|
|
d0d2229db6 | ||
|
|
286258a2f2 | ||
|
|
ac14a88647 | ||
|
|
bf19578673 | ||
|
|
c5f396d889 | ||
|
|
22325500d9 | ||
|
|
51eede080c | ||
|
|
503c86d47d | ||
|
|
0e0887181d | ||
|
|
974cc591bd | ||
|
|
844afdb653 | ||
|
|
c3c643d9b7 | ||
|
|
b51b0c5f62 | ||
|
|
96bdefddc9 | ||
|
|
ba74a6a252 | ||
|
|
79427097e1 | ||
|
|
7ec21b7121 | ||
|
|
29115b3185 | ||
|
|
33b3814642 | ||
|
|
85431a8132 | ||
|
|
2d9ef56a8b | ||
|
|
b2ae829731 | ||
|
|
a3c544a9a1 | ||
|
|
9c2292289b | ||
|
|
b514548ca2 | ||
|
|
692a4a8fcd | ||
|
|
3a07cf7d70 | ||
|
|
7f5421fc26 | ||
|
|
04d62cd269 | ||
|
|
ac55e468a7 | ||
|
|
e20db7dc88 | ||
|
|
235bee9191 | ||
|
|
7a429b01c7 | ||
|
|
fbf5b14933 | ||
|
|
0dfd5dd96f | ||
|
|
6d5acec958 | ||
|
|
beed43e577 | ||
|
|
fc04b9113e | ||
|
|
68c038085a | ||
|
|
176f5a2860 | ||
|
|
e7c9623aa9 | ||
|
|
53ca2ac378 | ||
|
|
d140cb4450 | ||
|
|
0ba501636e | ||
|
|
ceca9fff17 | ||
|
|
3e31abb645 | ||
|
|
6c07cd319b | ||
|
|
5247b7124f | ||
|
|
6cec557855 | ||
|
|
1868bd6777 | ||
|
|
7c7efeda82 | ||
|
|
414486f1dd | ||
|
|
b8bc9659d4 | ||
|
|
d52a6e6fc4 | ||
|
|
5ab41056ab | ||
|
|
0692b34192 | ||
|
|
3636c332ff | ||
|
|
cb18963ded | ||
|
|
8cf92d3303 | ||
|
|
cdc5c15b4b | ||
|
|
ed313edd07 | ||
|
|
9b981a4f61 | ||
|
|
c791893b4a | ||
|
|
2605c7e0b3 | ||
|
|
f315948028 | ||
|
|
869367f7e2 | ||
|
|
3a2c7e8edd | ||
|
|
0a34a43976 | ||
|
|
78cd21d78f | ||
|
|
9f18de0196 | ||
|
|
0bffdc4e27 | ||
|
|
0f5ff53386 | ||
|
|
21611fc1ad | ||
|
|
99e1eb5452 | ||
|
|
8c3ca44c57 | ||
|
|
7a85e17d14 | ||
|
|
b533dcd86d | ||
|
|
a379ce6fed | ||
|
|
efd5c51110 | ||
|
|
886db4ffca | ||
|
|
e6f6ee2bcd | ||
|
|
4a6b5d4ec7 | ||
|
|
027e7624cb | ||
|
|
0e31077735 | ||
|
|
c8a9dd0d0a | ||
|
|
2bd7ddaa31 | ||
|
|
5566b4455b | ||
|
|
fed2c13521 | ||
|
|
37d092aab8 | ||
|
|
5626f4e50a | ||
|
|
efbc42dac3 | ||
|
|
160934884d | ||
|
|
af1cfcb9bd | ||
|
|
d6f726fc23 | ||
|
|
92ee071eb2 | ||
|
|
bdfa8ad4f3 | ||
|
|
37540f4927 | ||
|
|
000ab5ff19 | ||
|
|
f054274948 | ||
|
|
081907e168 | ||
|
|
1a115a8ce6 | ||
|
|
fd9158c75f | ||
|
|
1bde30a196 | ||
|
|
764aacaa8f | ||
|
|
a848211926 | ||
|
|
f1a42869d5 | ||
|
|
97a6ba9931 | ||
|
|
f4744f1e79 | ||
|
|
321f686108 | ||
|
|
90340350fa | ||
|
|
6f4fd4467b | ||
|
|
b31ce13f68 | ||
|
|
260d3b0b4e | ||
|
|
c8c7ffbf05 | ||
|
|
4b03185b77 | ||
|
|
52ec572db3 | ||
|
|
dc31cf83c6 | ||
|
|
8a4f51257d | ||
|
|
051469fa16 | ||
|
|
3f6cdc2e03 | ||
|
|
f3449f2b00 | ||
|
|
d7691d9a25 | ||
|
|
f414d4934c | ||
|
|
014917301a | ||
|
|
cc483acbde | ||
|
|
07f8a4eadd | ||
|
|
5fd127b53a | ||
|
|
ece89ddeab | ||
|
|
a1565a7d99 | ||
|
|
2f9b0de742 | ||
|
|
7e5f1b5859 | ||
|
|
40fd4bbb66 | ||
|
|
e4143352c9 | ||
|
|
c045e14837 | ||
|
|
f0f3c215ce | ||
|
|
8f4113d859 | ||
|
|
4cfc2ac1a4 | ||
|
|
d2aa5217dc | ||
|
|
e8baf4a28c | ||
|
|
e438d32879 | ||
|
|
32ef10b273 | ||
|
|
ad296051b7 | ||
|
|
e603136918 | ||
|
|
f8a61f7d7e | ||
|
|
cb5ba8baae | ||
|
|
079e70fc4e | ||
|
|
3d701f5fcf | ||
|
|
b31e4a3c27 | ||
|
|
992d6e8477 | ||
|
|
f6cdb165a3 | ||
|
|
f60388d160 | ||
|
|
96fa2ad8eb | ||
|
|
f143462ebe | ||
|
|
51fa61a1cd | ||
|
|
048e967546 | ||
|
|
608fd49ac3 | ||
|
|
bb630797b5 | ||
|
|
01a6e914f2 | ||
|
|
0190e1a00b | ||
|
|
11a87c22f9 | ||
|
|
5f6c0d2245 | ||
|
|
caaacb6c15 | ||
|
|
767c61c08b | ||
|
|
dc93e30451 | ||
|
|
368162df87 | ||
|
|
9eb2106ed2 | ||
|
|
cfc05b78fe | ||
|
|
d2a42c0038 | ||
|
|
58a3d174ec | ||
|
|
9c605e7333 | ||
|
|
1fd7e88ffd | ||
|
|
c5e7da0631 | ||
|
|
68f58e415f | ||
|
|
698abec25c | ||
|
|
1578f5ed47 | ||
|
|
80d7b5a5c9 | ||
|
|
eb023ceb51 | ||
|
|
5d1fda7d7f | ||
|
|
fafc04a59e | ||
|
|
ddcca58f64 | ||
|
|
fe8f5c745d | ||
|
|
d876224358 | ||
|
|
c4306f2f0a | ||
|
|
8b7a227820 | ||
|
|
d7afcee622 | ||
|
|
a421ff1105 | ||
|
|
5997030c97 | ||
|
|
3c8086373b | ||
|
|
10ec6b63b6 | ||
|
|
49087007be | ||
|
|
0897cd8777 | ||
|
|
4b945a9041 | ||
|
|
d66ed71bc6 | ||
|
|
b5b34df155 | ||
|
|
ff51435747 | ||
|
|
def561986b | ||
|
|
5c258d4a2a | ||
|
|
0d53f2b45c | ||
|
|
4f03044fe7 | ||
|
|
b967538435 | ||
|
|
e53f3969e9 | ||
|
|
5026bf8247 | ||
|
|
09cb4f5fc5 | ||
|
|
ddd7a550e4 | ||
|
|
3398f22c16 | ||
|
|
7f17519fbf | ||
|
|
a65884f9ae | ||
|
|
d70766f4c8 | ||
|
|
1365aa8881 | ||
|
|
fe5bc02682 | ||
|
|
6f096e7c4b | ||
|
|
389ad737e6 | ||
|
|
c00f7813a2 | ||
|
|
e5ceaa182d | ||
|
|
eeb8eb1824 | ||
|
|
7c6444c37c | ||
|
|
4a179c8f87 | ||
|
|
2b3895a514 | ||
|
|
0bb0f9cec7 | ||
|
|
6a07ea73a8 | ||
|
|
5c51c54ccc | ||
|
|
c740801ea5 |
No files matched your search
+5
-2
@@ -412,10 +412,13 @@ configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/include/Config.h.in
|
||||
${CMAKE_BINARY_DIR}/generated/ConfigDefines.h)
|
||||
|
||||
include(CTest)
|
||||
if (BUILD_TESTS)
|
||||
include(CTest)
|
||||
enable_testing()
|
||||
message(STATUS "Unit tests are enabled")
|
||||
if (NOT BUILD_TESTING)
|
||||
# CMake checks this variable before generating CTestTestfile.cmake
|
||||
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
|
||||
endif()
|
||||
|
||||
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
|
||||
if (TEST_JOB_COUNT)
|
||||
|
||||
@@ -354,6 +354,7 @@ enum class SystemRegister : uint32_t {
|
||||
RNDRRS = GenSystemReg<0b11, 0b011, 0b0010, 0b0100, 0b001>(),
|
||||
NZCV = GenSystemReg<0b11, 0b011, 0b0100, 0b0010, 0b000>(),
|
||||
FPCR = GenSystemReg<0b11, 0b011, 0b0100, 0b0100, 0b000>(),
|
||||
TPIDRRO_EL0 = GenSystemReg<0b11, 0b011, 0b1101, 0b0000, 0b011>(),
|
||||
CNTFRQ_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b000>(),
|
||||
CNTVCT_EL0 = GenSystemReg<0b11, 0b011, 0b1110, 0b0000, 0b010>(),
|
||||
};
|
||||
|
||||
Vendored
+1
-1
Submodule External/xbyak updated: f17cb9d6b9...c68cc53d18.
@@ -323,8 +323,8 @@ def print_ir_structs(defines):
|
||||
output_file.write("struct __attribute__((packed)) IROp_Header {\n")
|
||||
output_file.write("\tvoid* Data[0];\n")
|
||||
output_file.write("\tIROps Op;\n\n")
|
||||
output_file.write("\tuint8_t Size;\n")
|
||||
output_file.write("\tuint8_t ElementSize;\n")
|
||||
output_file.write("\tIR::OpSize Size;\n")
|
||||
output_file.write("\tIR::OpSize ElementSize;\n")
|
||||
|
||||
output_file.write("\ttemplate<typename T>\n")
|
||||
output_file.write("\tT const* C() const { return reinterpret_cast<T const*>(Data); }\n")
|
||||
@@ -630,20 +630,19 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->Size;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpElementSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\treturn HeaderOp->ElementSize;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetName(HeaderOp->Op));\n")
|
||||
output_file.write("\t\treturn HeaderOp->Size / HeaderOp->ElementSize;\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n")
|
||||
output_file.write("\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n")
|
||||
@@ -699,8 +698,12 @@ def print_ir_allocator_helpers():
|
||||
|
||||
# We gather the "has x87?" flag as we go. This saves the user from
|
||||
# having to keep track of whether they emitted any x87.
|
||||
# Also changes the mmx state to X87.
|
||||
if op.LoweredX87:
|
||||
output_file.write("\t\tRecordX87Use();\n")
|
||||
output_file.write(
|
||||
"\t\tif(MMXState == MMXState_MMX) ChgStateMMX_X87();\n"
|
||||
)
|
||||
|
||||
output_file.write("\t\tauto _Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
@@ -724,11 +727,11 @@ def print_ir_allocator_helpers():
|
||||
# We can only infer a size if we have arguments
|
||||
if op.DestSize == None:
|
||||
# We need to infer destination size
|
||||
output_file.write("\t\tuint8_t InferSize = 0;\n")
|
||||
output_file.write("\t\tIR::OpSize InferSize = OpSize::iUnsized;\n")
|
||||
if len(op.Arguments) != 0:
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tuint8_t Size{} = GetOpSize({});\n".format(arg.Name, arg.Name))
|
||||
output_file.write("\t\tauto Size{} = GetOpSize({});\n".format(arg.Name, arg.Name))
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tInferSize = std::max(InferSize, Size{});\n".format(arg.Name))
|
||||
@@ -741,7 +744,7 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\t_Op.first->Header.Size = {};\n".format(op.DestSize))
|
||||
|
||||
if op.NumElements == None:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size / ({});\n".format(1))
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size;\n")
|
||||
else:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = _Op.first->Header.Size / ({});\n".format(op.NumElements))
|
||||
|
||||
@@ -826,4 +829,3 @@ print_ir_dispatcher_defs()
|
||||
print_ir_dispatcher_dispatch()
|
||||
|
||||
output_dispatch_file.close()
|
||||
|
||||
@@ -105,17 +105,17 @@ set (SRCS
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
Interface/Core/JIT/Arm64/JIT.cpp
|
||||
Interface/Core/JIT/Arm64/ALUOps.cpp
|
||||
Interface/Core/JIT/Arm64/AtomicOps.cpp
|
||||
Interface/Core/JIT/Arm64/BranchOps.cpp
|
||||
Interface/Core/JIT/Arm64/ConversionOps.cpp
|
||||
Interface/Core/JIT/Arm64/EncryptionOps.cpp
|
||||
Interface/Core/JIT/Arm64/MemoryOps.cpp
|
||||
Interface/Core/JIT/Arm64/MiscOps.cpp
|
||||
Interface/Core/JIT/Arm64/MoveOps.cpp
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp
|
||||
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
|
||||
Interface/Core/JIT/JIT.cpp
|
||||
Interface/Core/JIT/ALUOps.cpp
|
||||
Interface/Core/JIT/AtomicOps.cpp
|
||||
Interface/Core/JIT/BranchOps.cpp
|
||||
Interface/Core/JIT/ConversionOps.cpp
|
||||
Interface/Core/JIT/EncryptionOps.cpp
|
||||
Interface/Core/JIT/MemoryOps.cpp
|
||||
Interface/Core/JIT/MiscOps.cpp
|
||||
Interface/Core/JIT/MoveOps.cpp
|
||||
Interface/Core/JIT/VectorOps.cpp
|
||||
Interface/Core/JIT/Arm64Relocations.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/H0F38Tables.cpp
|
||||
@@ -126,7 +126,6 @@ set (SRCS
|
||||
Interface/Core/X86Tables/SecondaryTables.cpp
|
||||
Interface/Core/X86Tables/VEXTables.cpp
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
|
||||
@@ -233,6 +233,10 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
// Zero is a special case, the significand for +/- 0 is +/- zero.
|
||||
if (lhs.Exponent == 0x0 && lhs.Significand == 0x0) {
|
||||
return lhs;
|
||||
}
|
||||
X80SoftFloat Tmp = lhs;
|
||||
Tmp.Exponent = 0x3FFF;
|
||||
Tmp.Sign = lhs.Sign;
|
||||
@@ -256,6 +260,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
return Result;
|
||||
#else
|
||||
// Zero is a special case, the exponent is always -inf
|
||||
if (lhs.Exponent == 0x0 && lhs.Significand == 0x0) {
|
||||
X80SoftFloat Result(1, 0x7FFFUL, 0x8000'0000'0000'0000UL);
|
||||
return Result;
|
||||
}
|
||||
|
||||
int32_t TrueExp = lhs.Exponent - ExponentBias;
|
||||
return i32_to_extF80(TrueExp);
|
||||
#endif
|
||||
|
||||
@@ -24,14 +24,6 @@ fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNe
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>(Features);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
CustomExitHandler = std::move(handler);
|
||||
}
|
||||
|
||||
ExitHandler FEXCore::Context::ContextImpl::GetExitHandler() const {
|
||||
return CustomExitHandler;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
@@ -81,11 +81,6 @@ public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
|
||||
void SetExitHandler(ExitHandler handler) override;
|
||||
ExitHandler GetExitHandler() const override;
|
||||
|
||||
ExitReason RunUntilExit(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void ExecuteThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) override;
|
||||
@@ -113,33 +108,29 @@ public:
|
||||
* Usecases:
|
||||
* Parent thread Creation:
|
||||
* - Thread = CreateThread(InitialRIP, InitialStack, nullptr, 0);
|
||||
* - CTX->RunUntilExit(Thread);
|
||||
* - CTX->ExecuteThread(Thread);
|
||||
* OS thread Creation:
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
* - ThreadHandler calls `CTX->ExecutionThread(Thread)`
|
||||
* - ThreadHandler calls `CTX->ExecuteThread(Thread)`
|
||||
* OS fork (New thread created with a clone of thread state):
|
||||
* - clone{2, 3}
|
||||
* - Thread = CreateThread(0, 0, CopyOfThreadState, PPID);
|
||||
* - ExecutionThread(Thread); // Starts executing without creating another host thread
|
||||
* - ExecuteThread(Thread); // Starts executing without creating another host thread
|
||||
* Thunk callback executing guest code from native host thread
|
||||
* - Thread = CreateThread(0, 0, NewState, PPID);
|
||||
* - InitializeThreadTLSData(Thread);
|
||||
* - HandleCallback(Thread, RIP);
|
||||
*/
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) override;
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
/**
|
||||
* @brief Destroys this FEX thread object and stops tracking it internally
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState* Thread, bool NeedsTLSUninstall) override;
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
#ifndef _WIN32
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
@@ -235,8 +226,6 @@ public:
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
} Config;
|
||||
|
||||
std::atomic_bool CoreShuttingDown {false};
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
uint32_t StrictSplitLockMutex {};
|
||||
@@ -249,8 +238,6 @@ public:
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
@@ -302,18 +289,8 @@ public:
|
||||
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Initializes TID, PID and TLS data for a thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void InitializeThreadTLSData(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState* ParentThread, FEXCore::Core::InternalThreadState* ChildThread);
|
||||
|
||||
uint8_t GetGPRSize() const {
|
||||
return Config.Is64BitMode ? 8 : 4;
|
||||
IR::OpSize GetGPROpSize() const {
|
||||
return Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
|
||||
}
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
@@ -381,7 +358,6 @@ private:
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
|
||||
@@ -90,7 +90,7 @@ namespace ProductNames {
|
||||
#endif
|
||||
} // namespace ProductNames
|
||||
|
||||
static uint32_t GetCPUID() {
|
||||
uint32_t GetCPUID_Syscall() {
|
||||
uint32_t CPU {};
|
||||
FHU::Syscalls::getcpu(&CPU, nullptr);
|
||||
return CPU;
|
||||
@@ -138,6 +138,12 @@ uint32_t GetCycleCounterFrequency() {
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint32_t GetCPUID_TPIDRRO() {
|
||||
uint64_t Result {};
|
||||
__asm("mrs %[Res], TPIDRRO_EL0" : [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
PerCPUData.resize(Cores);
|
||||
|
||||
@@ -895,11 +901,11 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) con
|
||||
// Extended processor and feature bits
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) const {
|
||||
|
||||
// RDTSCP is disabled on WIN32/Wine because there is no sane way to query processor ID.
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t SUPPORTS_RDTSCP = 1;
|
||||
#else
|
||||
constexpr uint32_t SUPPORTS_RDTSCP = 0;
|
||||
// RDTSCP under WIN32 is only supported if CPUIndex is available in TPIDRRO.
|
||||
const uint32_t SUPPORTS_RDTSCP = SupportsCPUIndexInTPIDRRO;
|
||||
#endif
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
|
||||
@@ -1213,12 +1219,20 @@ FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() const {
|
||||
}
|
||||
|
||||
CPUIDEmu::CPUIDEmu(const FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx} {
|
||||
: CTX {ctx}
|
||||
, SupportsCPUIndexInTPIDRRO {CTX->HostFeatures.SupportsCPUIndexInTPIDRRO}
|
||||
, GetCPUID {GetCPUID_Syscall} {
|
||||
Cores = CTX->HostFeatures.CPUMIDRs.size();
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
SetupFeatures();
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
if (SupportsCPUIndexInTPIDRRO) {
|
||||
GetCPUID = GetCPUID_TPIDRRO;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
} // namespace FEXCore
|
||||
@@ -115,6 +115,7 @@ public:
|
||||
|
||||
private:
|
||||
const FEXCore::Context::ContextImpl* CTX;
|
||||
bool SupportsCPUIndexInTPIDRRO {};
|
||||
bool Hybrid {};
|
||||
uint32_t Cores {};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
@@ -510,5 +511,8 @@ private:
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
}};
|
||||
|
||||
using GetCPUIDPtr = uint32_t (*)();
|
||||
GetCPUIDPtr GetCPUID;
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -9,14 +9,14 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "Interface/Core/ArchHelpers//Arm64Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
@@ -354,8 +354,6 @@ bool ContextImpl::InitCore() {
|
||||
if (Config.GdbServer) {
|
||||
// If gdbserver is enabled then this needs to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
// FEX needs to start paused when gdb is enabled.
|
||||
StartPaused = true;
|
||||
}
|
||||
|
||||
return true;
|
||||
@@ -365,29 +363,17 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason ContextImpl::RunUntilExit(FEXCore::Core::InternalThreadState* Thread) {
|
||||
ExecutionThread(Thread);
|
||||
|
||||
CoreShuttingDown.store(true);
|
||||
|
||||
if (CustomExitHandler) {
|
||||
CustomExitHandler(Thread, FEXCore::Context::ExitReason::EXIT_SHUTDOWN);
|
||||
return Thread->ExitReason;
|
||||
}
|
||||
|
||||
return FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
|
||||
}
|
||||
|
||||
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
}
|
||||
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeThreadTLSData(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Let's do some initial bookkeeping here
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
#endif
|
||||
// If it is the parent thread that died then just leave
|
||||
FEX_TODO("This doesn't make sense when the parent thread doesn't outlive its children");
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
|
||||
@@ -402,8 +388,6 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
|
||||
Dispatcher->InitThreadPointers(Thread);
|
||||
|
||||
Thread->CTX = this;
|
||||
|
||||
Thread->PassManager->AddDefaultPasses(this);
|
||||
Thread->PassManager->AddDefaultValidationPasses();
|
||||
|
||||
@@ -418,7 +402,9 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
|
||||
FEXCore::Core::InternalThreadState*
|
||||
ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {};
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
|
||||
.CTX = this,
|
||||
};
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
@@ -443,13 +429,7 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXC
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState* Thread, bool NeedsTLSUninstall) {
|
||||
if (NeedsTLSUninstall) {
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
}
|
||||
|
||||
void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
FEXCore::Allocator::VirtualProtect(&Thread->InterruptFaultPage, sizeof(Thread->InterruptFaultPage),
|
||||
Allocator::ProtectOptions::Read | Allocator::ProtectOptions::Write);
|
||||
delete Thread;
|
||||
@@ -583,7 +563,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount);
|
||||
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
|
||||
@@ -599,7 +579,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
@@ -642,8 +622,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(
|
||||
Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -673,7 +652,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
const bool NeedsBlockEnd =
|
||||
@@ -687,11 +666,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->ExitFunction(
|
||||
Thread->OpDispatcher->_EntrypointOffset(IR::SizeToOpSize(GPRSize), Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -886,45 +862,6 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
void ContextImpl::ExecutionThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
|
||||
InitializeThreadTLSData(Thread);
|
||||
|
||||
// Now notify the thread that we are initialized
|
||||
Thread->ThreadWaiting.NotifyAll();
|
||||
|
||||
if (StartPaused || Thread->StartPaused) {
|
||||
// Parent thread doesn't need to wait to run
|
||||
Thread->StartRunning.Wait();
|
||||
}
|
||||
|
||||
if (!Thread->RunningEvents.EarlyExit.load()) {
|
||||
Thread->RunningEvents.WaitingToStart = false;
|
||||
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_NONE;
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
|
||||
static_cast<ContextImpl*>(Thread->CTX)->Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
Thread->RunningEvents.Running = false;
|
||||
}
|
||||
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
|
||||
// If it is the parent thread that died then just leave
|
||||
FEX_TODO("This doesn't make sense when the parent thread doesn't outlive its children");
|
||||
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
@@ -950,6 +887,10 @@ void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* T
|
||||
}
|
||||
|
||||
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
if (!Thread) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
@@ -1013,12 +954,13 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
IRHeader.first->Blocks = emit->WrapNode(Block);
|
||||
emit->SetCurrentCodeBlock(Block);
|
||||
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
if (GPRSize == 8) {
|
||||
if (GPRSize == IR::OpSize::i64Bit) {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), X86State::REG_R11, IR::GPRClass, GPRSize);
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(8, 8, emit->_Constant(Entrypoint)), offsetof(Core::CPUState, mm[0][0]));
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->_Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
},
|
||||
|
||||
@@ -926,7 +926,7 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
|
||||
// If the RIP setting is conditional AND within our symbol range then it can be considered for multiblock
|
||||
uint64_t TargetRIP = 0;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
bool Conditional = true;
|
||||
|
||||
switch (DecodeInst->OP) {
|
||||
@@ -954,7 +954,7 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
default: return; break;
|
||||
}
|
||||
|
||||
if (GPRSize == 4) {
|
||||
if (GPRSize == IR::OpSize::i32Bit) {
|
||||
// If we are running a 32bit guest then wrap around addresses that go above 32bit
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
@@ -995,13 +995,13 @@ bool Decoder::BranchTargetCanContinue(bool FinalInstruction) const {
|
||||
}
|
||||
|
||||
uint64_t TargetRIP = 0;
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
|
||||
if (DecodeInst->OP == 0xE8) { // Call - immediate target
|
||||
const uint64_t NextRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
TargetRIP = DecodeInst->PC + DecodeInst->InstSize + DecodeInst->Src[0].Literal();
|
||||
|
||||
if (GPRSize == 4) {
|
||||
if (GPRSize == IR::OpSize::i32Bit) {
|
||||
// If we are running a 32bit guest then wrap around addresses that go above 32bit
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
@@ -79,17 +79,17 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::IROp_Header* IROp, FallbackInfo* Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
switch (IROp->Op) {
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
@@ -99,11 +99,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
@@ -115,7 +115,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
|
||||
SupportsPreserveAllABI};
|
||||
@@ -124,7 +124,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
|
||||
SupportsPreserveAllABI};
|
||||
@@ -133,7 +133,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
|
||||
SupportsPreserveAllABI};
|
||||
@@ -156,11 +156,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
|
||||
+132
-128
@@ -8,7 +8,7 @@ $end_info$
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -54,8 +54,8 @@ DEF_OP(EntrypointOffset) {
|
||||
auto Constant = Entry + Op->Offset;
|
||||
auto Dst = GetReg(Node);
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 4) {
|
||||
const auto OpSize = IROp->Size;
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
@@ -92,10 +92,10 @@ DEF_OP(AddNZCV) {
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size >= 4, "Constant not allowed here");
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
cmn(EmitSize, Src1, Const);
|
||||
} else if (IROp->Size < 4) {
|
||||
unsigned Shift = 32 - (8 * IROp->Size);
|
||||
} else if (IROp->Size < IR::OpSize::i32Bit) {
|
||||
unsigned Shift = 32 - IR::OpSizeAsBits(IROp->Size);
|
||||
|
||||
lsl(ARMEmitter::Size::i32Bit, TMP1, Src1, Shift);
|
||||
cmn(EmitSize, TMP1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
@@ -165,7 +165,7 @@ DEF_OP(TestNZ) {
|
||||
// Shift the sign bit into place, clearing out the garbage in upper bits.
|
||||
// Adding zero does an effective test, setting NZ according to the result and
|
||||
// zeroing CV.
|
||||
if (IROp->Size < 4) {
|
||||
if (IROp->Size < IR::OpSize::i32Bit) {
|
||||
// Cheaper to and+cmn than to lsl+lsl+tst, so do the and ourselves if
|
||||
// needed.
|
||||
if (Op->Src1 != Op->Src2) {
|
||||
@@ -179,7 +179,7 @@ DEF_OP(TestNZ) {
|
||||
Src1 = TMP1;
|
||||
}
|
||||
|
||||
unsigned Shift = 32 - (IROp->Size * 8);
|
||||
unsigned Shift = 32 - IR::OpSizeAsBits(IROp->Size);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Src1, ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
@@ -193,11 +193,11 @@ DEF_OP(TestNZ) {
|
||||
|
||||
DEF_OP(TestZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestZ>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size < 4, "TestNZ used at higher sizes");
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size < IR::OpSize::i32Bit, "TestNZ used at higher sizes");
|
||||
const auto EmitSize = ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
uint64_t Mask = IROp->Size == 8 ? ~0ULL : ((1ull << (IROp->Size * 8)) - 1);
|
||||
uint64_t Mask = IROp->Size == IR::OpSize::i64Bit ? ~0ULL : ((1ull << IR::OpSizeAsBits(IROp->Size)) - 1);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
@@ -223,25 +223,25 @@ DEF_OP(SubShift) {
|
||||
|
||||
DEF_OP(SubNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SubNZCV>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize >= 4, "Constant not allowed here");
|
||||
LOGMAN_THROW_AA_FMT(OpSize >= IR::OpSize::i32Bit, "Constant not allowed here");
|
||||
cmp(EmitSize, GetReg(Op->Src1.ID()), Const);
|
||||
} else {
|
||||
unsigned Shift = OpSize < 4 ? (32 - (8 * OpSize)) : 0;
|
||||
unsigned Shift = OpSize < IR::OpSize::i32Bit ? (32 - IR::OpSizeAsBits(OpSize)) : 0;
|
||||
ARMEmitter::Register ShiftedSrc1 = GetZeroableReg(Op->Src1);
|
||||
|
||||
// Shift to fix flags for <32-bit ops.
|
||||
// Any shift of zero is still zero so optimize out silly zero shifts.
|
||||
if (OpSize < 4 && ShiftedSrc1 != ARMEmitter::Reg::zr) {
|
||||
if (OpSize < IR::OpSize::i32Bit && ShiftedSrc1 != ARMEmitter::Reg::zr) {
|
||||
lsl(ARMEmitter::Size::i32Bit, TMP1, ShiftedSrc1, Shift);
|
||||
ShiftedSrc1 = TMP1;
|
||||
}
|
||||
|
||||
if (OpSize < 4) {
|
||||
if (OpSize < IR::OpSize::i32Bit) {
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
cmp(EmitSize, ShiftedSrc1, GetReg(Op->Src2.ID()));
|
||||
@@ -286,10 +286,10 @@ DEF_OP(SetSmallNZV) {
|
||||
auto Op = IROp->C<IR::IROp_SetSmallNZV>();
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM, "Unsupported flagm op");
|
||||
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 1 || OpSize == 2, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i8Bit || OpSize == IR::OpSize::i16Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
setf8(GetReg(Op->Src.ID()).W());
|
||||
} else {
|
||||
setf16(GetReg(Op->Src.ID()).W());
|
||||
@@ -401,20 +401,20 @@ DEF_OP(Div) {
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
sxtb(EmitSize, TMP1, Src1);
|
||||
sxtb(EmitSize, TMP2, Src2);
|
||||
|
||||
Src1 = TMP1;
|
||||
Src2 = TMP2;
|
||||
} else if (OpSize == 2) {
|
||||
} else if (OpSize == IR::OpSize::i16Bit) {
|
||||
sxth(EmitSize, TMP1, Src1);
|
||||
sxth(EmitSize, TMP2, Src2);
|
||||
|
||||
@@ -430,20 +430,20 @@ DEF_OP(UDiv) {
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
uxtb(EmitSize, TMP1, Src1);
|
||||
uxtb(EmitSize, TMP2, Src2);
|
||||
|
||||
Src1 = TMP1;
|
||||
Src2 = TMP2;
|
||||
} else if (OpSize == 2) {
|
||||
} else if (OpSize == IR::OpSize::i16Bit) {
|
||||
uxth(EmitSize, TMP1, Src1);
|
||||
uxth(EmitSize, TMP2, Src2);
|
||||
|
||||
@@ -458,20 +458,20 @@ DEF_OP(Rem) {
|
||||
auto Op = IROp->C<IR::IROp_Rem>();
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
sxtb(EmitSize, TMP1, Src1);
|
||||
sxtb(EmitSize, TMP2, Src2);
|
||||
|
||||
Src1 = TMP1;
|
||||
Src2 = TMP2;
|
||||
} else if (OpSize == 2) {
|
||||
} else if (OpSize == IR::OpSize::i16Bit) {
|
||||
sxth(EmitSize, TMP1, Src1);
|
||||
sxth(EmitSize, TMP2, Src2);
|
||||
|
||||
@@ -487,20 +487,20 @@ DEF_OP(URem) {
|
||||
auto Op = IROp->C<IR::IROp_URem>();
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 1) {
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
uxtb(EmitSize, TMP1, Src1);
|
||||
uxtb(EmitSize, TMP2, Src2);
|
||||
|
||||
Src1 = TMP1;
|
||||
Src2 = TMP2;
|
||||
} else if (OpSize == 2) {
|
||||
} else if (OpSize == IR::OpSize::i16Bit) {
|
||||
uxth(EmitSize, TMP1, Src1);
|
||||
uxth(EmitSize, TMP2, Src2);
|
||||
|
||||
@@ -514,15 +514,15 @@ DEF_OP(URem) {
|
||||
|
||||
DEF_OP(MulH) {
|
||||
auto Op = IROp->C<IR::IROp_MulH>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 4) {
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
sxtw(TMP1, Src1.W());
|
||||
sxtw(TMP2, Src2.W());
|
||||
mul(ARMEmitter::Size::i32Bit, Dst, TMP1, TMP2);
|
||||
@@ -534,15 +534,15 @@ DEF_OP(MulH) {
|
||||
|
||||
DEF_OP(UMulH) {
|
||||
auto Op = IROp->C<IR::IROp_UMulH>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
if (OpSize == 4) {
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
uxtw(ARMEmitter::Size::i64Bit, TMP1, Src1);
|
||||
uxtw(ARMEmitter::Size::i64Bit, TMP2, Src2);
|
||||
mul(ARMEmitter::Size::i64Bit, Dst, TMP1, TMP2);
|
||||
@@ -593,7 +593,7 @@ DEF_OP(Ornror) {
|
||||
|
||||
DEF_OP(AndWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AndWithFlags>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
uint64_t Const;
|
||||
@@ -601,7 +601,7 @@ DEF_OP(AndWithFlags) {
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
// See TestNZ
|
||||
if (OpSize < 4) {
|
||||
if (OpSize < IR::OpSize::i32Bit) {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
and_(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
@@ -614,7 +614,7 @@ DEF_OP(AndWithFlags) {
|
||||
}
|
||||
}
|
||||
|
||||
unsigned Shift = 32 - (OpSize * 8);
|
||||
unsigned Shift = 32 - IR::OpSizeAsBits(OpSize);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Dst, ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
@@ -640,7 +640,7 @@ DEF_OP(XornShift) {
|
||||
|
||||
DEF_OP(Ashr) {
|
||||
auto Op = IROp->C<IR::IROp_Ashr>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -648,29 +648,29 @@ DEF_OP(Ashr) {
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
if (OpSize >= 4) {
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
asr(EmitSize, Dst, Src1, (unsigned int)Const);
|
||||
} else {
|
||||
sbfx(EmitSize, TMP1, Src1, 0, OpSize * 8);
|
||||
sbfx(EmitSize, TMP1, Src1, 0, IR::OpSizeAsBits(OpSize));
|
||||
asr(EmitSize, Dst, TMP1, (unsigned int)Const);
|
||||
ubfx(EmitSize, Dst, Dst, 0, OpSize * 8);
|
||||
ubfx(EmitSize, Dst, Dst, 0, IR::OpSizeAsBits(OpSize));
|
||||
}
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
if (OpSize >= 4) {
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
asrv(EmitSize, Dst, Src1, Src2);
|
||||
} else {
|
||||
sbfx(EmitSize, TMP1, Src1, 0, OpSize * 8);
|
||||
sbfx(EmitSize, TMP1, Src1, 0, IR::OpSizeAsBits(OpSize));
|
||||
asrv(EmitSize, Dst, TMP1, Src2);
|
||||
ubfx(EmitSize, Dst, Dst, 0, OpSize * 8);
|
||||
ubfx(EmitSize, Dst, Dst, 0, IR::OpSizeAsBits(OpSize));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ShiftFlags) {
|
||||
auto Op = IROp->C<IR::IROp_ShiftFlags>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = Op->Size;
|
||||
const auto EmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto PFOutput = GetReg(Node);
|
||||
const auto PFInput = GetReg(Op->PFInput.ID());
|
||||
@@ -690,16 +690,16 @@ DEF_OP(ShiftFlags) {
|
||||
|
||||
// We need to mask the source before comparing it. We don't just skip flag
|
||||
// updates for Src2=0 but anything that masks to zero.
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == 8 ? 0x3f : 0x1f);
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
cbz(EmitSize, TMP1, &Done);
|
||||
{
|
||||
// PF/SF/ZF/OF
|
||||
if (OpSize >= 4) {
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
ands(EmitSize, PFTemp, Dst, Dst);
|
||||
} else {
|
||||
unsigned Shift = 32 - (OpSize * 8);
|
||||
unsigned Shift = 32 - (IR::OpSizeToSize(OpSize) * 8);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Dst, ARMEmitter::ShiftType::LSL, Shift);
|
||||
mov(ARMEmitter::Size::i64Bit, PFTemp, Dst);
|
||||
}
|
||||
@@ -709,12 +709,12 @@ DEF_OP(ShiftFlags) {
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
if (Op->Shift == IR::ShiftType::LSL) {
|
||||
if (OpSize >= 4) {
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
neg(EmitSize, CFWord, Src2);
|
||||
lsrv(EmitSize, CFWord, Src1, CFWord);
|
||||
} else {
|
||||
CFWord = Dst.X();
|
||||
CFBit = (OpSize * 8);
|
||||
CFBit = IR::OpSizeToSize(OpSize) * 8;
|
||||
}
|
||||
} else {
|
||||
sub(ARMEmitter::Size::i64Bit, CFWord, Src2, 1);
|
||||
@@ -737,7 +737,7 @@ DEF_OP(ShiftFlags) {
|
||||
rmif(CFWord, (CFBit - 1) % 64, (1 << 1) /* C */);
|
||||
|
||||
if (SetOF) {
|
||||
rmif(TMP3, OpSize * 8 - 1, (1 << 0) /* V */);
|
||||
rmif(TMP3, IR::OpSizeToSize(OpSize) * 8 - 1, (1 << 0) /* V */);
|
||||
}
|
||||
} else {
|
||||
mrs(TMP2, ARMEmitter::SystemRegister::NZCV);
|
||||
@@ -750,7 +750,7 @@ DEF_OP(ShiftFlags) {
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP2, CFWord, 29 /* C */, 1);
|
||||
|
||||
if (SetOF) {
|
||||
lsr(EmitSize, TMP3, TMP3, OpSize * 8 - 1);
|
||||
lsr(EmitSize, TMP3, TMP3, IR::OpSizeToSize(OpSize) * 8 - 1);
|
||||
bfi(ARMEmitter::Size::i32Bit, TMP2, TMP3, 28 /* V */, 1);
|
||||
}
|
||||
|
||||
@@ -770,14 +770,14 @@ DEF_OP(RotateFlags) {
|
||||
const auto Result = GetReg(Op->Result.ID());
|
||||
const auto Shift = GetReg(Op->Shift.ID());
|
||||
const bool Left = Op->Left;
|
||||
const auto EmitSize = Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::SingleUseForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto BitSize = Op->Size * 8;
|
||||
const auto BitSize = IR::OpSizeToSize(Op->Size) * 8;
|
||||
unsigned CFBit = Left ? 0 : BitSize - 1;
|
||||
|
||||
// For ROR, OF is the XOR of the new CF bit and the most significant bit of the result.
|
||||
@@ -897,7 +897,7 @@ DEF_OP(PDep) {
|
||||
DEF_OP(PExt) {
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto OpSizeBitsM1 = (OpSize * 8) - 1;
|
||||
const auto OpSizeBitsM1 = IR::OpSizeAsBits(OpSize) - 1;
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Input = GetReg(Op->Input.ID());
|
||||
@@ -952,8 +952,8 @@ DEF_OP(PExt) {
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= 4 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
@@ -963,14 +963,14 @@ DEF_OP(LDiv) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
uxth(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 16, 16);
|
||||
sxth(EmitSize, TMP2, Divisor);
|
||||
sdiv(EmitSize, Dst, TMP1, TMP2);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
// TODO: 32-bit operation should be guaranteed not to leave garbage in the upper bits.
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
@@ -978,7 +978,7 @@ DEF_OP(LDiv) {
|
||||
sdiv(EmitSize, Dst, TMP1, TMP2);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
|
||||
@@ -1022,8 +1022,8 @@ DEF_OP(LDiv) {
|
||||
|
||||
DEF_OP(LUDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= 4 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
@@ -1033,20 +1033,20 @@ DEF_OP(LUDiv) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64=
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
uxth(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 16, 16);
|
||||
udiv(EmitSize, Dst, TMP1, Divisor);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
// TODO: 32-bit operation should be guaranteed not to leave garbage in the upper bits.
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
udiv(EmitSize, Dst, TMP1, Divisor);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
|
||||
@@ -1086,8 +1086,8 @@ DEF_OP(LUDiv) {
|
||||
|
||||
DEF_OP(LRem) {
|
||||
auto Op = IROp->C<IR::IROp_LRem>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= 4 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
@@ -1097,7 +1097,7 @@ DEF_OP(LRem) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
uxth(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 16, 16);
|
||||
sxth(EmitSize, TMP2, Divisor);
|
||||
@@ -1105,7 +1105,7 @@ DEF_OP(LRem) {
|
||||
msub(EmitSize, Dst, TMP3, TMP2, TMP1);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
// TODO: 32-bit operation should be guaranteed not to leave garbage in the upper bits.
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
@@ -1114,7 +1114,7 @@ DEF_OP(LRem) {
|
||||
msub(EmitSize, Dst, TMP2, TMP3, TMP1);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
|
||||
@@ -1160,8 +1160,8 @@ DEF_OP(LRem) {
|
||||
|
||||
DEF_OP(LURem) {
|
||||
auto Op = IROp->C<IR::IROp_LURem>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= 4 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize >= IR::OpSize::i32Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
@@ -1171,14 +1171,14 @@ DEF_OP(LURem) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
case IR::OpSize::i16Bit: {
|
||||
uxth(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 16, 16);
|
||||
udiv(EmitSize, TMP2, TMP1, Divisor);
|
||||
msub(EmitSize, Dst, TMP2, Divisor, TMP1);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
case IR::OpSize::i32Bit: {
|
||||
// TODO: 32-bit operation should be guaranteed not to leave garbage in the upper bits.
|
||||
mov(EmitSize, TMP1, Lower);
|
||||
bfi(EmitSize, TMP1, Upper, 32, 32);
|
||||
@@ -1186,7 +1186,7 @@ DEF_OP(LURem) {
|
||||
msub(EmitSize, Dst, TMP2, Divisor, TMP1);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
case IR::OpSize::i64Bit: {
|
||||
ARMEmitter::SingleUseForwardLabel Only64Bit {};
|
||||
ARMEmitter::SingleUseForwardLabel LongDIVRet {};
|
||||
|
||||
@@ -1238,30 +1238,30 @@ DEF_OP(Not) {
|
||||
|
||||
DEF_OP(Popcount) {
|
||||
auto Op = IROp->C<IR::IROp_Popcount>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 0x1:
|
||||
case IR::OpSize::i8Bit:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
// only use lowest byte
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
case 0x2:
|
||||
case IR::OpSize::i16Bit:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
// only count two lowest bytes
|
||||
addp(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
case 0x4:
|
||||
case IR::OpSize::i32Bit:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
// fmov has zero extended, unused bytes are zero
|
||||
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
case 0x8:
|
||||
case IR::OpSize::i64Bit:
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), Src);
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
// fmov has zero extended, unused bytes are zero
|
||||
@@ -1280,34 +1280,27 @@ DEF_OP(FindLSB) {
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (IROp->Size != 8) {
|
||||
ubfx(EmitSize, TMP1, Src, 0, IROp->Size * 8);
|
||||
cmp(EmitSize, TMP1, 0);
|
||||
rbit(EmitSize, TMP1, TMP1);
|
||||
} else {
|
||||
rbit(EmitSize, TMP1, Src);
|
||||
cmp(EmitSize, Src, 0);
|
||||
}
|
||||
|
||||
// We assume the source is nonzero, so we can just rbit+clz without worrying
|
||||
// about upper garbage for smaller types.
|
||||
rbit(EmitSize, TMP1, Src);
|
||||
clz(EmitSize, Dst, TMP1);
|
||||
csinv(EmitSize, Dst, Dst, ARMEmitter::Reg::zr, ARMEmitter::Condition::CC_NE);
|
||||
}
|
||||
|
||||
DEF_OP(FindMSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindMSB>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
movz(ARMEmitter::Size::i64Bit, TMP1, OpSize * 8 - 1);
|
||||
movz(ARMEmitter::Size::i64Bit, TMP1, IR::OpSizeAsBits(OpSize) - 1);
|
||||
|
||||
if (OpSize == 2) {
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
lsl(EmitSize, Dst, Src, 16);
|
||||
orr(EmitSize, Dst, Dst, 0x8000);
|
||||
clz(EmitSize, Dst, Dst);
|
||||
} else {
|
||||
clz(EmitSize, Dst, Src);
|
||||
@@ -1318,9 +1311,10 @@ DEF_OP(FindMSB) {
|
||||
|
||||
DEF_OP(FindTrailingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_FindTrailingZeroes>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -1328,7 +1322,7 @@ DEF_OP(FindTrailingZeroes) {
|
||||
|
||||
rbit(EmitSize, Dst, Src);
|
||||
|
||||
if (OpSize == 2) {
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
// This orr does two things. First, if the (masked) source is zero, it
|
||||
// reverses to zero in the top so it forces clz to return 16. Second, it
|
||||
// ensures garbage in the upper bits of the source don't affect clz, because
|
||||
@@ -1342,15 +1336,16 @@ DEF_OP(FindTrailingZeroes) {
|
||||
|
||||
DEF_OP(CountLeadingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_CountLeadingZeroes>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (OpSize == 2) {
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
// Expressing as lsl+orr+clz clears away any garbage in the upper bits
|
||||
// (alternatively could do uxth+clz+sub.. equal cost in total).
|
||||
lsl(EmitSize, Dst, Src, 16);
|
||||
@@ -1363,16 +1358,17 @@ DEF_OP(CountLeadingZeroes) {
|
||||
|
||||
DEF_OP(Rev) {
|
||||
auto Op = IROp->C<IR::IROp_Rev>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit,
|
||||
"Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
rev(EmitSize, Dst, Src);
|
||||
if (OpSize == 2) {
|
||||
if (OpSize == IR::OpSize::i16Bit) {
|
||||
lsr(EmitSize, Dst, Dst, 16);
|
||||
}
|
||||
}
|
||||
@@ -1398,10 +1394,10 @@ DEF_OP(Bfi) {
|
||||
mov(EmitSize, TMP1, SrcDst);
|
||||
bfi(EmitSize, TMP1, Src, Op->lsb, Op->Width);
|
||||
|
||||
if (IROp->Size >= 4) {
|
||||
if (IROp->Size >= IR::OpSize::i32Bit) {
|
||||
mov(EmitSize, Dst, TMP1.R());
|
||||
} else {
|
||||
ubfx(EmitSize, Dst, TMP1, 0, IROp->Size * 8);
|
||||
ubfx(EmitSize, Dst, TMP1, 0, IR::OpSizeAsBits(IROp->Size));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1432,7 +1428,7 @@ DEF_OP(Bfxil) {
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= IR::OpSize::i64Bit, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(Op->Width != 0, "Invalid BFE width of 0");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
@@ -1442,7 +1438,7 @@ DEF_OP(Bfe) {
|
||||
if (Op->lsb == 0 && Op->Width == 32) {
|
||||
mov(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
} else if (Op->lsb == 0 && Op->Width == 64) {
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size == 8, "Must be 64-bit wide register");
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size == IR::OpSize::i64Bit, "Must be 64-bit wide register");
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, Src);
|
||||
} else {
|
||||
ubfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
@@ -1459,9 +1455,9 @@ DEF_OP(Sbfe) {
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto CompareEmitSize = Op->CompareSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto CompareEmitSize = Op->CompareSize == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapCC(Op->Cond);
|
||||
@@ -1478,7 +1474,7 @@ DEF_OP(Select) {
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1.ID());
|
||||
const auto Src2 = GetVReg(Op->Cmp2.ID());
|
||||
fcmp(Op->CompareSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit, Src1, Src2);
|
||||
fcmp(Op->CompareSize == IR::OpSize::i64Bit ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit, Src1, Src2);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Select: Expected GPR or FPR");
|
||||
}
|
||||
@@ -1487,7 +1483,7 @@ DEF_OP(Select) {
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
uint64_t all_ones = OpSize == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
uint64_t all_ones = OpSize == IR::OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
|
||||
@@ -1516,7 +1512,7 @@ DEF_OP(NZCVSelect) {
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
uint64_t all_ones = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
uint64_t all_ones = IROp->Size == IR::OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
|
||||
@@ -1535,6 +1531,14 @@ DEF_OP(NZCVSelect) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelectV) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelectV>();
|
||||
|
||||
auto cc = MapCC(Op->Cond);
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
fcsel(SubRegSize.Scalar, GetVReg(Node), GetVReg(Op->TrueVal.ID()), GetVReg(Op->FalseVal.ID()), cc);
|
||||
}
|
||||
|
||||
DEF_OP(NZCVSelectIncrement) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelectIncrement>();
|
||||
|
||||
@@ -1547,7 +1551,7 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto ElementSizeBits = Op->Header.ElementSize * 8;
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(Op->Header.ElementSize);
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
@@ -1558,10 +1562,10 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
const auto PerformMove = [&](const ARMEmitter::VRegister reg, int index) {
|
||||
switch (OpSize) {
|
||||
case 1: umov<ARMEmitter::SubRegSize::i8Bit>(Dst, Vector, index); break;
|
||||
case 2: umov<ARMEmitter::SubRegSize::i16Bit>(Dst, Vector, index); break;
|
||||
case 4: umov<ARMEmitter::SubRegSize::i32Bit>(Dst, Vector, index); break;
|
||||
case 8: umov<ARMEmitter::SubRegSize::i64Bit>(Dst, Vector, index); break;
|
||||
case IR::OpSize::i8Bit: umov<ARMEmitter::SubRegSize::i8Bit>(Dst, Vector, index); break;
|
||||
case IR::OpSize::i16Bit: umov<ARMEmitter::SubRegSize::i16Bit>(Dst, Vector, index); break;
|
||||
case IR::OpSize::i32Bit: umov<ARMEmitter::SubRegSize::i32Bit>(Dst, Vector, index); break;
|
||||
case IR::OpSize::i64Bit: umov<ARMEmitter::SubRegSize::i64Bit>(Dst, Vector, index); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize); break;
|
||||
}
|
||||
};
|
||||
@@ -1586,10 +1590,10 @@ DEF_OP(VExtractToGPR) {
|
||||
// upper half of the vector.
|
||||
const auto SanitizedIndex = [OpSize, Op] {
|
||||
switch (OpSize) {
|
||||
case 1: return Op->Index - 16;
|
||||
case 2: return Op->Index - 8;
|
||||
case 4: return Op->Index - 4;
|
||||
case 8: return Op->Index - 2;
|
||||
case IR::OpSize::i8Bit: return Op->Index - 16;
|
||||
case IR::OpSize::i16Bit: return Op->Index - 8;
|
||||
case IR::OpSize::i32Bit: return Op->Index - 4;
|
||||
case IR::OpSize::i64Bit: return Op->Index - 2;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OpSize: {}", OpSize); return 0;
|
||||
}
|
||||
}();
|
||||
@@ -1605,7 +1609,7 @@ DEF_OP(Float_ToGPR_ZS) {
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar.ID());
|
||||
|
||||
if (Op->SrcElementSize == 8) {
|
||||
if (Op->SrcElementSize == IR::OpSize::i64Bit) {
|
||||
fcvtzs(ConvertSize(IROp), Dst, Src.D());
|
||||
} else {
|
||||
fcvtzs(ConvertSize(IROp), Dst, Src.S());
|
||||
@@ -1618,7 +1622,7 @@ DEF_OP(Float_ToGPR_S) {
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar.ID());
|
||||
|
||||
if (Op->SrcElementSize == 8) {
|
||||
if (Op->SrcElementSize == IR::OpSize::i64Bit) {
|
||||
frinti(VTMP1.D(), Src.D());
|
||||
fcvtzs(ConvertSize(IROp), Dst, VTMP1.D());
|
||||
} else {
|
||||
@@ -1629,7 +1633,7 @@ DEF_OP(Float_ToGPR_S) {
|
||||
|
||||
DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
const auto EmitSubSize = Op->ElementSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit;
|
||||
const auto EmitSubSize = Op->ElementSize == IR::OpSize::i64Bit ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit;
|
||||
|
||||
ARMEmitter::VRegister Scalar1 = GetVReg(Op->Scalar1.ID());
|
||||
ARMEmitter::VRegister Scalar2 = GetVReg(Op->Scalar2.ID());
|
||||
+1
-1
@@ -6,7 +6,7 @@ desc: relocation logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
|
||||
+16
-13
@@ -7,13 +7,13 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == 4 || IROp->ElementSize == 8, "Wrong element size");
|
||||
LOGMAN_THROW_AA_FMT(IROp->ElementSize == IR::OpSize::i32Bit || IROp->ElementSize == IR::OpSize::i64Bit, "Wrong element size");
|
||||
// Size is the size of each pair element
|
||||
auto Dst0 = GetReg(Op->OutLo.ID());
|
||||
auto Dst1 = GetReg(Op->OutHi.ID());
|
||||
@@ -23,7 +23,7 @@ DEF_OP(CASPair) {
|
||||
auto Desired1 = GetReg(Op->DesiredHi.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = IROp->ElementSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = IROp->ElementSize == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
// RA has heuristics to try to pair sources, but we need to handle the cases
|
||||
// where they fail. We do so by moving to temporaries. Note we use 64-bit
|
||||
@@ -112,9 +112,9 @@ DEF_OP(CAS) {
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (IROp->Size == 1) {
|
||||
if (IROp->Size == IR::OpSize::i8Bit) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
|
||||
} else if (IROp->Size == 2) {
|
||||
} else if (IROp->Size == IR::OpSize::i16Bit) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTH, 0);
|
||||
} else {
|
||||
cmp(EmitSize, TMP2, Expected);
|
||||
@@ -273,18 +273,21 @@ DEF_OP(AtomicNeg) {
|
||||
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(
|
||||
OpSize == IR::OpSize::i64Bit || OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i16Bit || OpSize == IR::OpSize::i8Bit, "Unexpecte"
|
||||
"d CAS "
|
||||
"size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto SubEmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -294,7 +297,7 @@ DEF_OP(AtomicSwap) {
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
stlxr(SubEmitSize, TMP4, Src, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
ubfm(EmitSize, GetReg(Node), TMP2, 0, OpSize * 8 - 1);
|
||||
ubfm(EmitSize, GetReg(Node), TMP2, 0, IR::OpSizeAsBits(OpSize) - 1);
|
||||
}
|
||||
}
|
||||
|
||||
+2
-2
@@ -9,7 +9,7 @@ $end_info$
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
@@ -117,7 +117,7 @@ DEF_OP(CondJump) {
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
auto Reg = GetReg(Op->Cmp1.ID());
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
const auto Size = Op->CompareSize == IR::OpSize::i32Bit ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
+35
-35
@@ -5,7 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -15,18 +15,18 @@ DEF_OP(VInsGPR) {
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementsPer128Bit = 16 / ElementSize;
|
||||
const auto ElementsPer128Bit = IR::NumElements(IR::OpSize::i128Bit, ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(ElementSize);
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
const auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
@@ -90,16 +90,16 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
case IR::OpSize::i8Bit:
|
||||
uxtb(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 2:
|
||||
case IR::OpSize::i16Bit:
|
||||
uxth(ARMEmitter::Size::i32Bit, TMP1, Src);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1);
|
||||
break;
|
||||
case 4: fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src); break;
|
||||
case 8: fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src); break;
|
||||
case IR::OpSize::i32Bit: fmov(ARMEmitter::Size::i32Bit, Dst.S(), Src); break;
|
||||
case IR::OpSize::i64Bit: fmov(ARMEmitter::Size::i64Bit, Dst.D(), Src); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
}
|
||||
@@ -111,7 +111,7 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
@@ -126,8 +126,8 @@ DEF_OP(VDupFromGPR) {
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t ElementSize = IR::OpSizeToSize(Op->Header.ElementSize);
|
||||
const uint16_t Conv = (ElementSize << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
@@ -165,7 +165,7 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (IR::OpSizeToSize(Op->Header.ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetVReg(Op->Scalar.ID());
|
||||
@@ -205,7 +205,7 @@ DEF_OP(Vector_SToF) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -215,15 +215,15 @@ DEF_OP(Vector_SToF) {
|
||||
scvtf(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
} else {
|
||||
if (OpSize == ElementSize) {
|
||||
if (ElementSize == 8) {
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
|
||||
} else if (ElementSize == 4) {
|
||||
} else if (ElementSize == IR::OpSize::i32Bit) {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
|
||||
} else {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
|
||||
}
|
||||
} else {
|
||||
if (OpSize == 8) {
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
scvtf(SubEmitSize, Dst.D(), Vector.D());
|
||||
} else {
|
||||
scvtf(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
@@ -238,7 +238,7 @@ DEF_OP(Vector_FToZS) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -248,15 +248,15 @@ DEF_OP(Vector_FToZS) {
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
} else {
|
||||
if (OpSize == ElementSize) {
|
||||
if (ElementSize == 8) {
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
fcvtzs(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Vector.D());
|
||||
} else if (ElementSize == 4) {
|
||||
} else if (ElementSize == IR::OpSize::i32Bit) {
|
||||
fcvtzs(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Vector.S());
|
||||
} else {
|
||||
fcvtzs(ARMEmitter::ScalarRegSize::i16Bit, Dst.H(), Vector.H());
|
||||
}
|
||||
} else {
|
||||
if (OpSize == 8) {
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
fcvtzs(SubEmitSize, Dst.D(), Vector.D());
|
||||
} else {
|
||||
fcvtzs(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
@@ -269,7 +269,7 @@ DEF_OP(Vector_FToS) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
@@ -284,7 +284,7 @@ DEF_OP(Vector_FToS) {
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
if (OpSize == 8) {
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
frinti(SubEmitSize, Dst.D(), Vector.D());
|
||||
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
|
||||
} else {
|
||||
@@ -300,10 +300,10 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Conv = (IR::OpSizeToSize(ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -403,7 +403,7 @@ DEF_OP(Vector_FToI) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -427,15 +427,15 @@ DEF_OP(Vector_FToI) {
|
||||
// frinti having AdvSIMD, AdvSIMD scalar, and an SVE version),
|
||||
// we can't just use a lambda without some seriously ugly casting.
|
||||
// This is fairly self-contained otherwise.
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == 2) { \
|
||||
name(Dst.H(), Vector.H()); \
|
||||
} else if (ElementSize == 4) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == 8) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == IR::OpSize::i16Bit) { \
|
||||
name(Dst.H(), Vector.H()); \
|
||||
} else if (ElementSize == IR::OpSize::i32Bit) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == IR::OpSize::i64Bit) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
}
|
||||
|
||||
switch (Op->Round) {
|
||||
@@ -464,7 +464,7 @@ DEF_OP(Vector_F64ToI32) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Round = Op->Round;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
+10
-10
@@ -5,7 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -24,7 +24,7 @@ DEF_OP(VAESEnc) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -49,7 +49,7 @@ DEF_OP(VAESEncLast) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -72,7 +72,7 @@ DEF_OP(VAESDec) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -97,7 +97,7 @@ DEF_OP(VAESDecLast) {
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
if (Dst == State && Dst != Key) {
|
||||
// Optimal case in which Dst already contains the starting state.
|
||||
@@ -152,10 +152,10 @@ DEF_OP(CRC32) {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1: crc32cb(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case 2: crc32ch(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case 4: crc32cw(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case 8: crc32cx(Dst.X(), Src1.X(), Src2.X()); break;
|
||||
case IR::OpSize::i8Bit: crc32cb(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case IR::OpSize::i16Bit: crc32ch(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case IR::OpSize::i32Bit: crc32cw(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
case IR::OpSize::i64Bit: crc32cx(Dst.X(), Src1.X(), Src2.X()); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
@@ -193,7 +193,7 @@ DEF_OP(PCLMUL) {
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE, "Currently only supports 128-bit operations.");
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
|
||||
+3
-3
@@ -16,7 +16,7 @@ $end_info$
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
@@ -626,8 +626,8 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = OpHeader->Size;
|
||||
if (OpSize == 4) {
|
||||
const auto Size = OpHeader->Size;
|
||||
if (Size == IR::OpSize::i32Bit) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
*Value = (Entry + Op->Offset) & Mask;
|
||||
+29
-23
@@ -129,23 +129,25 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
return Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
return Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->Size == 4 || Op->Size == 8, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSize(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(uint8_t ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
|
||||
return ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ARMEmitter::SubRegSize::i128Bit;
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
return ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == IR::OpSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ARMEmitter::SubRegSize::i128Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -154,8 +156,8 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(uint8_t ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != 16, "Invalid size");
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSize16(ElementSize);
|
||||
}
|
||||
|
||||
@@ -166,13 +168,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 8, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 1, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
@@ -183,13 +185,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 16, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair16(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 1, "Invalid size");
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair8(Op);
|
||||
}
|
||||
|
||||
@@ -226,7 +228,7 @@ private:
|
||||
bool IsGPR(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
@@ -235,7 +237,7 @@ private:
|
||||
// TMP1 is safe to use again once this memory operand is used with its
|
||||
// equivalent loads or stores that this was called for.
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
ARMEmitter::SVEMemOperand GenerateSVEMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -316,19 +318,20 @@ private:
|
||||
|
||||
using ScalarFMAOpCaller =
|
||||
std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2, ARMEmitter::VRegister Src3)>;
|
||||
void VFScalarFMAOperation(uint8_t OpSize, uint8_t ElementSize, ScalarFMAOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
void VFScalarFMAOperation(IR::OpSize OpSize, IR::OpSize ElementSize, ScalarFMAOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Upper, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2,
|
||||
ARMEmitter::VRegister Addend);
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
void VFScalarOperation(IR::OpSize OpSize, IR::OpSize ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
|
||||
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst,
|
||||
ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
void VFScalarUnaryOperation(IR::OpSize OpSize, IR::OpSize ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit,
|
||||
ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1,
|
||||
std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
|
||||
void Emulate128BitGather(size_t Size, size_t ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
|
||||
void Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
|
||||
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, size_t VectorIndexSize,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
// Runtime selection;
|
||||
// Load and store TSO memory style
|
||||
@@ -352,4 +355,7 @@ private:
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
[[nodiscard]]
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,21 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
[[nodiscard]]
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+298
-289
File diff suppressed because it is too large.
Load diff
+13
-8
@@ -10,7 +10,7 @@ $end_info$
|
||||
#endif
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
@@ -192,8 +192,17 @@ DEF_OP(Print) {
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
DEF_OP(ProcessorID) {
|
||||
if (CTX->HostFeatures.SupportsCPUIndexInTPIDRRO) {
|
||||
mrs(GetReg(Node), ARMEmitter::SystemRegister::TPIDRRO_EL0);
|
||||
return;
|
||||
}
|
||||
#ifdef _WIN32
|
||||
else {
|
||||
// If on Windows and TPIDRRO isn't supported (like in wine), then this is a programming error.
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#else
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
|
||||
@@ -248,12 +257,8 @@ DEF_OP(ProcessorID) {
|
||||
// CPU is in w0
|
||||
// Node is in w1
|
||||
orr(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::ShiftType::LSL, 12);
|
||||
}
|
||||
#else
|
||||
DEF_OP(ProcessorID) {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
@@ -262,7 +267,7 @@ DEF_OP(RDRAND) {
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
yield();
|
||||
wfe();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
+1
-1
@@ -5,7 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
+304
-299
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -43,19 +43,19 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref NewVec = _VExtr(16, 8, Dest, Src, 1);
|
||||
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
|
||||
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
Ref Result = _VXor(16, 1, Dest, NewVec);
|
||||
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
@@ -86,7 +86,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
@@ -121,25 +121,26 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(32, k_array[Imm8]);
|
||||
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
|
||||
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
|
||||
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto B = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto C = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto D = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
|
||||
auto A1 =
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), W0E), K);
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto D1 = C;
|
||||
auto E1 = D;
|
||||
|
||||
@@ -147,13 +148,14 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
};
|
||||
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(16, 4, Src, W_idx);
|
||||
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
|
||||
auto ANext =
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(32, 27))), Q), K);
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(32, 2));
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto DNext = C;
|
||||
auto ENext = D;
|
||||
|
||||
@@ -165,12 +167,12 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
|
||||
auto Dest3 = _VInsGPR(16, 4, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(16, 4, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(16, 4, 1, Dest2, std::get<2>(Final));
|
||||
auto Dest0 = _VInsGPR(16, 4, 0, Dest1, std::get<3>(Final));
|
||||
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
|
||||
auto Dest0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
|
||||
|
||||
StoreResult(FPRClass, Op, Dest0, -1);
|
||||
StoreResult(FPRClass, Op, Dest0, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
@@ -183,52 +185,56 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
Result = _VSha256U0(Dest, Src);
|
||||
} else {
|
||||
const auto Sigma0 = [this](Ref W) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
return _Xor(
|
||||
OpSize::i32Bit,
|
||||
_Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 7)), _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 18))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 3)));
|
||||
};
|
||||
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto W4 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto W3 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto W2 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto W1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto W0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
|
||||
auto Sig3 = _Add(OpSize::i32Bit, W3, Sigma0(W4));
|
||||
auto Sig2 = _Add(OpSize::i32Bit, W2, Sigma0(W3));
|
||||
auto Sig1 = _Add(OpSize::i32Bit, W1, Sigma0(W2));
|
||||
auto Sig0 = _Add(OpSize::i32Bit, W0, Sigma0(W1));
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, Sig1);
|
||||
Result = _VInsGPR(16, 4, 0, D1, Sig0);
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, Sig3);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, Sig2);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, Sig1);
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, Sig0);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
const auto Sigma1 = [this](Ref W) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
|
||||
return _Xor(
|
||||
OpSize::i32Bit,
|
||||
_Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 17)), _Ror(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 19))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(OpSize::i32Bit, 10)));
|
||||
};
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), Sigma1(W17));
|
||||
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
|
||||
|
||||
auto D3 = _VInsGPR(16, 4, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(16, 4, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(16, 4, 1, D2, W17);
|
||||
auto D0 = _VInsGPR(16, 4, 0, D1, W16);
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
|
||||
auto D0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
|
||||
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
StoreResult(FPRClass, Op, D0, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
@@ -246,12 +252,12 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](Ref A) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), A, ShiftType::ROR, 13), A,
|
||||
ShiftType::ROR, 22);
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
|
||||
A, ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](Ref E) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), E, ShiftType::ROR, 11), E,
|
||||
ShiftType::ROR, 25);
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
|
||||
E, ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
@@ -259,64 +265,64 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto E0 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto F0 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
|
||||
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
|
||||
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
|
||||
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
|
||||
auto H0 = _VExtractToGPR(16, 4, Dest, 0);
|
||||
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, H0);
|
||||
|
||||
auto A0 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto B0 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto C0 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
|
||||
|
||||
auto D0 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
|
||||
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
|
||||
auto WK1 = _VExtractToGPR(16, 4, XMM0, 1);
|
||||
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
|
||||
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
|
||||
G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, G0);
|
||||
|
||||
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
|
||||
|
||||
// Rematerialize C0. As with G0.
|
||||
C0 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
|
||||
|
||||
auto Res3 = _VInsGPR(16, 4, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(16, 4, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(16, 4, 1, Res2, E2);
|
||||
auto Res0 = _VInsGPR(16, 4, 0, Res1, E1);
|
||||
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
|
||||
auto Res0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
|
||||
|
||||
StoreResult(FPRClass, Op, Res0, -1);
|
||||
StoreResult(FPRClass, Op, Res0, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEnc(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
@@ -325,19 +331,19 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEncLast(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
@@ -346,19 +352,19 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDec(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
@@ -367,19 +373,19 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDecLast(16, Dest, Src, LoadZeroVector(16));
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
@@ -388,20 +394,20 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const uint64_t RCON = Op->Src[1].Literal();
|
||||
|
||||
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(16, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
|
||||
return _VAESKeyGenAssist(Src, KeyGenSwizzle, LoadZeroVector(16), RCON);
|
||||
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
|
||||
return _VAESKeyGenAssist(Src, KeyGenSwizzle, LoadZeroVector(OpSize::i128Bit), RCON);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
Ref Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -409,19 +415,19 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
|
||||
|
||||
auto Res = _PCLMUL(16, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
|
||||
|
||||
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -5,41 +5,41 @@
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_DDDTable[] = {
|
||||
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
|
||||
{0x0D, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<4, false>},
|
||||
{0x0D, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, 4>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, 4>},
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
|
||||
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
|
||||
|
||||
{0x90, 1, &OpDispatchBuilder::VPFCMPOp<1>},
|
||||
{0x94, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, 4>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, 4>},
|
||||
{0x97, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, 4>},
|
||||
{0x94, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x97, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
|
||||
{0x9A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, 4>},
|
||||
{0x9E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, 4>},
|
||||
{0x9A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0x9E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
|
||||
{0xA0, 1, &OpDispatchBuilder::VPFCMPOp<2>},
|
||||
{0xA4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, 4>},
|
||||
{0xA4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xA7, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
|
||||
{0xAA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VFSUB, 4>},
|
||||
{0xAE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, 4>},
|
||||
{0xAA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0xAE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
|
||||
{0xB0, 1, &OpDispatchBuilder::VPFCMPOp<0>},
|
||||
{0xB4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, 4>},
|
||||
{0xB4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0xB7, 1, &OpDispatchBuilder::PMULHRWOp},
|
||||
|
||||
{0xBB, 1, &OpDispatchBuilder::PSWAPDOp},
|
||||
{0xBF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 1>},
|
||||
{0xBF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i8Bit>},
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -36,13 +36,9 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, Ref Src) {
|
||||
size_t NumFlags = FlagOffsets.size();
|
||||
if (Lower8) {
|
||||
// Calculate flags early.
|
||||
// Could use InvalidateDeferredFlags() if we had masked invalidation.
|
||||
// This is only a partial overwrite of flags since OF isn't stored here.
|
||||
CalculateDeferredFlags();
|
||||
NumFlags = 5;
|
||||
} else {
|
||||
// We are overwriting all RFLAGS. Invalidate the deferred flag state.
|
||||
InvalidateDeferredFlags();
|
||||
}
|
||||
|
||||
// PF and CF are both stored inverted, so hoist the invert.
|
||||
@@ -138,9 +134,9 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
return Original;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub) {
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
uint64_t SignBit = (SrcSize * 8) - 1;
|
||||
void OpDispatchBuilder::CalculateOF(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub) {
|
||||
const auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
const uint64_t SignBit = IR::OpSizeAsBits(SrcSize) - 1;
|
||||
Ref Anded = nullptr;
|
||||
|
||||
// For add, OF is set iff the sources have the same sign but the destination
|
||||
@@ -171,7 +167,7 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2
|
||||
}
|
||||
}
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SignBit, true);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadPFRaw(bool Mask, bool Invert) {
|
||||
@@ -262,22 +258,22 @@ Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
|
||||
return _NZCVSelectIncrement(OpSize, {CFInverted ? COND_UGE : COND_ULT}, Src, Src);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
Ref Res;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
if (SrcSize >= 4) {
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
RectifyCarryInvert(false);
|
||||
HandleNZCV_RMW();
|
||||
Res = _AdcWithFlags(OpSize, Src1, Src2);
|
||||
CFInverted = false;
|
||||
} else {
|
||||
// Need to zero-extend for correct comparisons below
|
||||
Src2 = _Bfe(OpSize, SrcSize * 8, 0, Src2);
|
||||
Src2 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src2);
|
||||
|
||||
// Note that we do not extend Src2PlusCF, since we depend on proper
|
||||
// 32-bit arithmetic to correctly handle the Src2 = 0xffff case.
|
||||
@@ -285,7 +281,7 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
|
||||
// Need to zero-extend for the comparison.
|
||||
Res = _Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Res, Src2PlusCF, One, Zero);
|
||||
@@ -299,15 +295,15 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
// Arm's subtraction has inverted CF from x86, so rectify the input and
|
||||
// invert the output.
|
||||
RectifyCarryInvert(true);
|
||||
@@ -316,13 +312,13 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
CFInverted = true;
|
||||
} else {
|
||||
// Zero extend for correct comparison behaviour with Src1 = 0xffff.
|
||||
Src1 = _Bfe(OpSize, SrcSize * 8, 0, Src1);
|
||||
Src2 = _Bfe(OpSize, SrcSize * 8, 0, Src2);
|
||||
Src1 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src1);
|
||||
Src2 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src2);
|
||||
|
||||
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
Res = _Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, SrcSize * 8, 0, Res);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Src1, Src2PlusCF, One, Zero);
|
||||
|
||||
@@ -335,7 +331,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_SUB(IR::OpSize SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCFInv = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
|
||||
@@ -344,10 +340,10 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
Res = _SubWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _SubWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_SubNZCV(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
_SubNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Sub(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
@@ -365,7 +361,7 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
return Res;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADD(IR::OpSize SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCFInv = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
|
||||
@@ -374,10 +370,10 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
Res = _AddWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _AddWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_AddNZCV(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
_AddNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Add(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
@@ -394,13 +390,13 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, b
|
||||
return Res;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, Ref Res, Ref High) {
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High) {
|
||||
HandleNZCVWrite();
|
||||
InvalidatePF_AF();
|
||||
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// If the value can fit then the top bits will be zero
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, IR::OpSizeAsBits(SrcSize) - 1, Res);
|
||||
_SubNZCV(OpSize::i64Bit, High, SignBit);
|
||||
|
||||
// If High = SignBit, then sets to nZCv. Else sets to nzcV. Since SF/ZF
|
||||
@@ -415,7 +411,7 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
InvalidatePF_AF();
|
||||
|
||||
auto Zero = _InlineConstant(0);
|
||||
OpSize Size = IR::SizeToOpSize(GetOpSize(High));
|
||||
const auto Size = GetOpSize(High);
|
||||
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// The result register will be all zero if it can't fit due to how multiplication behaves
|
||||
@@ -427,7 +423,7 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
InvalidateAF();
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -436,13 +432,13 @@ void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, UnmaskedRes);
|
||||
|
||||
@@ -451,7 +447,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref U
|
||||
// Extract the last bit shifted in to CF. Shift is already masked, but for
|
||||
// 8/16-bit it might be >= SrcSizeBits, in which case CF is cleared. There's
|
||||
// nothing to do in that case since we already cleared CF above.
|
||||
auto SrcSizeBits = SrcSize * 8;
|
||||
const auto SrcSizeBits = IR::OpSizeAsBits(SrcSize);
|
||||
if (Shift < SrcSizeBits) {
|
||||
SetCFDirect(Src1, SrcSizeBits - Shift, true);
|
||||
}
|
||||
@@ -464,13 +460,13 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref U
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
if (Shift == 1) {
|
||||
auto Xor = _Xor(OpSize, UnmaskedRes, Src1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, SrcSize * 8 - 1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, IR::OpSizeAsBits(SrcSize) - 1, true);
|
||||
} else {
|
||||
// Undefined, we choose to zero as part of SetNZ_ZeroCV
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -490,7 +486,7 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
// already zeroed there's nothing to do here.
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// Set SF and PF. Clobbers OF, but OF only defined for Shift = 1 where it is
|
||||
// set below.
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
@@ -502,7 +498,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
InvalidateAF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -515,18 +511,18 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Ref
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// Is set to the MSB of the original value
|
||||
if (Shift == 1) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Src1, SrcSize * 8 - 1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Src1, IR::OpSizeAsBits(SrcSize) - 1, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
const auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
CalculateFlags_ShiftRightImmediateCommon(SrcSize, Res, Src1, Shift);
|
||||
|
||||
// OF
|
||||
@@ -536,12 +532,12 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
|
||||
// XOR of Result and Src1
|
||||
if (Shift == 1) {
|
||||
auto val = _Xor(OpSize, Src1, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val, SrcSize * 8 - 1, true);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val, IR::OpSizeAsBits(SrcSize) - 1, true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result) {
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(IR::OpSize SrcSize, Ref Result) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
// Test ZF of result, SF is undefined so this is ok.
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
@@ -549,7 +545,7 @@ void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result) {
|
||||
// Now set CF if the Result = SrcSize * 8. Since SrcSize is a power-of-two and
|
||||
// Result is <= SrcSize * 8, we equivalently check if the log2(SrcSize * 8)
|
||||
// bit is set. No masking is needed because no higher bits could be set.
|
||||
unsigned CarryBit = FEXCore::ilog2(SrcSize * 8u);
|
||||
unsigned CarryBit = FEXCore::ilog2(IR::OpSizeAsBits(SrcSize));
|
||||
SetCFDirect(Result, CarryBit);
|
||||
}
|
||||
|
||||
|
||||
@@ -11,64 +11,64 @@ constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F38Table[] = {
|
||||
{OPD(PF_38_NONE, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
|
||||
{OPD(PF_38_66, 0x00), 1, &OpDispatchBuilder::PSHUFBOp},
|
||||
{OPD(PF_38_NONE, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, 2>},
|
||||
{OPD(PF_38_66, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, 2>},
|
||||
{OPD(PF_38_NONE, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, 4>},
|
||||
{OPD(PF_38_66, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, 4>},
|
||||
{OPD(PF_38_NONE, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADDP, OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x03), 1, &OpDispatchBuilder::PHADDS},
|
||||
{OPD(PF_38_66, 0x03), 1, &OpDispatchBuilder::PHADDS},
|
||||
{OPD(PF_38_NONE, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
|
||||
{OPD(PF_38_66, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
|
||||
{OPD(PF_38_NONE, 0x05), 1, &OpDispatchBuilder::PHSUB<2>},
|
||||
{OPD(PF_38_66, 0x05), 1, &OpDispatchBuilder::PHSUB<2>},
|
||||
{OPD(PF_38_NONE, 0x06), 1, &OpDispatchBuilder::PHSUB<4>},
|
||||
{OPD(PF_38_66, 0x06), 1, &OpDispatchBuilder::PHSUB<4>},
|
||||
{OPD(PF_38_NONE, 0x05), 1, &OpDispatchBuilder::PHSUB<OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x05), 1, &OpDispatchBuilder::PHSUB<OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x06), 1, &OpDispatchBuilder::PHSUB<OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x06), 1, &OpDispatchBuilder::PHSUB<OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x07), 1, &OpDispatchBuilder::PHSUBS},
|
||||
{OPD(PF_38_66, 0x07), 1, &OpDispatchBuilder::PHSUBS},
|
||||
{OPD(PF_38_NONE, 0x08), 1, &OpDispatchBuilder::PSIGN<1>},
|
||||
{OPD(PF_38_66, 0x08), 1, &OpDispatchBuilder::PSIGN<1>},
|
||||
{OPD(PF_38_NONE, 0x09), 1, &OpDispatchBuilder::PSIGN<2>},
|
||||
{OPD(PF_38_66, 0x09), 1, &OpDispatchBuilder::PSIGN<2>},
|
||||
{OPD(PF_38_NONE, 0x0A), 1, &OpDispatchBuilder::PSIGN<4>},
|
||||
{OPD(PF_38_66, 0x0A), 1, &OpDispatchBuilder::PSIGN<4>},
|
||||
{OPD(PF_38_NONE, 0x08), 1, &OpDispatchBuilder::PSIGN<OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x08), 1, &OpDispatchBuilder::PSIGN<OpSize::i8Bit>},
|
||||
{OPD(PF_38_NONE, 0x09), 1, &OpDispatchBuilder::PSIGN<OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x09), 1, &OpDispatchBuilder::PSIGN<OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x0A), 1, &OpDispatchBuilder::PSIGN<OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x0A), 1, &OpDispatchBuilder::PSIGN<OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
|
||||
{OPD(PF_38_66, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
|
||||
{OPD(PF_38_66, 0x10), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, 1>},
|
||||
{OPD(PF_38_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, 4>},
|
||||
{OPD(PF_38_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, 8>},
|
||||
{OPD(PF_38_66, 0x10), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, OpSize::i64Bit>},
|
||||
{OPD(PF_38_66, 0x17), 1, &OpDispatchBuilder::PTestOp},
|
||||
{OPD(PF_38_NONE, 0x1C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 1>},
|
||||
{OPD(PF_38_66, 0x1C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 1>},
|
||||
{OPD(PF_38_NONE, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 2>},
|
||||
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 2>},
|
||||
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 4>},
|
||||
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, 4>},
|
||||
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, true>},
|
||||
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, true>},
|
||||
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<1, 8, true>},
|
||||
{OPD(PF_38_66, 0x23), 1, &OpDispatchBuilder::ExtendVectorElements<2, 4, true>},
|
||||
{OPD(PF_38_66, 0x24), 1, &OpDispatchBuilder::ExtendVectorElements<2, 8, true>},
|
||||
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<4, 8, true>},
|
||||
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::PMULLOp<4, true>},
|
||||
{OPD(PF_38_66, 0x29), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 8>},
|
||||
{OPD(PF_38_NONE, 0x1C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x1C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i8Bit>},
|
||||
{OPD(PF_38_NONE, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, true>},
|
||||
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x23), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x24), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x29), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i64Bit>},
|
||||
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::PACKUSOp<4>},
|
||||
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, false>},
|
||||
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, false>},
|
||||
{OPD(PF_38_66, 0x32), 1, &OpDispatchBuilder::ExtendVectorElements<1, 8, false>},
|
||||
{OPD(PF_38_66, 0x33), 1, &OpDispatchBuilder::ExtendVectorElements<2, 4, false>},
|
||||
{OPD(PF_38_66, 0x34), 1, &OpDispatchBuilder::ExtendVectorElements<2, 8, false>},
|
||||
{OPD(PF_38_66, 0x35), 1, &OpDispatchBuilder::ExtendVectorElements<4, 8, false>},
|
||||
{OPD(PF_38_66, 0x37), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 8>},
|
||||
{OPD(PF_38_66, 0x38), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, 1>},
|
||||
{OPD(PF_38_66, 0x39), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, 4>},
|
||||
{OPD(PF_38_66, 0x3A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, 2>},
|
||||
{OPD(PF_38_66, 0x3B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, 4>},
|
||||
{OPD(PF_38_66, 0x3C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, 1>},
|
||||
{OPD(PF_38_66, 0x3D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, 4>},
|
||||
{OPD(PF_38_66, 0x3E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, 2>},
|
||||
{OPD(PF_38_66, 0x3F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, 4>},
|
||||
{OPD(PF_38_66, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, 4>},
|
||||
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::PACKUSOp<OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, false>},
|
||||
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, false>},
|
||||
{OPD(PF_38_66, 0x32), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x33), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, false>},
|
||||
{OPD(PF_38_66, 0x34), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x35), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x37), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i64Bit>},
|
||||
{OPD(PF_38_66, 0x38), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x39), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x3A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x3B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x3C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x3D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x3E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x3F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x41), 1, &OpDispatchBuilder::PHMINPOSUWOp},
|
||||
|
||||
{OPD(PF_38_NONE, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
|
||||
@@ -7,27 +7,27 @@ namespace FEXCore::IR {
|
||||
#define PF_3A_NONE 0
|
||||
#define PF_3A_66 1
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<4>},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<8>},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<4>},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<8>},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<4>},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<8>},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<2>},
|
||||
{OPD(0, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<OpSize::i16Bit>},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(0, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 1>},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 2>},
|
||||
{OPD(0, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 4>},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 4>},
|
||||
{OPD(0, PF_3A_66, 0x14), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i8Bit>},
|
||||
{OPD(0, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{OPD(0, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<1>},
|
||||
{OPD(0, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<OpSize::i8Bit>},
|
||||
{OPD(0, PF_3A_66, 0x21), 1, &OpDispatchBuilder::InsertPSOp},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<4>},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<4>},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<8>},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
@@ -40,8 +40,8 @@ constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDis
|
||||
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(1, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 8>},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<8>},
|
||||
{OPD(1, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i64Bit>},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i64Bit>},
|
||||
};
|
||||
|
||||
#undef PF_3A_NONE
|
||||
|
||||
@@ -66,30 +66,30 @@ constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDis
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 7), 1, &OpDispatchBuilder::RDPIDOp},
|
||||
|
||||
// GROUP 12
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i16Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, OpSize::i16Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i16Bit>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 2>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i16Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, OpSize::i16Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_12, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i16Bit>},
|
||||
|
||||
// GROUP 13
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i32Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, OpSize::i32Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i32Bit>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 4>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i32Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAIOp, OpSize::i32Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_13, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i32Bit>},
|
||||
|
||||
// GROUP 14
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 8>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 8>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_NONE, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i64Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i64Bit>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, 8>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLI, OpSize::i64Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 3), 1, &OpDispatchBuilder::PSRLDQ},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, 8>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLLI, OpSize::i64Bit>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 7), 1, &OpDispatchBuilder::PSLLDQ},
|
||||
|
||||
// GROUP 15
|
||||
|
||||
@@ -44,104 +44,104 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0xBE, 2, &OpDispatchBuilder::MOVSXOp},
|
||||
{0xC0, 2, &OpDispatchBuilder::XADDOp},
|
||||
{0xC3, 1, &OpDispatchBuilder::MOVGPRNTOp},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<2>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 2>},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<OpSize::i16Bit>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{0xC8, 8, &OpDispatchBuilder::BSWAPOp},
|
||||
|
||||
// SSE
|
||||
{0x10, 2, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x12, 2, &OpDispatchBuilder::MOVLPOp},
|
||||
{0x14, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 4>},
|
||||
{0x15, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 4>},
|
||||
{0x14, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
|
||||
{0x15, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<4, false, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<4, false, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<4>},
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, 4>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, 4>},
|
||||
{0x52, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, 4>},
|
||||
{0x53, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, 4>},
|
||||
{0x54, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, 16>},
|
||||
{0x55, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, 8>},
|
||||
{0x56, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, 16>},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i32Bit>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i32Bit>},
|
||||
{0x52, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{0x53, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x54, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
|
||||
{0x55, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, OpSize::i64Bit>},
|
||||
{0x56, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, OpSize::i128Bit>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, 4>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, 4>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, 8, 4, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<4, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, 4>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, 4>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, 4>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, 4>},
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 1>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 2>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 4>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<2>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 1>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 2>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 4>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<2>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 1>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 2>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 4>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<4>},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i32Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i8Bit>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i16Bit>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i16Bit>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i8Bit>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i16Bit>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i32Bit>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<OpSize::i16Bit>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i8Bit>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i16Bit>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i32Bit>},
|
||||
{0x70, 1, &OpDispatchBuilder::PSHUFW8ByteOp},
|
||||
|
||||
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 1>},
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 2>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 4>},
|
||||
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i8Bit>},
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
|
||||
{0x77, 1, &OpDispatchBuilder::X87EMMS},
|
||||
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<4>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, 4>},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<OpSize::i32Bit>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, OpSize::i32Bit>},
|
||||
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 2>},
|
||||
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 4>},
|
||||
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 8>},
|
||||
{0xD4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 8>},
|
||||
{0xD5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, 2>},
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i16Bit>},
|
||||
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i32Bit>},
|
||||
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i64Bit>},
|
||||
{0xD4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i64Bit>},
|
||||
{0xD5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, OpSize::i16Bit>},
|
||||
{0xD7, 1, &OpDispatchBuilder::MOVMSKOpOne}, // PMOVMSKB
|
||||
{0xD8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, 1>},
|
||||
{0xD9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, 2>},
|
||||
{0xDA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, 1>},
|
||||
{0xDB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, 8>},
|
||||
{0xDC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, 1>},
|
||||
{0xDD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, 2>},
|
||||
{0xDE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, 1>},
|
||||
{0xDF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, 8>},
|
||||
{0xE0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 1>},
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, 2>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, 4>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 2>},
|
||||
{0xD8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, OpSize::i8Bit>},
|
||||
{0xD9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, OpSize::i16Bit>},
|
||||
{0xDA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, OpSize::i8Bit>},
|
||||
{0xDB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, OpSize::i64Bit>},
|
||||
{0xDC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, OpSize::i8Bit>},
|
||||
{0xDD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, OpSize::i16Bit>},
|
||||
{0xDE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, OpSize::i8Bit>},
|
||||
{0xDF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, OpSize::i64Bit>},
|
||||
{0xE0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i8Bit>},
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i16Bit>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i32Bit>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, 1>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, 2>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, 2>},
|
||||
{0xEB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, 8>},
|
||||
{0xEC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, 1>},
|
||||
{0xED, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, 2>},
|
||||
{0xEE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, 2>},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i16Bit>},
|
||||
{0xEB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, OpSize::i64Bit>},
|
||||
{0xEC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, OpSize::i8Bit>},
|
||||
{0xED, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, OpSize::i16Bit>},
|
||||
{0xEE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, OpSize::i16Bit>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 2>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 4>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 8>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<4, false>},
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i16Bit>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i32Bit>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i64Bit>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, false>},
|
||||
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
|
||||
{0xF6, 1, &OpDispatchBuilder::PSADBW},
|
||||
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
{0xF8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 1>},
|
||||
{0xF9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 2>},
|
||||
{0xFA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 4>},
|
||||
{0xFB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 8>},
|
||||
{0xFC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 1>},
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 2>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 4>},
|
||||
{0xF8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i8Bit>},
|
||||
{0xF9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i16Bit>},
|
||||
{0xFA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i32Bit>},
|
||||
{0xFB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i64Bit>},
|
||||
{0xFC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i8Bit>},
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
|
||||
// FEX reserved instructions
|
||||
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
@@ -151,21 +151,21 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x10, 2, &OpDispatchBuilder::MOVSSOp},
|
||||
{0x12, 1, &OpDispatchBuilder::VMOVSLDUPOp},
|
||||
{0x16, 1, &OpDispatchBuilder::VMOVSHDUPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<4>},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<OpSize::i32Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<4, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<4, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>},
|
||||
{0x52, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<8, 4>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x52, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, false>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQOp, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
@@ -173,142 +173,142 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0xB8, 1, &OpDispatchBuilder::PopcountOp},
|
||||
{0xBC, 1, &OpDispatchBuilder::TZCNT},
|
||||
{0xBD, 1, &OpDispatchBuilder::LZCNT},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<4>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<4, true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryRepNEModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVSDOp},
|
||||
{0x12, 1, &OpDispatchBuilder::MOVDDUPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<8>},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<OpSize::i64Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<8, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<8, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
|
||||
// x52 = Invalid
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<4, 8>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, 4>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<4>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<4>},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i32Bit>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<8>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, true>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i64Bit>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{0xF0, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x12, 2, &OpDispatchBuilder::MOVLPOp},
|
||||
{0x14, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 8>},
|
||||
{0x15, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 8>},
|
||||
{0x14, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i64Bit>},
|
||||
{0x15, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i64Bit>},
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<8>},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
|
||||
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, 8>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, 8>},
|
||||
{0x54, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, 16>},
|
||||
{0x55, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, 8>},
|
||||
{0x56, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, 16>},
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i64Bit>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i64Bit>},
|
||||
{0x54, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
|
||||
{0x55, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, OpSize::i64Bit>},
|
||||
{0x56, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, OpSize::i128Bit>},
|
||||
{0x57, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, 8>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, 8>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, 4, 8, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, true>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, 8>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, 8>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, 8>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, 8>},
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 1>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 2>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 4>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<2>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 1>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 2>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, 4>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<2>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 1>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 2>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 4>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<4>},
|
||||
{0x6C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, 8>},
|
||||
{0x6D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, 8>},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i64Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i64Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false, true>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i64Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i64Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i64Bit>},
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i8Bit>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i16Bit>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i16Bit>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i8Bit>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i16Bit>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i32Bit>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<OpSize::i16Bit>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i8Bit>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i16Bit>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i32Bit>},
|
||||
{0x6C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i64Bit>},
|
||||
{0x6D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i64Bit>},
|
||||
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x70, 1, &OpDispatchBuilder::PSHUFDOp},
|
||||
|
||||
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 1>},
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 2>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, 4>},
|
||||
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i8Bit>},
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
|
||||
{0x78, 1, nullptr}, // GROUP 17
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, 8>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<8>},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i64Bit>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0x7F, 1, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<8>},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<2>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, 2>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, 8>},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<OpSize::i64Bit>},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<OpSize::i16Bit>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, OpSize::i64Bit>},
|
||||
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<8>},
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 2>},
|
||||
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 4>},
|
||||
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, 8>},
|
||||
{0xD4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 8>},
|
||||
{0xD5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, 2>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i64Bit>},
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i16Bit>},
|
||||
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i32Bit>},
|
||||
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i64Bit>},
|
||||
{0xD4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i64Bit>},
|
||||
{0xD5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, OpSize::i16Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQOp, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0xD7, 1, &OpDispatchBuilder::MOVMSKOpOne}, // PMOVMSKB
|
||||
{0xD8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, 1>},
|
||||
{0xD9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, 2>},
|
||||
{0xDA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, 1>},
|
||||
{0xDB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, 16>},
|
||||
{0xDC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, 1>},
|
||||
{0xDD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, 2>},
|
||||
{0xDE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, 1>},
|
||||
{0xDF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, 8>},
|
||||
{0xE0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 1>},
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, 2>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, 4>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, 2>},
|
||||
{0xD8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, OpSize::i8Bit>},
|
||||
{0xD9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQSUB, OpSize::i16Bit>},
|
||||
{0xDA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMIN, OpSize::i8Bit>},
|
||||
{0xDB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
|
||||
{0xDC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, OpSize::i8Bit>},
|
||||
{0xDD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUQADD, OpSize::i16Bit>},
|
||||
{0xDE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VUMAX, OpSize::i8Bit>},
|
||||
{0xDF, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VANDN, OpSize::i64Bit>},
|
||||
{0xE0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i8Bit>},
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i16Bit>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i32Bit>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, false>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true, false>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, 1>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, 2>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, 2>},
|
||||
{0xEB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, 16>},
|
||||
{0xEC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, 1>},
|
||||
{0xED, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, 2>},
|
||||
{0xEE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, 2>},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i16Bit>},
|
||||
{0xEB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VOR, OpSize::i128Bit>},
|
||||
{0xEC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, OpSize::i8Bit>},
|
||||
{0xED, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQADD, OpSize::i16Bit>},
|
||||
{0xEE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMAX, OpSize::i16Bit>},
|
||||
{0xEF, 1, &OpDispatchBuilder::VectorXOROp},
|
||||
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 2>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 4>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, 8>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<4, false>},
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i16Bit>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i32Bit>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i64Bit>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, false>},
|
||||
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
|
||||
{0xF6, 1, &OpDispatchBuilder::PSADBW},
|
||||
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
{0xF8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 1>},
|
||||
{0xF9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 2>},
|
||||
{0xFA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 4>},
|
||||
{0xFB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, 8>},
|
||||
{0xFC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 1>},
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 2>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, 4>},
|
||||
{0xF8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i8Bit>},
|
||||
{0xF9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i16Bit>},
|
||||
{0xFA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i32Bit>},
|
||||
{0xFB, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSUB, OpSize::i64Bit>},
|
||||
{0xFC, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i8Bit>},
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable_64[] = {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -26,7 +26,7 @@ class OrderedNode;
|
||||
Ref OpDispatchBuilder::GetX87Top() {
|
||||
// Yes, we are storing 3 bits in a single flag register.
|
||||
// Deal with it
|
||||
return _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::GetX87Tag(Ref Value, Ref AbridgedFTW) {
|
||||
@@ -56,17 +56,17 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87Top(Ref Value) {
|
||||
_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
// Float LoaD operation with memory operand
|
||||
void OpDispatchBuilder::FLD(OpcodeArgs, size_t Width) {
|
||||
size_t ReadWidth = (Width == 80) ? 16 : Width / 8;
|
||||
void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref ConvertedData = Data;
|
||||
// Convert to 80bit float
|
||||
if (Width == 32 || Width == 64) {
|
||||
if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
ConvertedData = _F80CVTTo(Data, ReadWidth);
|
||||
}
|
||||
_PushStack(ConvertedData, Data, ReadWidth, true);
|
||||
@@ -79,31 +79,31 @@ void OpDispatchBuilder::FLDFromStack(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
_PushStack(ConvertedData, Data, 16, true);
|
||||
_PushStack(ConvertedData, Data, OpSize::i128Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
Ref converted = _F80BCDStore(_ReadStackValue(0));
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant Constant) {
|
||||
// Update TOP
|
||||
Ref Data = LoadAndCacheNamedVectorConstant(16, Constant);
|
||||
_PushStack(Data, Data, 16, true);
|
||||
Ref Data = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, Constant);
|
||||
_PushStack(Data, Data, OpSize::i128Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
size_t ReadWidth = GetSrcSize(Op);
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
|
||||
// Sign extend to 64bits
|
||||
if (ReadWidth != 8) {
|
||||
Data = _Sbfe(OpSize::i64Bit, ReadWidth * 8, 0, Data);
|
||||
if (ReadWidth != OpSize::i64Bit) {
|
||||
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
|
||||
}
|
||||
|
||||
// We're about to clobber flags to grab the sign, so save NZCV.
|
||||
@@ -123,14 +123,14 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
Ref ConvertedData = _VCastFromGPR(16, 8, shifted);
|
||||
ConvertedData = _VInsElement(16, 8, 1, 0, ConvertedData, _VCastFromGPR(16, 8, upper));
|
||||
Ref ConvertedData = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, shifted);
|
||||
ConvertedData = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 0, ConvertedData, _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, upper));
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, size_t Width) {
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreStackMemory(Mem, OpSize::i128Bit, true, Width / 8);
|
||||
_StoreStackMemory(Mem, OpSize::i128Bit, true, Width);
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
@@ -149,18 +149,18 @@ void OpDispatchBuilder::FSTToStack(OpcodeArgs) {
|
||||
|
||||
// Store integer to memory (possibly with truncation)
|
||||
void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Data = _ReadStackValue(0);
|
||||
Data = _F80CVTInt(Size, Data, Truncate);
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, 1);
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FADD(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) { // Implicit argument case
|
||||
auto Offset = Op->OP & 7;
|
||||
auto St0 = 0;
|
||||
@@ -175,22 +175,22 @@ void OpDispatchBuilder::FADD(OpcodeArgs, size_t Width, bool Integer, OpDispatchB
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != 80, "No 80-bit floats from memory");
|
||||
LOGMAN_THROW_A_FMT(Width != OpSize::f80Bit, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width / 8);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width / 8);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
_F80AddValue(0, Arg);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FMUL(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) { // Implicit argument case
|
||||
auto offset = Op->OP & 7;
|
||||
auto st0 = 0;
|
||||
@@ -205,15 +205,15 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, size_t Width, bool Integer, OpDispatchB
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != 80, "No 80-bit floats from memory");
|
||||
LOGMAN_THROW_A_FMT(Width != OpSize::f80Bit, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width / 8);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width / 8);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -224,7 +224,7 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, size_t Width, bool Integer, OpDispatchB
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FDIV(OpcodeArgs, size_t Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto Offset = Op->OP & 7;
|
||||
const auto St0 = 0;
|
||||
@@ -242,15 +242,15 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, size_t Width, bool Integer, bool Revers
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != 80, "No 80-bit floats from memory");
|
||||
LOGMAN_THROW_A_FMT(Width != OpSize::f80Bit, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width / 8);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width / 8);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -265,7 +265,7 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, size_t Width, bool Integer, bool Revers
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSUB(OpcodeArgs, size_t Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto Offset = Op->OP & 7;
|
||||
const auto St0 = 0;
|
||||
@@ -283,15 +283,15 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, size_t Width, bool Integer, bool Revers
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Width != 80, "No 80-bit floats from memory");
|
||||
LOGMAN_THROW_A_FMT(Width != OpSize::f80Bit, "No 80-bit floats from memory");
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width / 8);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width / 8);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
// top of stack is at offset zero
|
||||
@@ -342,42 +342,42 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
// Before we store anything we need to sync our stack to the registers.
|
||||
_SyncStackToSlow();
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -400,26 +400,27 @@ Ref OpDispatchBuilder::ReconstructX87StateFromFSW_Helper(Ref FSW) {
|
||||
void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
auto Size = GetSrcSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(IR::OpSizeToSize(Size) * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(IR::OpSizeToSize(Size) * 2));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
_SyncStackToSlow();
|
||||
|
||||
// 14 bytes for 16bit
|
||||
// 2 Bytes : FCW
|
||||
// 2 Bytes : FSW
|
||||
@@ -438,60 +439,66 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// 2 bytes : Opcode
|
||||
// 4 bytes : data pointer offset
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
const auto Size = GetDstSize(Op);
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
auto data = _LoadContextIndexed(Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreMem(FPRClass, 16, data, Mem, _Constant((Size * 7) + (10 * i)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMem(FPRClass, OpSize::i128Bit, data, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
auto data = _LoadContextIndexed(Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, data, Mem, _Constant((Size * 7) + (7 * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, topBytes, Mem, _Constant((Size * 7) + (7 * 10) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, data, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4);
|
||||
_StoreMem(FPRClass, OpSize::i16Bit, topBytes, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINIT(Op);
|
||||
@@ -499,17 +506,27 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
if (ReducedPrecisionMode) {
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = NewFCW;
|
||||
auto roundShift = _Constant(10);
|
||||
auto roundMask = _Constant(3);
|
||||
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
|
||||
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
}
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
@@ -517,15 +534,18 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
Ref Mask = _VCastFromGPR(16, 8, low);
|
||||
Mask = _VInsGPR(16, 8, 1, Mask, high);
|
||||
|
||||
Ref Mask = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, low);
|
||||
Mask = _VInsGPR(OpSize::i128Bit, OpSize::i64Bit, 1, Mask, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, 16, Mem, _Constant((Size * 7) + (10 * i)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(16, 16, Reg, Mask);
|
||||
|
||||
_StoreContextIndexed(Reg, Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
|
||||
if (ReducedPrecisionMode) {
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg);
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
@@ -534,29 +554,31 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
Ref Reg = _LoadMem(FPRClass, 8, Mem, _Constant((Size * 7) + (10 * 7)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh = _LoadMem(FPRClass, 2, Mem, _Constant((Size * 7) + (10 * 7) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh);
|
||||
_StoreContextIndexed(Reg, Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh =
|
||||
_LoadMem(FPRClass, OpSize::i16Bit, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
|
||||
if (ReducedPrecisionMode) {
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
}
|
||||
|
||||
// Load / Store Control Word
|
||||
void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResult(GPRClass, Op, FCW, -1);
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResult(GPRClass, Op, FCW, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
// FIXME: Because loading control flags will affect several instructions in fast path, we might have
|
||||
// to switch for now to slow mode whenever these are manually changed.
|
||||
// Remove the next line and try DF_04.asm in fast path.
|
||||
_StackForceSlow();
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
|
||||
void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
uint8_t Offset = Op->OP & 7;
|
||||
// fxch st0, st0 is for us essentially a nop
|
||||
@@ -569,15 +591,15 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
void OpDispatchBuilder::X87FYL2X(OpcodeArgs, bool IsFYL2XP1) {
|
||||
if (IsFYL2XP1) {
|
||||
// create an add between top of stack and 1.
|
||||
Ref One = ReducedPrecisionMode ? _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000)) :
|
||||
LoadAndCacheNamedVectorConstant(16, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
Ref One = ReducedPrecisionMode ? _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, _Constant(0x3FF0000000000000)) :
|
||||
LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
_F80AddValue(0, One);
|
||||
}
|
||||
|
||||
_F80FYL2XStack();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FCOMI(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
|
||||
void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
@@ -588,13 +610,13 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, size_t Width, bool Integer, OpDispatch
|
||||
Res = _F80CmpStack(Offset);
|
||||
} else {
|
||||
// Memory arg
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, Width / 8);
|
||||
b = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, Width / 8);
|
||||
b = _F80CVTTo(arg, Width);
|
||||
}
|
||||
}
|
||||
Res = _F80CmpValue(b);
|
||||
@@ -612,10 +634,7 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, size_t Width, bool Integer, OpDispatch
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
} else {
|
||||
// Invalidate deferred flags early
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetCFDirect(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
@@ -675,7 +694,6 @@ void OpDispatchBuilder::X87ModifySTP(OpcodeArgs, bool Inc) {
|
||||
// Optionally we can pass a pre calculated value for Top, otherwise we calculate it
|
||||
// during the function runtime.
|
||||
Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
|
||||
// Start with the top value
|
||||
auto Top = T ? T : GetX87Top();
|
||||
Ref FSW = _Lshl(OpSize::i64Bit, Top, _Constant(11));
|
||||
@@ -700,18 +718,21 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
// There's no load Status Word instruction but you can load it through frstor
|
||||
// or fldenv.
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
|
||||
Ref TopValue = _SyncStackToSlow();
|
||||
Ref StatusWord = ReconstructFSW_Helper(TopValue);
|
||||
StoreResult(GPRClass, Op, StatusWord, -1);
|
||||
StoreResult(GPRClass, Op, StatusWord, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
auto Zero = _Constant(0);
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
_SetRoundingMode(Zero, false, Zero);
|
||||
}
|
||||
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = _Constant(16, 0x037F);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _Constant(OpSize::i16Bit, 0x037F);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Set top to zero
|
||||
SetX87Top(Zero);
|
||||
@@ -776,13 +797,14 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
auto AllOneConst = _Constant(0xffff'ffff'ffff'ffffull);
|
||||
|
||||
Ref SrcCond = SelectCC(CC, OpSize::i64Bit, AllOneConst, ZeroConst);
|
||||
Ref VecCond = _VDupFromGPR(16, 8, SrcCond);
|
||||
_F80VBSLStack(16, VecCond, Op->OP & 7, 0);
|
||||
Ref VecCond = _VDupFromGPR(OpSize::i128Bit, OpSize::i64Bit, SrcCond);
|
||||
_F80VBSLStack(OpSize::i128Bit, VecCond, Op->OP & 7, 0);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
auto a = _ReadStackValue(0);
|
||||
Ref Result = ReducedPrecisionMode ? _VExtractToGPR(8, 8, a, 0) : _VExtractToGPR(16, 8, a, 1);
|
||||
Ref Result =
|
||||
ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Result) : _Bfe(OpSize::i64Bit, 1, 15, Result);
|
||||
@@ -804,4 +826,14 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(C3);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FXTRACT(OpcodeArgs) {
|
||||
auto Top = _ReadStackValue(0);
|
||||
|
||||
_PopStackDestroy();
|
||||
auto Exp = _F80XTRACT_EXP(Top);
|
||||
auto Sig = _F80XTRACT_SIG(Top);
|
||||
_PushStack(Exp, Exp, OpSize::f80Bit, true);
|
||||
_PushStack(Sig, Sig, OpSize::f80Bit, true);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
@@ -22,38 +23,28 @@ class OrderedNode;
|
||||
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
// Init host rounding mode to zero
|
||||
auto Zero = _Constant(0);
|
||||
_SetRoundingMode(Zero, false, Zero);
|
||||
|
||||
// Call generic version
|
||||
FNINIT(Op);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(IR::OpSizeToSize(Size)), Size, MEM_OFFSET_SXTX, 1);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
@@ -62,59 +53,59 @@ void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
// F64 ops
|
||||
// Float load op with memory operand
|
||||
void OpDispatchBuilder::FLDF64(OpcodeArgs, size_t Width) {
|
||||
size_t ReadWidth = (Width == 80) ? 16 : Width / 8;
|
||||
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
Ref ConvertedData = Data;
|
||||
if (Width == 32) {
|
||||
ConvertedData = _Float_FToF(8, 4, Data);
|
||||
} else if (Width == 80) {
|
||||
ConvertedData = _F80CVT(8, Data);
|
||||
if (Width == OpSize::i32Bit) {
|
||||
ConvertedData = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Data);
|
||||
} else if (Width == OpSize::f80Bit) {
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, Data);
|
||||
}
|
||||
_PushStack(ConvertedData, Data, ReadWidth, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i128Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
ConvertedData = _F80CVT(8, ConvertedData);
|
||||
_PushStack(ConvertedData, Data, 8, true);
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
|
||||
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
Ref converted = _F80CVTTo(_ReadStackValue(0), 8);
|
||||
Ref converted = _F80CVTTo(_ReadStackValue(0), OpSize::i64Bit);
|
||||
converted = _F80BCDStore(converted);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FLDF64_Const(OpcodeArgs, uint64_t Num) {
|
||||
auto Data = _VCastFromGPR(8, 8, _Constant(Num));
|
||||
_PushStack(Data, Data, 8, true);
|
||||
auto Data = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, _Constant(Num));
|
||||
_PushStack(Data, Data, OpSize::i64Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
size_t ReadWidth = GetSrcSize(Op);
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
if (ReadWidth == 2) {
|
||||
Data = _Sbfe(OpSize::i64Bit, ReadWidth * 8, 0, Data);
|
||||
if (ReadWidth == OpSize::i16Bit) {
|
||||
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
|
||||
}
|
||||
auto ConvertedData = _Float_FromGPR_S(8, ReadWidth == 4 ? 4 : 8, Data);
|
||||
auto ConvertedData = _Float_FromGPR_S(OpSize::i64Bit, ReadWidth == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, Data);
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSTF64(OpcodeArgs, size_t Width) {
|
||||
void OpDispatchBuilder::FSTF64(OpcodeArgs, IR::OpSize Width) {
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreStackMemory(Mem, OpSize::i64Bit, true, Width / 8);
|
||||
_StoreStackMemory(Mem, OpSize::i64Bit, true, Width);
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
@@ -122,22 +113,22 @@ void OpDispatchBuilder::FSTF64(OpcodeArgs, size_t Width) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref data = _ReadStackValue(0);
|
||||
if (Truncate) {
|
||||
data = _Float_ToGPR_ZS(Size == 4 ? 4 : 8, 8, data);
|
||||
data = _Float_ToGPR_ZS(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data);
|
||||
} else {
|
||||
data = _Float_ToGPR_S(Size == 4 ? 4 : 8, 8, data);
|
||||
data = _Float_ToGPR_S(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data);
|
||||
}
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, 1);
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FADDF64(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) { // Implicit argument case
|
||||
auto Offset = Op->OP & 7;
|
||||
auto St0 = 0;
|
||||
@@ -157,14 +148,14 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, size_t Width, bool Integer, OpDispat
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, arg);
|
||||
} else if (Width == 32) {
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(8, 4, arg);
|
||||
} else if (Width == 64) {
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
|
||||
@@ -173,7 +164,7 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, size_t Width, bool Integer, OpDispat
|
||||
}
|
||||
|
||||
// FIXME: following is very similar to FADDF64
|
||||
void OpDispatchBuilder::FMULF64(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) { // Implicit argument case
|
||||
auto offset = Op->OP & 7;
|
||||
auto st0 = 0;
|
||||
@@ -193,14 +184,14 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, size_t Width, bool Integer, OpDispat
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, arg);
|
||||
} else if (Width == 32) {
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(8, 4, arg);
|
||||
} else if (Width == 64) {
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
|
||||
@@ -212,7 +203,7 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, size_t Width, bool Integer, OpDispat
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FDIVF64(OpcodeArgs, size_t Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto offset = Op->OP & 7;
|
||||
const auto st0 = 0;
|
||||
@@ -240,17 +231,17 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, size_t Width, bool Integer, bool Rev
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
if (Width == OpSize::i16Bit) {
|
||||
Arg = _Sbfe(OpSize::i64Bit, 16, 0, Arg);
|
||||
}
|
||||
Arg = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, Arg);
|
||||
} else if (Width == 32) {
|
||||
Arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = _Float_FToF(8, 4, Arg);
|
||||
} else if (Width == 64) {
|
||||
Arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
}
|
||||
@@ -267,7 +258,7 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, size_t Width, bool Integer, bool Rev
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSUBF64(OpcodeArgs, size_t Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpDispatchBuilder::OpResult ResInST0) {
|
||||
if (Op->Src[0].IsNone()) {
|
||||
const auto Offset = Op->OP & 7;
|
||||
const auto St0 = 0;
|
||||
@@ -295,17 +286,17 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, size_t Width, bool Integer, bool Rev
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, arg);
|
||||
} else if (Width == 32) {
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(8, 4, arg);
|
||||
} else if (Width == 64) {
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
}
|
||||
@@ -328,11 +319,10 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
|
||||
|
||||
// Now we do our comparison.
|
||||
_F80StackTest(0);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FCOMIF64(OpcodeArgs, size_t Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
|
||||
void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
@@ -342,17 +332,17 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, size_t Width, bool Integer, OpDispa
|
||||
b = _ReadStackValue(offset);
|
||||
} else {
|
||||
// Memory arg
|
||||
if (Width == 16 || Width == 32 || Width == 64) {
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if (Width == 16) {
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, Width == 64 ? 8 : 4, arg);
|
||||
} else if (Width == 32) {
|
||||
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if (Width == 64) {
|
||||
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
}
|
||||
@@ -363,7 +353,6 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, size_t Width, bool Integer, OpDispa
|
||||
GetNZCV();
|
||||
|
||||
_F80CmpValue(b);
|
||||
PossiblySetNZCVBits = ~0;
|
||||
ConvertNZCVToX87();
|
||||
} else {
|
||||
HandleNZCVWrite();
|
||||
@@ -379,144 +368,37 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, size_t Width, bool Integer, OpDispa
|
||||
}
|
||||
}
|
||||
|
||||
// This function converts to F80 on save for compatibility
|
||||
void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
_SyncStackToSlow();
|
||||
// 14 bytes for 16bit
|
||||
// 2 Bytes : FCW
|
||||
// 2 Bytes : FSW
|
||||
// 2 bytes : FTW
|
||||
// 2 bytes : Instruction offset
|
||||
// 2 bytes : Instruction CS selector
|
||||
// 2 bytes : Data offset
|
||||
// 2 bytes : Data selector
|
||||
void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
|
||||
// Split node into SIG and EXP while handling the special zero case.
|
||||
// i.e. if val == 0.0, then sig = 0.0, exp = -inf
|
||||
// if val == -0.0, then sig = -0.0, exp = -inf
|
||||
// otherwise we just extract the 64-bit sig and exp as normal.
|
||||
Ref Node = _ReadStackValue(0);
|
||||
|
||||
// 28 bytes for 32bit
|
||||
// 4 bytes : FCW
|
||||
// 4 bytes : FSW
|
||||
// 4 bytes : FTW
|
||||
// 4 bytes : Instruction pointer
|
||||
// 2 bytes : instruction pointer selector
|
||||
// 2 bytes : Opcode
|
||||
// 4 bytes : data pointer offset
|
||||
// 4 bytes : data pointer selector
|
||||
Ref Gpr = _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, Node, 0);
|
||||
|
||||
const auto Size = GetDstSize(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
// zero case
|
||||
Ref ExpZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, _Constant(0xfff0'0000'0000'0000UL));
|
||||
Ref SigZV = Node;
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
// non zero case
|
||||
Ref ExpNZ = _Bfe(OpSize::i64Bit, 11, 52, Gpr);
|
||||
ExpNZ = _Sub(OpSize::i64Bit, ExpNZ, _Constant(1023));
|
||||
Ref ExpNZV = _Float_FromGPR_S(OpSize::i64Bit, OpSize::i64Bit, ExpNZ);
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
Ref SigNZ = _And(OpSize::i64Bit, Gpr, _Constant(0x800f'ffff'ffff'ffffLL));
|
||||
SigNZ = _Or(OpSize::i64Bit, SigNZ, _Constant(0x3ff0'0000'0000'0000LL));
|
||||
Ref SigNZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, SigNZ);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
// Comparison and select to push onto stack
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Gpr, _Constant(0x7fff'ffff'ffff'ffffUL));
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
Ref Sig = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, SigZV, SigNZV);
|
||||
Ref Exp = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, ExpZV, ExpNZV);
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref data = _LoadContextIndexed(Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
data = _F80CVTTo(data, 8);
|
||||
_StoreMem(FPRClass, 16, data, Mem, _Constant((Size * 7) + (i * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
Ref data = _LoadContextIndexed(Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
data = _F80CVTTo(data, 8);
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, data, Mem, _Constant((Size * 7) + (7 * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, topBytes, Mem, _Constant((Size * 7) + (7 * 10) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINITF64(Op);
|
||||
_PopStackDestroy();
|
||||
_PushStack(Exp, Exp, OpSize::i64Bit, true);
|
||||
_PushStack(Sig, Sig, OpSize::i64Bit, true);
|
||||
}
|
||||
|
||||
// This function converts from F80 on load for compatibility
|
||||
|
||||
void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
const auto Size = GetSrcSize(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = NewFCW;
|
||||
auto roundShift = _Constant(10);
|
||||
auto roundMask = _Constant(3);
|
||||
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
|
||||
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
Ref Mask = _VCastFromGPR(16, 8, low);
|
||||
Mask = _VInsGPR(16, 8, 1, Mask, high);
|
||||
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, 16, Mem, _Constant((Size * 7) + (i * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(16, 16, Reg, Mask);
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(8, Reg);
|
||||
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
|
||||
Ref Reg = _LoadMem(FPRClass, 8, Mem, _Constant((Size * 7) + (7 * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh = _LoadMem(FPRClass, 2, Mem, _Constant((Size * 7) + (7 * 10) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh);
|
||||
Reg = _F80CVT(8, Reg); // Convert to double precision
|
||||
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -225,7 +225,6 @@ enum InstType {
|
||||
TYPE_SECONDARY_TABLE_PREFIX,
|
||||
TYPE_X87_TABLE_PREFIX,
|
||||
TYPE_VEX_TABLE_PREFIX,
|
||||
TYPE_XOP_TABLE_PREFIX,
|
||||
TYPE_INST,
|
||||
TYPE_X87 = TYPE_INST,
|
||||
TYPE_INVALID,
|
||||
@@ -466,14 +465,6 @@ constexpr size_t MAX_VEX_TABLE_SIZE = (1 << 13);
|
||||
// group select (3 bits for now) | ModRM opcode (3 bits)
|
||||
constexpr size_t MAX_VEX_GROUP_TABLE_SIZE = (1 << 7);
|
||||
|
||||
// XOP
|
||||
// group (2 bits for now) | vex.pp (2 bits) | opcode (8bit)
|
||||
constexpr size_t MAX_XOP_TABLE_SIZE = (1 << 13);
|
||||
|
||||
// XOP group ops
|
||||
// group select (2 bits for now) | modrm opcode (3 bits)
|
||||
constexpr size_t MAX_XOP_GROUP_TABLE_SIZE = (1 << 6);
|
||||
|
||||
extern std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps;
|
||||
extern std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps;
|
||||
extern std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps;
|
||||
@@ -492,10 +483,6 @@ extern std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps;
|
||||
extern std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps;
|
||||
extern std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps;
|
||||
|
||||
// XOP
|
||||
extern std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps;
|
||||
extern std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps;
|
||||
|
||||
template <typename OpcodeType>
|
||||
struct X86TablesInfoStruct {
|
||||
OpcodeType first;
|
||||
|
||||
@@ -1,143 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: frontend|x86-tables
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> Table{};
|
||||
#define OPD(group, pp, opcode) ( (group << 10) | (pp << 8) | (opcode))
|
||||
constexpr uint16_t XOP_GROUP_8 = 0;
|
||||
constexpr uint16_t XOP_GROUP_9 = 1;
|
||||
constexpr uint16_t XOP_GROUP_A = 2;
|
||||
|
||||
constexpr U16U8InfoStruct XOPTable[] = {
|
||||
// Group 8
|
||||
{OPD(XOP_GROUP_8, 0, 0x85), 1, X86InstInfo{"VPMAXSSWW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x86), 1, X86InstInfo{"VPMACSSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x87), 1, X86InstInfo{"VPMAXSSDQL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0x8E), 1, X86InstInfo{"VPMACSSDD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x8F), 1, X86InstInfo{"VPMACSSDQH", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0x95), 1, X86InstInfo{"VPMAXSWW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x96), 1, X86InstInfo{"VPMAXSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x97), 1, X86InstInfo{"VPMAXSDQL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0x9E), 1, X86InstInfo{"VPMACSDD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0x9F), 1, X86InstInfo{"VPMACSDQH", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xA2), 1, X86InstInfo{"VPCMOV", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xA3), 1, X86InstInfo{"VPPERM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xA6), 1, X86InstInfo{"VPMADCSSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xB6), 1, X86InstInfo{"VPMADCSWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xC0), 1, X86InstInfo{"VPROTB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xC1), 1, X86InstInfo{"VPROTW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xC2), 1, X86InstInfo{"VPROTD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xC3), 1, X86InstInfo{"VPROTQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xCC), 1, X86InstInfo{"VPCOMccB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xCD), 1, X86InstInfo{"VPCOMccW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xCE), 1, X86InstInfo{"VPCOMccD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xCF), 1, X86InstInfo{"VPCOMccQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_8, 0, 0xEC), 1, X86InstInfo{"VPCOMccUB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xED), 1, X86InstInfo{"VPCOMccUW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xEE), 1, X86InstInfo{"VPCOMccUD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_8, 0, 0xEF), 1, X86InstInfo{"VPCOMccUQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group 9
|
||||
{OPD(XOP_GROUP_9, 0, 0x01), 1, X86InstInfo{"", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, // Group 1
|
||||
{OPD(XOP_GROUP_9, 0, 0x02), 1, X86InstInfo{"", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, // Group 2
|
||||
{OPD(XOP_GROUP_9, 0, 0x12), 1, X86InstInfo{"", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, // Group 3
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0x80), 1, X86InstInfo{"VFRZPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x81), 1, X86InstInfo{"VFRCZPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x82), 1, X86InstInfo{"VFRCZSS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x83), 1, X86InstInfo{"VFRCZSD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0x90), 1, X86InstInfo{"VPROTB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x91), 1, X86InstInfo{"VPROTW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x92), 1, X86InstInfo{"VPROTD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x93), 1, X86InstInfo{"VRPTOQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x94), 1, X86InstInfo{"VPSHLB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x95), 1, X86InstInfo{"VPSHLW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x96), 1, X86InstInfo{"VPSHLD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x97), 1, X86InstInfo{"VPSHLQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0x98), 1, X86InstInfo{"VPSHAB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x99), 1, X86InstInfo{"VPSHAW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x9A), 1, X86InstInfo{"VPSHAD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0x9B), 1, X86InstInfo{"VPSHAQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0xC1), 1, X86InstInfo{"VPHADDBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xC2), 1, X86InstInfo{"VPHADDBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xC3), 1, X86InstInfo{"VPHADDBQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xC6), 1, X86InstInfo{"VPHADDWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xC7), 1, X86InstInfo{"VPHADDWQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xCB), 1, X86InstInfo{"VPHADDDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0xD1), 1, X86InstInfo{"VPHADDUBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xD2), 1, X86InstInfo{"VPHADDUBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xD3), 1, X86InstInfo{"VPHADDUBQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xD6), 1, X86InstInfo{"VPHADDUWD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xD7), 1, X86InstInfo{"VPHADDUWQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xDB), 1, X86InstInfo{"VPHADDUDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(XOP_GROUP_9, 0, 0xE1), 1, X86InstInfo{"VPHSUBBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xE2), 1, X86InstInfo{"VPHSUBBD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_9, 0, 0xE3), 1, X86InstInfo{"VPHSUBDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group A
|
||||
{OPD(XOP_GROUP_A, 0, 0x10), 1, X86InstInfo{"BEXTR", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(XOP_GROUP_A, 0, 0x12), 1, X86InstInfo{"", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}}, // Group 4
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), XOPTable, std::size(XOPTable));
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> Table{};
|
||||
#define OPD(subgroup, opcode) (((subgroup - 1) << 3) | (opcode))
|
||||
constexpr U8U8InfoStruct XOPGroupTable[] = {
|
||||
// Group 1
|
||||
{OPD(1, 1), 1, X86InstInfo{"BLCFILL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 2), 1, X86InstInfo{"BLSFILL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 3), 1, X86InstInfo{"BLCS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 4), 1, X86InstInfo{"TZMSK", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 5), 1, X86InstInfo{"BLCIC", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 6), 1, X86InstInfo{"BLSIC", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(1, 7), 1, X86InstInfo{"T1MSKC", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group 2
|
||||
{OPD(2, 1), 1, X86InstInfo{"BLCMSK", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 6), 1, X86InstInfo{"BLCI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group 3
|
||||
{OPD(3, 0), 1, X86InstInfo{"LLWPCB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 1), 1, X86InstInfo{"SLWPCB", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// Group 4
|
||||
{OPD(4, 0), 1, X86InstInfo{"LWPINS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(4, 1), 1, X86InstInfo{"LWPVAL", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), XOPGroupTable, std::size(XOPGroupTable));
|
||||
return Table;
|
||||
}();
|
||||
|
||||
}
|
||||
@@ -548,13 +548,16 @@ protected:
|
||||
|
||||
// This must directly match bytes to the named opsize.
|
||||
// Implicit sized IR operations does math to get between sizes.
|
||||
enum OpSize : uint8_t {
|
||||
enum class OpSize : uint8_t {
|
||||
iUnsized = 0,
|
||||
i8Bit = 1,
|
||||
i16Bit = 2,
|
||||
i32Bit = 4,
|
||||
i64Bit = 8,
|
||||
f80Bit = 10,
|
||||
i128Bit = 16,
|
||||
i256Bit = 32,
|
||||
iInvalid = 0xFF,
|
||||
};
|
||||
|
||||
enum class FloatCompareOp : uint8_t {
|
||||
@@ -578,16 +581,71 @@ enum class ShiftType : uint8_t {
|
||||
// This is a nop operation and will be eliminated by the compiler.
|
||||
static inline OpSize SizeToOpSize(uint8_t Size) {
|
||||
switch (Size) {
|
||||
case 0: return OpSize::iUnsized;
|
||||
case 1: return OpSize::i8Bit;
|
||||
case 2: return OpSize::i16Bit;
|
||||
case 4: return OpSize::i32Bit;
|
||||
case 8: return OpSize::i64Bit;
|
||||
case 10: return OpSize::f80Bit;
|
||||
case 16: return OpSize::i128Bit;
|
||||
case 32: return OpSize::i256Bit;
|
||||
case 0xFF: return OpSize::iInvalid;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
// This is a nop operation and will be eliminated by the compiler.
|
||||
static inline uint8_t OpSizeToSize(IR::OpSize Size) {
|
||||
switch (Size) {
|
||||
case OpSize::iUnsized: return 0;
|
||||
case OpSize::i8Bit: return 1;
|
||||
case OpSize::i16Bit: return 2;
|
||||
case OpSize::i32Bit: return 4;
|
||||
case OpSize::i64Bit: return 8;
|
||||
case OpSize::f80Bit: return 10;
|
||||
case OpSize::i128Bit: return 16;
|
||||
case OpSize::i256Bit: return 32;
|
||||
case OpSize::iInvalid: return 0xFF;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
static inline uint16_t OpSizeAsBits(IR::OpSize Size) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::OpSizeToSize(Size) * 8u;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_integral_v<T>)
|
||||
static inline OpSize operator<<(IR::OpSize Size, T Shift) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::SizeToOpSize(IR::OpSizeToSize(Size) << Shift);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_integral_v<T>)
|
||||
static inline OpSize operator>>(IR::OpSize Size, T Shift) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::SizeToOpSize(IR::OpSizeToSize(Size) >> Shift);
|
||||
}
|
||||
|
||||
static inline OpSize operator/(IR::OpSize Size, IR::OpSize Divisor) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::SizeToOpSize(IR::OpSizeToSize(Size) / IR::OpSizeToSize(Divisor));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_integral_v<T>)
|
||||
static inline OpSize operator/(IR::OpSize Size, T Divisor) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::SizeToOpSize(IR::OpSizeToSize(Size) / Divisor);
|
||||
}
|
||||
|
||||
static inline uint8_t NumElements(IR::OpSize RegisterSize, IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(RegisterSize != IR::OpSize::iInvalid && ElementSize != IR::OpSize::iInvalid, "Invalid Size");
|
||||
return IR::OpSizeToSize(RegisterSize) / IR::OpSizeToSize(ElementSize);
|
||||
}
|
||||
|
||||
#define IROP_ENUM
|
||||
#define IROP_STRUCTS
|
||||
#define IROP_SIZES
|
||||
|
||||
+402
-399
File diff suppressed because it is too large.
Load diff
@@ -112,17 +112,17 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
}
|
||||
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
uint32_t ElementSize = IROp->ElementSize;
|
||||
uint32_t NumElements = IROp->Size;
|
||||
if (!IROp->ElementSize) {
|
||||
auto ElementSize = IROp->ElementSize;
|
||||
uint32_t NumElements = 0;
|
||||
if (IROp->ElementSize == OpSize::iUnsized) {
|
||||
ElementSize = IROp->Size;
|
||||
}
|
||||
|
||||
if (ElementSize) {
|
||||
NumElements /= ElementSize;
|
||||
if (ElementSize != OpSize::iUnsized) {
|
||||
NumElements = IR::NumElements(IROp->Size, ElementSize);
|
||||
}
|
||||
|
||||
*out << " i" << std::dec << (ElementSize * 8);
|
||||
*out << " i" << std::dec << IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
if (NumElements > 1) {
|
||||
*out << "v" << std::dec << NumElements;
|
||||
@@ -294,14 +294,14 @@ void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocation
|
||||
AddIndent();
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
|
||||
uint32_t ElementSize = IROp->ElementSize;
|
||||
uint32_t NumElements = IROp->Size;
|
||||
if (!IROp->ElementSize) {
|
||||
auto ElementSize = IROp->ElementSize;
|
||||
uint8_t NumElements = 0;
|
||||
if (IROp->ElementSize != OpSize::iUnsized) {
|
||||
ElementSize = IROp->Size;
|
||||
}
|
||||
|
||||
if (ElementSize) {
|
||||
NumElements /= ElementSize;
|
||||
if (ElementSize != OpSize::iUnsized) {
|
||||
NumElements = IR::NumElements(IROp->Size, ElementSize);
|
||||
}
|
||||
|
||||
*out << "%" << std::dec << ID;
|
||||
@@ -324,7 +324,7 @@ void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocation
|
||||
}
|
||||
}
|
||||
|
||||
*out << " i" << std::dec << (ElementSize * 8);
|
||||
*out << " i" << std::dec << IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
if (NumElements > 1) {
|
||||
*out << "v" << std::dec << NumElements;
|
||||
@@ -333,17 +333,17 @@ void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocation
|
||||
*out << " = ";
|
||||
} else {
|
||||
|
||||
uint32_t ElementSize = IROp->ElementSize;
|
||||
if (!IROp->ElementSize) {
|
||||
auto ElementSize = IROp->ElementSize;
|
||||
if (IROp->ElementSize == OpSize::iUnsized) {
|
||||
ElementSize = IROp->Size;
|
||||
}
|
||||
uint32_t NumElements = 0;
|
||||
if (ElementSize) {
|
||||
NumElements = IROp->Size / ElementSize;
|
||||
if (ElementSize != OpSize::iUnsized) {
|
||||
NumElements = IR::NumElements(IROp->Size, ElementSize);
|
||||
}
|
||||
|
||||
*out << "(%" << std::dec << ID << ' ';
|
||||
*out << 'i' << std::dec << (ElementSize * 8);
|
||||
*out << 'i' << std::dec << IR::OpSizeAsBits(ElementSize);
|
||||
if (NumElements > 1) {
|
||||
*out << 'v' << std::dec << NumElements;
|
||||
}
|
||||
|
||||
@@ -59,12 +59,12 @@ public:
|
||||
#define IROP_ALLOCATE_HELPERS
|
||||
#define IROP_DISPATCH_HELPERS
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
IRPair<IROp_Constant> _Constant(uint8_t Size, uint64_t Constant) {
|
||||
IRPair<IROp_Constant> _Constant(IR::OpSize Size, uint64_t Constant) {
|
||||
auto Op = AllocateOp<IROp_Constant, IROps::OP_CONSTANT>();
|
||||
uint64_t Mask = ~0ULL >> (64 - Size);
|
||||
uint64_t Mask = ~0ULL >> (64 - IR::OpSizeAsBits(Size));
|
||||
Op.first->Constant = (Constant & Mask);
|
||||
Op.first->Header.Size = Size / 8;
|
||||
Op.first->Header.ElementSize = Size / 8;
|
||||
Op.first->Header.Size = Size;
|
||||
Op.first->Header.ElementSize = Size;
|
||||
return Op;
|
||||
}
|
||||
IRPair<IROp_Jump> _Jump() {
|
||||
@@ -77,24 +77,24 @@ public:
|
||||
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
|
||||
}
|
||||
// TODO: Work to remove this implicit sized Select implementation.
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, Ref ssa0, Ref ssa1, Ref ssa2, Ref ssa3, uint8_t CompareSize = 0) {
|
||||
if (CompareSize == 0) {
|
||||
CompareSize = std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa0), GetOpSize(ssa1)));
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, Ref ssa0, Ref ssa1, Ref ssa2, Ref ssa3, IR::OpSize CompareSize = OpSize::iUnsized) {
|
||||
if (CompareSize == OpSize::iUnsized) {
|
||||
CompareSize = std::max(OpSize::i32Bit, std::max(GetOpSize(ssa0), GetOpSize(ssa1)));
|
||||
}
|
||||
|
||||
return _Select(IR::SizeToOpSize(std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa2), GetOpSize(ssa3)))),
|
||||
IR::SizeToOpSize(CompareSize), CondClassType {Cond}, ssa0, ssa1, ssa2, ssa3);
|
||||
return _Select(std::max(OpSize::i32Bit, std::max(GetOpSize(ssa2), GetOpSize(ssa3))), CompareSize, CondClassType {Cond}, ssa0, ssa1, ssa2, ssa3);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) {
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) {
|
||||
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) {
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMemTSO> _StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) {
|
||||
IRPair<IROp_StoreMemTSO>
|
||||
_StoreMemTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
Ref Invalid() {
|
||||
@@ -343,8 +343,13 @@ protected:
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
// MMX State can be either MMX (for 64bit) or x87 FPU (for 80bit)
|
||||
enum { MMXState_MMX, MMXState_X87 } MMXState = MMXState_MMX;
|
||||
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
virtual void RecordX87Use() {}
|
||||
virtual void ChgStateX87_MMX() {}
|
||||
virtual void ChgStateMMX_X87() {}
|
||||
virtual void SaveNZCV(IROps Op) {}
|
||||
|
||||
Ref CurrentWriteCursor = nullptr;
|
||||
|
||||
@@ -29,7 +29,7 @@ $end_info$
|
||||
namespace FEXCore::IR {
|
||||
|
||||
uint64_t getMask(IROp_Header* Op) {
|
||||
uint64_t NumBits = Op->Size * 8;
|
||||
uint64_t NumBits = IR::OpSizeAsBits(Op->Size);
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
|
||||
@@ -91,7 +91,7 @@ private:
|
||||
// We don't allow 8/16-bit operations to have constants, since no
|
||||
// constant would be in bounds after the JIT's 24/16 shift.
|
||||
auto Filter = [&IROp](uint64_t X) {
|
||||
return ARMEmitter::IsImmAddSub(X) && IROp->Size >= 4;
|
||||
return ARMEmitter::IsImmAddSub(X) && IROp->Size >= OpSize::i32Bit;
|
||||
};
|
||||
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, Filter);
|
||||
@@ -112,7 +112,7 @@ private:
|
||||
IsSIMM9 &= (SupportsTSOImm9 || !TSO);
|
||||
|
||||
// Extended offsets for regular loadstore only.
|
||||
bool IsExtended = (Imm & (IROp->Size - 1)) == 0 && Imm / IROp->Size <= 4095;
|
||||
bool IsExtended = (Imm & (IR::OpSizeToSize(IROp->Size) - 1)) == 0 && Imm / IR::OpSizeToSize(IROp->Size) <= 4095;
|
||||
IsExtended &= !TSO;
|
||||
|
||||
if (IsSIMM9 || IsExtended) {
|
||||
@@ -204,7 +204,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
/* IsImmAddSub assumes the constants are sign-extended, take care of that
|
||||
* here so we get the optimization for 32-bit adds too.
|
||||
*/
|
||||
if (Op->Header.Size == 4) {
|
||||
if (Op->Header.Size == OpSize::i32Bit) {
|
||||
Constant1 = (int64_t)(int32_t)Constant1;
|
||||
Constant2 = (int64_t)(int32_t)Constant2;
|
||||
}
|
||||
@@ -290,12 +290,12 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IROp->Size * 8); });
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IROp->Size * 8); });
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_XOR: {
|
||||
@@ -325,7 +325,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IROp->Size * 8); });
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
}
|
||||
}
|
||||
break;
|
||||
@@ -333,7 +333,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
case OP_ANDWITHFLAGS:
|
||||
case OP_ANDN:
|
||||
case OP_TESTNZ: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IROp->Size * 8); });
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_NEG: {
|
||||
@@ -356,7 +356,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31;
|
||||
uint64_t ShiftMask = IROp->Size == OpSize::i64Bit ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 << (Constant2 & ShiftMask)) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
@@ -384,7 +384,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
|
||||
if (IROp->Size <= 8 && IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
if (IROp->Size <= OpSize::i64Bit && IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
@@ -400,7 +400,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t DestSizeInBits = IROp->Size * 8;
|
||||
uint64_t DestSizeInBits = IR::OpSizeAsBits(IROp->Size);
|
||||
uint64_t DestMask = DestSizeInBits == 64 ? ~0ULL : ((1ULL << DestSizeInBits) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
@@ -424,11 +424,11 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
uint64_t NewConstant = SourceMask << Op->lsb;
|
||||
|
||||
if (ConstantSrc & 1) {
|
||||
auto orr = IREmit->_Or(IR::SizeToOpSize(IROp->Size), CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
auto orr = IREmit->_Or(IROp->Size, CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, orr);
|
||||
} else {
|
||||
// We are wanting to clear the bitfield.
|
||||
auto andn = IREmit->_Andn(IR::SizeToOpSize(IROp->Size), CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
auto andn = IREmit->_Andn(IROp->Size, CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, andn);
|
||||
}
|
||||
}
|
||||
@@ -596,7 +596,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
case OP_SELECT: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
uint64_t AllOnes = IROp->Size == OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
uint64_t Constant2 {};
|
||||
uint64_t Constant3 {};
|
||||
@@ -614,7 +614,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
// We always allow source 1 to be zero, but source 0 can only be a
|
||||
// special 1/~0 constant if source 1 is 0.
|
||||
if (InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 1)) {
|
||||
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
uint64_t AllOnes = IROp->Size == OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 0, [&AllOnes](uint64_t X) { return X == 1 || X == AllOnes; });
|
||||
}
|
||||
break;
|
||||
@@ -632,7 +632,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
auto EO = NewRIP->C<IR::IROp_EntrypointOffset>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(IR::SizeToOpSize(EO->Header.Size), EO->Offset));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(EO->Header.Size, EO->Offset));
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -79,12 +79,12 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
const auto ID = CurrentIR.GetID(CodeNode);
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
HadError |= OpSize == 0;
|
||||
HadError |= OpSize == IR::OpSize::iInvalid;
|
||||
// Does the op have a destination of size 0?
|
||||
if (OpSize == 0) {
|
||||
if (OpSize == IR::OpSize::iInvalid) {
|
||||
Errors << "%" << ID << ": Had destination but with no size" << std::endl;
|
||||
}
|
||||
|
||||
|
||||
@@ -319,6 +319,7 @@ constexpr FlagInfo ClassifyConst(IROps Op) {
|
||||
case OP_STOREAF: return FlagInfo::Pack({.Write = FLAG_A, .CanEliminate = true});
|
||||
|
||||
case OP_NZCVSELECT:
|
||||
case OP_NZCVSELECTV:
|
||||
case OP_NZCVSELECTINCREMENT:
|
||||
case OP_NEG:
|
||||
case OP_CONDJUMP:
|
||||
@@ -353,6 +354,11 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_NZCVSELECTV: {
|
||||
auto Op = IROp->CW<IR::IROp_NZCVSelectV>();
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
}
|
||||
|
||||
case OP_NEG: {
|
||||
auto Op = IROp->CW<IR::IROp_Neg>();
|
||||
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
|
||||
@@ -515,7 +521,7 @@ void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView&
|
||||
// Pattern match a branch fed by a compare. We could also handle bit tests
|
||||
// here, but tbz/tbnz has a limited offset range which we don't have a way to
|
||||
// deal with yet. Let's hope that's not a big deal.
|
||||
if (!(Op->Cond == COND_NEQ || Op->Cond == COND_EQ) || (Prev->Size < 4)) {
|
||||
if (!(Op->Cond == COND_NEQ || Op->Cond == COND_EQ) || (Prev->Size < OpSize::i32Bit)) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -606,7 +612,7 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie
|
||||
// this flag is outside of the if, since the TestNZ might result from
|
||||
// optimizing AndWithFlags, and we need to converge locally in a single
|
||||
// iteration.
|
||||
if (IROp->Op == OP_TESTNZ && IROp->Size < 4 && !(FlagsRead & (FLAG_N | FLAG_C))) {
|
||||
if (IROp->Op == OP_TESTNZ && IROp->Size < OpSize::i32Bit && !(FlagsRead & (FLAG_N | FLAG_C))) {
|
||||
IROp->Op = OP_TESTZ;
|
||||
}
|
||||
|
||||
|
||||
@@ -156,7 +156,6 @@ private:
|
||||
bool ReducedPrecisionMode;
|
||||
|
||||
// Helpers
|
||||
std::tuple<Ref, Ref> SplitF64SigExp(Ref Node);
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
|
||||
// Handles a Unary operation.
|
||||
@@ -284,7 +283,8 @@ inline void X87StackOptimization::MigrateToSlowPathIf(bool ShouldMigrate) {
|
||||
|
||||
inline Ref X87StackOptimization::GetTopWithCache_Slow() {
|
||||
if (!TopOffsetCache[0]) {
|
||||
TopOffsetCache[0] = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
TopOffsetCache[0] =
|
||||
IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
return TopOffsetCache[0];
|
||||
}
|
||||
@@ -306,31 +306,32 @@ inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset) {
|
||||
|
||||
|
||||
inline void X87StackOptimization::SetTopWithCache_Slow(Ref Value) {
|
||||
IREmit->_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
InvalidateTopOffsetCache();
|
||||
TopOffsetCache[0] = Value;
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::SetX87ValidTag(Ref Value, bool Valid) {
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), Value);
|
||||
Ref NewAbridgedFTW = Valid ? IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RegMask) : IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RegMask);
|
||||
IREmit->_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::GetX87ValidTag_Slow(uint8_t Offset) {
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
return IREmit->_And(OpSize::i32Bit, IREmit->_Lshr(OpSize::i32Bit, AbridgedFTW, GetOffsetTopWithCache_Slow(Offset)), GetConstant(1));
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::LoadStackValueAtOffset_Slow(uint8_t Offset) {
|
||||
return IREmit->_LoadContextIndexed(GetOffsetTopWithCache_Slow(Offset), ReducedPrecisionMode ? 8 : 16, MMBaseOffset(), 16, FPRClass);
|
||||
return IREmit->_LoadContextIndexed(GetOffsetTopWithCache_Slow(Offset), ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
|
||||
MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset, bool SetValid) {
|
||||
OrderedNode* TopOffset = GetOffsetTopWithCache_Slow(Offset);
|
||||
// store
|
||||
IREmit->_StoreContextIndexed(Value, TopOffset, ReducedPrecisionMode ? 8 : 16, MMBaseOffset(), 16, FPRClass);
|
||||
IREmit->_StoreContextIndexed(Value, TopOffset, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit, MMBaseOffset(), 16, FPRClass);
|
||||
// mark it valid
|
||||
// In some cases we might already know it has been previously set as valid so we don't need to do it again
|
||||
if (SetValid) {
|
||||
@@ -379,7 +380,7 @@ void X87StackOptimization::HandleUnop(IROps Op64, bool VFOp64, IROps Op80) {
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
if (VFOp64) {
|
||||
DeriveOp(Value, Op64, IREmit->_VFSqrt(8, 8, St0));
|
||||
DeriveOp(Value, Op64, IREmit->_VFSqrt(OpSize::i64Bit, OpSize::i64Bit, St0));
|
||||
} else {
|
||||
DeriveOp(Value, Op64, IREmit->_F64SIN(St0));
|
||||
}
|
||||
@@ -399,10 +400,10 @@ void X87StackOptimization::HandleBinopValue(IROps Op64, bool VFOp64, IROps Op80,
|
||||
Ref Node = {};
|
||||
if (ReducedPrecisionMode) {
|
||||
if (Reverse) {
|
||||
DeriveOp(Node, Op64, IREmit->_VFAdd(8, 8, ValueNode, StackNode));
|
||||
DeriveOp(Node, Op64, IREmit->_VFAdd(OpSize::i64Bit, OpSize::i64Bit, ValueNode, StackNode));
|
||||
} else {
|
||||
if (VFOp64) {
|
||||
DeriveOp(Node, Op64, IREmit->_VFAdd(8, 8, StackNode, ValueNode));
|
||||
DeriveOp(Node, Op64, IREmit->_VFAdd(OpSize::i64Bit, OpSize::i64Bit, StackNode, ValueNode));
|
||||
} else {
|
||||
DeriveOp(Node, Op64, IREmit->_F64FPREM(StackNode, ValueNode));
|
||||
}
|
||||
@@ -476,13 +477,14 @@ Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
}
|
||||
Ref TopIndex = GetOffsetTopWithCache_Slow(i);
|
||||
if (Valid == StackSlot::VALID) {
|
||||
IREmit->_StoreContextIndexed(StackMember.StackDataNode, TopIndex, ReducedPrecisionMode ? 8 : 16, MMBaseOffset(), 16, FPRClass);
|
||||
IREmit->_StoreContextIndexed(StackMember.StackDataNode, TopIndex, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
|
||||
MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
}
|
||||
{ // Set valid tags
|
||||
uint8_t Mask = StackData.getValidMask();
|
||||
if (Mask == 0xff) {
|
||||
IREmit->_StoreContext(1, GPRClass, GetConstant(Mask), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(Mask), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
} else if (Mask != 0) {
|
||||
if (std::popcount(Mask) == 1) {
|
||||
uint8_t BitIdx = __builtin_ctz(Mask);
|
||||
@@ -491,16 +493,16 @@ Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
// perform a rotate right on mask by top
|
||||
auto* TopValue = GetTopWithCache_Slow();
|
||||
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref NewAbridgedFTW = IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
|
||||
IREmit->_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
}
|
||||
}
|
||||
{ // Set invalid tags
|
||||
uint8_t Mask = StackData.getInvalidMask();
|
||||
if (Mask == 0xff) {
|
||||
IREmit->_StoreContext(1, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
} else if (Mask != 0) {
|
||||
if (std::popcount(Mask)) {
|
||||
uint8_t BitIdx = __builtin_ctz(Mask);
|
||||
@@ -509,29 +511,15 @@ Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
// Same rotate right as above but this time on the invalid mask
|
||||
auto* TopValue = GetTopWithCache_Slow();
|
||||
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref NewAbridgedFTW = IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
|
||||
IREmit->_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
}
|
||||
}
|
||||
return TopValue;
|
||||
}
|
||||
|
||||
std::tuple<Ref, Ref> X87StackOptimization::SplitF64SigExp(Ref Node) {
|
||||
Ref Gpr = IREmit->_VExtractToGPR(8, 8, Node, 0);
|
||||
|
||||
Ref Exp = IREmit->_And(OpSize::i64Bit, Gpr, GetConstant(0x7ff0000000000000LL));
|
||||
Exp = IREmit->_Lshr(OpSize::i64Bit, Exp, GetConstant(52));
|
||||
Exp = IREmit->_Sub(OpSize::i64Bit, Exp, GetConstant(1023));
|
||||
Exp = IREmit->_Float_FromGPR_S(8, 8, Exp);
|
||||
Ref Sig = IREmit->_And(OpSize::i64Bit, Gpr, GetConstant(0x800fffffffffffffLL));
|
||||
Sig = IREmit->_Or(OpSize::i64Bit, Sig, GetConstant(0x3ff0000000000000LL));
|
||||
Sig = IREmit->_VCastFromGPR(8, 8, Sig);
|
||||
|
||||
return std::tuple {Exp, Sig};
|
||||
}
|
||||
|
||||
void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::x87StackOpt");
|
||||
|
||||
@@ -662,9 +650,9 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
HandleUnop(OP_F64TAN, false, OP_F80TAN);
|
||||
Ref OneConst {};
|
||||
if (ReducedPrecisionMode) {
|
||||
OneConst = IREmit->_VCastFromGPR(8, 8, GetConstant(0x3FF0000000000000));
|
||||
OneConst = IREmit->_VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, GetConstant(0x3FF0000000000000));
|
||||
} else {
|
||||
OneConst = IREmit->_LoadNamedVectorConstant(16, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
OneConst = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
}
|
||||
|
||||
if (SlowPath) {
|
||||
@@ -723,7 +711,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
}
|
||||
} else { // invalidate all
|
||||
if (SlowPath) {
|
||||
IREmit->_StoreContext(1, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
} else {
|
||||
for (size_t i = 0; i < StackData.size; i++) {
|
||||
StackData.setTagInvalid(i);
|
||||
@@ -743,7 +731,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
} else {
|
||||
auto* SourceNode = CurrentIR.GetNode(Op->X80Src);
|
||||
auto* OriginalNode = CurrentIR.GetNode(Op->OriginalValue);
|
||||
StackData.push(StackMemberInfo {SourceNode, OriginalNode, SizeToOpSize(Op->LoadSize), Op->Float});
|
||||
StackData.push(StackMemberInfo {SourceNode, OriginalNode, Op->LoadSize, Op->Float});
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -809,33 +797,34 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
} else {
|
||||
if (ReducedPrecisionMode) {
|
||||
switch (Op->StoreSize) {
|
||||
case 4: {
|
||||
StackNode = IREmit->_Float_FToF(4, 8, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, 4, AddrNode, StackNode);
|
||||
case OpSize::i32Bit: {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i32Bit, AddrNode, StackNode);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
IREmit->_StoreMem(FPRClass, 8, AddrNode, StackNode);
|
||||
case OpSize::i64Bit: {
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
break;
|
||||
}
|
||||
case 10: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, 8);
|
||||
IREmit->_StoreMem(FPRClass, 8, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(16, 8, StackNode, 1);
|
||||
IREmit->_StoreMem(GPRClass, 2, Upper, AddrNode, GetConstant(8), 8, MEM_OFFSET_SXTX, 1);
|
||||
case OpSize::f80Bit: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, AddrNode, GetConstant(8), OpSize::i64Bit, MEM_OFFSET_SXTX, 1);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
}
|
||||
} else {
|
||||
if (Op->StoreSize != 10) { // if it's not 80bits then convert
|
||||
if (Op->StoreSize != OpSize::f80Bit) { // if it's not 80bits then convert
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
}
|
||||
if (Op->StoreSize == 10) { // Part of code from StoreResult_WithOpSize()
|
||||
if (Op->StoreSize == OpSize::f80Bit) { // Part of code from StoreResult_WithOpSize()
|
||||
// For X87 extended doubles, split before storing
|
||||
IREmit->_StoreMem(FPRClass, 8, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(16, 8, StackNode, 1);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, AddrNode, StackNode);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
auto DestAddr = IREmit->_Add(OpSize::i64Bit, AddrNode, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, 2, DestAddr, Upper, 8);
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, DestAddr, Upper, OpSize::i64Bit);
|
||||
} else {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, AddrNode, StackNode);
|
||||
}
|
||||
@@ -886,13 +875,13 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// of a value
|
||||
Ref ResultNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
ResultNode = IREmit->_VFNeg(8, 8, Value);
|
||||
ResultNode = IREmit->_VFNeg(OpSize::i64Bit, OpSize::i64Bit, Value);
|
||||
} else {
|
||||
Ref Low = GetConstant(0);
|
||||
Ref High = GetConstant(0b1'000'0000'0000'0000ULL);
|
||||
Ref HelperNode = IREmit->_VCastFromGPR(16, 8, Low);
|
||||
HelperNode = IREmit->_VInsGPR(16, 8, 1, HelperNode, High);
|
||||
ResultNode = IREmit->_VXor(16, 1, Value, HelperNode);
|
||||
Ref HelperNode = IREmit->_VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, Low);
|
||||
HelperNode = IREmit->_VInsGPR(OpSize::i128Bit, OpSize::i64Bit, 1, HelperNode, High);
|
||||
ResultNode = IREmit->_VXor(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
|
||||
}
|
||||
StoreStackValue(ResultNode);
|
||||
break;
|
||||
@@ -903,14 +892,14 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref ResultNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
ResultNode = IREmit->_VFAbs(8, 8, Value);
|
||||
ResultNode = IREmit->_VFAbs(OpSize::i64Bit, OpSize::i64Bit, Value);
|
||||
} else {
|
||||
// Intermediate insts
|
||||
Ref Low = GetConstant(~0ULL);
|
||||
Ref High = GetConstant(0b0'111'1111'1111'1111ULL);
|
||||
Ref HelperNode = IREmit->_VCastFromGPR(16, 8, Low);
|
||||
HelperNode = IREmit->_VInsGPR(16, 8, 1, HelperNode, High);
|
||||
ResultNode = IREmit->_VAnd(16, 1, Value, HelperNode);
|
||||
Ref HelperNode = IREmit->_VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, Low);
|
||||
HelperNode = IREmit->_VInsGPR(OpSize::i128Bit, OpSize::i64Bit, 1, HelperNode, High);
|
||||
ResultNode = IREmit->_VAnd(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
|
||||
}
|
||||
StoreStackValue(ResultNode);
|
||||
break;
|
||||
@@ -924,7 +913,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref CmpNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
CmpNode = IREmit->_FCmp(8, StackValue1, StackValue2);
|
||||
CmpNode = IREmit->_FCmp(OpSize::i64Bit, StackValue1, StackValue2);
|
||||
} else {
|
||||
CmpNode = IREmit->_F80Cmp(StackValue1, StackValue2);
|
||||
}
|
||||
@@ -936,11 +925,11 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
const auto* Op = IROp->C<IROp_F80StackTest>();
|
||||
auto Offset = Op->SrcStack;
|
||||
auto StackNode = LoadStackValue(Offset);
|
||||
Ref ZeroConst = IREmit->_VCastFromGPR(ReducedPrecisionMode ? 8 : 16, 8, GetConstant(0));
|
||||
Ref ZeroConst = IREmit->_VCastFromGPR(ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit, OpSize::i64Bit, GetConstant(0));
|
||||
|
||||
Ref CmpNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
CmpNode = IREmit->_FCmp(8, StackNode, ZeroConst);
|
||||
CmpNode = IREmit->_FCmp(OpSize::i64Bit, StackNode, ZeroConst);
|
||||
} else {
|
||||
CmpNode = IREmit->_F80Cmp(StackNode, ZeroConst);
|
||||
}
|
||||
@@ -956,7 +945,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref CmpNode {};
|
||||
if (ReducedPrecisionMode) {
|
||||
CmpNode = IREmit->_FCmp(8, StackNode, Value);
|
||||
CmpNode = IREmit->_FCmp(OpSize::i64Bit, StackNode, Value);
|
||||
} else {
|
||||
CmpNode = IREmit->_F80Cmp(StackNode, Value);
|
||||
}
|
||||
@@ -964,30 +953,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_F80XTRACTSTACK: {
|
||||
Ref St0 = LoadStackValue();
|
||||
|
||||
Ref Exp {};
|
||||
Ref Sig {};
|
||||
if (ReducedPrecisionMode) {
|
||||
std::tie(Exp, Sig) = SplitF64SigExp(St0);
|
||||
} else {
|
||||
Exp = IREmit->_F80XTRACT_EXP(St0);
|
||||
Sig = IREmit->_F80XTRACT_SIG(St0);
|
||||
}
|
||||
|
||||
if (SlowPath) {
|
||||
// Write exp to top, update top for a push and set sig at new top.
|
||||
StoreStackValueAtOffset_Slow(Exp, 0, false);
|
||||
UpdateTopForPush_Slow();
|
||||
StoreStackValueAtOffset_Slow(Sig);
|
||||
} else {
|
||||
StackData.setTop(StackMemberInfo {Exp});
|
||||
StackData.push(StackMemberInfo {Sig});
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_SYNCSTACKTOSLOW: {
|
||||
// This synchronizes stack values but doesn't necessarily moves us off the FastPath!
|
||||
Ref NewTop = SynchronizeStackValues();
|
||||
@@ -1023,7 +988,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref Value {};
|
||||
if (ReducedPrecisionMode) {
|
||||
Value = IREmit->_Vector_FToI(8, 8, St0, Round_Host);
|
||||
Value = IREmit->_Vector_FToI(OpSize::i64Bit, OpSize::i64Bit, St0, Round_Host);
|
||||
} else {
|
||||
Value = IREmit->_F80Round(St0);
|
||||
}
|
||||
@@ -1039,7 +1004,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
Ref Value1 = LoadStackValue(StackOffset1);
|
||||
Ref Value2 = LoadStackValue(StackOffset2);
|
||||
|
||||
Ref StackNode = IREmit->_VBSL(16, CurrentIR.GetNode(Op->VectorMask), Value1, Value2);
|
||||
Ref StackNode = IREmit->_VBSL(OpSize::i128Bit, CurrentIR.GetNode(Op->VectorMask), Value1, Value2);
|
||||
StoreStackValue(StackNode, 0, StackOffset1 && StackOffset2);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -32,14 +32,6 @@ namespace Alloc::OSAllocator {
|
||||
|
||||
thread_local FEXCore::Core::InternalThreadState* TLSThread {};
|
||||
|
||||
void RegisterTLSData(FEXCore::Core::InternalThreadState* Thread) {
|
||||
TLSThread = Thread;
|
||||
}
|
||||
|
||||
void UninstallTLSData(FEXCore::Core::InternalThreadState* Thread) {
|
||||
TLSThread = nullptr;
|
||||
}
|
||||
|
||||
class OSAllocator_64Bit final : public Alloc::HostAllocator {
|
||||
public:
|
||||
OSAllocator_64Bit();
|
||||
@@ -585,3 +577,13 @@ fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocator() {
|
||||
return fextl::make_unique<OSAllocator_64Bit>();
|
||||
}
|
||||
} // namespace Alloc::OSAllocator
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
void RegisterTLSData(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Alloc::OSAllocator::TLSThread = Thread;
|
||||
}
|
||||
|
||||
void UninstallTLSData(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Alloc::OSAllocator::TLSThread = nullptr;
|
||||
}
|
||||
} // namespace FEXCore::Allocator
|
||||
@@ -48,7 +48,5 @@ public:
|
||||
} // namespace Alloc
|
||||
|
||||
namespace Alloc::OSAllocator {
|
||||
void RegisterTLSData(FEXCore::Core::InternalThreadState* Thread);
|
||||
void UninstallTLSData(FEXCore::Core::InternalThreadState* Thread);
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocator();
|
||||
} // namespace Alloc::OSAllocator
|
||||
@@ -44,14 +44,6 @@ class IREmitter;
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class Context;
|
||||
enum ExitReason {
|
||||
EXIT_NONE,
|
||||
EXIT_WAITING,
|
||||
EXIT_ASYNC_RUN,
|
||||
EXIT_SHUTDOWN,
|
||||
EXIT_DEBUG,
|
||||
EXIT_UNKNOWNERROR,
|
||||
};
|
||||
|
||||
enum OperatingMode {
|
||||
MODE_32BIT,
|
||||
@@ -73,7 +65,7 @@ using CodeRangeInvalidationFn = std::function<void(uint64_t start, uint64_t Leng
|
||||
|
||||
using CustomIREntrypointHandler = std::function<void(uintptr_t Entrypoint, IR::IREmitter*)>;
|
||||
|
||||
using ExitHandler = std::function<void(Core::InternalThreadState* Thread, ExitReason)>;
|
||||
using ExitHandler = std::function<void(Core::InternalThreadState* Thread)>;
|
||||
|
||||
using AOTIRCodeFileWriterFn = std::function<void(const fextl::string& fileid, const fextl::string& filename)>;
|
||||
using AOTIRLoaderCBFn = std::function<int(const fextl::string&)>;
|
||||
@@ -102,21 +94,6 @@ public:
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual bool InitCore() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetExitHandler(ExitHandler handler) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual ExitHandler GetExitHandler() const = 0;
|
||||
|
||||
/**
|
||||
* @brief Runs the CPU core until it exits
|
||||
*
|
||||
* If an Exit handler has been registered, this function won't return until the core
|
||||
* has shutdown.
|
||||
*
|
||||
* @param CTX The context that we created
|
||||
*
|
||||
* @return The ExitReason for the parentthread.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual ExitReason RunUntilExit(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
|
||||
/**
|
||||
* @brief Executes the supplied thread context on the current thread until a return is requested
|
||||
*/
|
||||
@@ -168,8 +145,7 @@ public:
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::Core::InternalThreadState* CreateThread(
|
||||
uint64_t InitialRIP, uint64_t StackPointer, const FEXCore::Core::CPUState* NewThreadState = nullptr, uint64_t ParentTID = 0) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ExecutionThread(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void DestroyThread(FEXCore::Core::InternalThreadState* Thread, bool NeedsTLSUninstall = false) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void DestroyThread(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
#ifndef _WIN32
|
||||
FEX_DEFAULT_VISIBILITY virtual void LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {}
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState* Thread, bool Child) {}
|
||||
|
||||
@@ -35,6 +35,7 @@ struct HostFeatures {
|
||||
bool SupportsPreserveAllABI {};
|
||||
bool SupportsAES256 {};
|
||||
bool SupportsSVEBitPerm {};
|
||||
bool SupportsCPUIndexInTPIDRRO {};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
|
||||
@@ -15,14 +15,6 @@ namespace FEXCore {
|
||||
namespace Core {
|
||||
struct InternalThreadState;
|
||||
|
||||
enum class SignalEvent {
|
||||
Nothing, // If the guest uses our signal we need to know it was errant on our end
|
||||
Pause,
|
||||
Stop,
|
||||
Return,
|
||||
ReturnRT,
|
||||
};
|
||||
|
||||
enum SignalNumber {
|
||||
#ifndef _WIN32
|
||||
FAULT_SIGSEGV = SIGSEGV,
|
||||
@@ -78,14 +70,6 @@ public:
|
||||
return Config;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Signals a thread with a specific core event.
|
||||
*
|
||||
* @param Thread Which thread to signal.
|
||||
* @param Event Which event to signal the event with.
|
||||
*/
|
||||
virtual void SignalThread(FEXCore::Core::InternalThreadState* Thread, Core::SignalEvent Event) = 0;
|
||||
|
||||
protected:
|
||||
SignalDelegatorConfig Config;
|
||||
};
|
||||
|
||||
@@ -80,20 +80,7 @@ static_assert(!std::is_move_assignable_v<NonMovableUniquePtr<int>>);
|
||||
struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
FEXCore::Core::CpuStateFrame* const CurrentFrame = &BaseFrameState;
|
||||
|
||||
struct {
|
||||
std::atomic_bool Running {false};
|
||||
std::atomic_bool WaitingToStart {true};
|
||||
std::atomic_bool EarlyExit {false};
|
||||
std::atomic_bool ThreadSleeping {false};
|
||||
} RunningEvents;
|
||||
|
||||
FEXCore::Context::Context* CTX;
|
||||
std::atomic<SignalEvent> SignalReason {SignalEvent::Nothing};
|
||||
|
||||
NonMovableUniquePtr<FEXCore::Threads::Thread> ExecutionThread;
|
||||
bool StartPaused {false};
|
||||
InterruptableConditionVariable StartRunning;
|
||||
Event ThreadWaiting;
|
||||
FEXCore::Context::Context* const CTX;
|
||||
|
||||
NonMovableUniquePtr<FEXCore::IR::OpDispatchBuilder> OpDispatcher;
|
||||
|
||||
@@ -104,23 +91,10 @@ struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
NonMovableUniquePtr<FEXCore::IR::PassManager> PassManager;
|
||||
NonMovableUniquePtr<JITSymbolBuffer> SymbolBuffer;
|
||||
|
||||
int StatusCode {};
|
||||
FEXCore::Context::ExitReason ExitReason {FEXCore::Context::ExitReason::EXIT_WAITING};
|
||||
std::shared_ptr<FEXCore::CompileService> CompileService;
|
||||
|
||||
std::shared_mutex ObjectCacheRefCounter {};
|
||||
|
||||
struct DeferredSignalState {
|
||||
#ifndef _WIN32
|
||||
siginfo_t Info;
|
||||
#endif
|
||||
int Signal;
|
||||
};
|
||||
|
||||
// Queue of thread local signal frames that have been deferred.
|
||||
// Async signals aren't guaranteed to be delivered in any particular order, but FEX treats them as FILO.
|
||||
fextl::vector<DeferredSignalState> DeferredSignalFrames;
|
||||
|
||||
///< Data pointer for exclusive use by the frontend
|
||||
void* FrontendPtr;
|
||||
|
||||
|
||||
@@ -9,6 +9,10 @@
|
||||
#include <optional>
|
||||
#include <sys/types.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
FEX_DEFAULT_VISIBILITY void SetupHooks();
|
||||
FEX_DEFAULT_VISIBILITY void ClearHooks();
|
||||
@@ -83,4 +87,9 @@ FEX_DEFAULT_VISIBILITY void ReclaimMemoryRegion(const fextl::vector<MemoryRegion
|
||||
// Use this to reserve the top 128TB of VA so the guest never see it
|
||||
// Returns nullptr on host VA < 48bits
|
||||
FEX_DEFAULT_VISIBILITY fextl::vector<MemoryRegion> Steal48BitVA();
|
||||
|
||||
#ifndef _WIN32
|
||||
FEX_DEFAULT_VISIBILITY void RegisterTLSData(FEXCore::Core::InternalThreadState* Thread);
|
||||
FEX_DEFAULT_VISIBILITY void UninstallTLSData(FEXCore::Core::InternalThreadState* Thread);
|
||||
#endif
|
||||
} // namespace FEXCore::Allocator
|
||||
@@ -124,6 +124,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: System: System register move") {
|
||||
TEST_SINGLE(msr(SystemRegister::RNDRRS, Reg::r30), "msr rndrrs, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::NZCV, Reg::r30), "msr nzcv, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::FPCR, Reg::r30), "msr fpcr, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::TPIDRRO_EL0, Reg::r30), "msr S3_3_c13_c0_3, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::CNTFRQ_EL0, Reg::r30), "msr S3_3_c14_c0_0, x30");
|
||||
TEST_SINGLE(msr(SystemRegister::CNTVCT_EL0, Reg::r30), "msr S3_3_c14_c0_2, x30");
|
||||
|
||||
@@ -134,6 +135,7 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: System: System register move") {
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::RNDRRS), "mrs x30, rndrrs");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::NZCV), "mrs x30, nzcv");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::FPCR), "mrs x30, fpcr");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::TPIDRRO_EL0), "mrs x30, S3_3_c13_c0_3");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::CNTFRQ_EL0), "mrs x30, S3_3_c14_c0_0");
|
||||
TEST_SINGLE(mrs(Reg::r30, SystemRegister::CNTVCT_EL0), "mrs x30, S3_3_c14_c0_2");
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <string_view>
|
||||
|
||||
namespace FHU {
|
||||
|
||||
/**
|
||||
* @brief Parses a string of arguments, returning a vector of string_views.
|
||||
*
|
||||
* @param ArgumentString The string of arguments to parse
|
||||
*
|
||||
* @return The array of parsed arguments
|
||||
*/
|
||||
static inline fextl::vector<std::string_view> ParseArgumentsFromString(const std::string_view ArgumentString) {
|
||||
fextl::vector<std::string_view> Arguments;
|
||||
|
||||
auto Begin = ArgumentString.begin();
|
||||
auto ArgEnd = Begin;
|
||||
const auto End = ArgumentString.end();
|
||||
while (ArgEnd != End && Begin != End) {
|
||||
// The end of an argument ends with a space or the end of the interpreter line.
|
||||
ArgEnd = std::find(Begin, End, ' ');
|
||||
|
||||
if (Begin != ArgEnd) {
|
||||
const auto View = std::string_view(Begin, ArgEnd - Begin);
|
||||
if (!View.empty()) {
|
||||
Arguments.emplace_back(View);
|
||||
}
|
||||
}
|
||||
|
||||
Begin = ArgEnd + 1;
|
||||
}
|
||||
|
||||
return Arguments;
|
||||
}
|
||||
} // namespace FHU
|
||||
@@ -7,7 +7,7 @@ FEX is very much work in progress, so expect things to change.
|
||||
|
||||
|
||||
## Quick start guide
|
||||
### For Ubuntu 20.04, 21.04, 21.10, 22.04
|
||||
### For Ubuntu 22.04, 24.04 and 24.10
|
||||
Execute the following command in the terminal to install FEX through a PPA.
|
||||
|
||||
`curl --silent https://raw.githubusercontent.com/FEX-Emu/FEX/main/Scripts/InstallFEX.py --output /tmp/InstallFEX.py && python3 /tmp/InstallFEX.py && rm /tmp/InstallFEX.py`
|
||||
@@ -22,7 +22,12 @@ Please see [Building FEX](#building-fex).
|
||||
## Getting Started
|
||||
FEX has been tested to build and run on ARMv8.0+ hardware.
|
||||
ARMv7 hardware will not work.
|
||||
Expected operating system usage is Linux. FEX has been tested with Ubuntu 20.04, 20.10, and 21.04. Also Arch Linux.
|
||||
Expected operating system usage is Linux. FEX has been tested with the following Linux OSes:
|
||||
|
||||
- Ubuntu 22.04
|
||||
- Ubuntu 24.04
|
||||
- Ubuntu 24.10
|
||||
- Arch Linux
|
||||
|
||||
On AArch64 hosts the user **MUST** have an x86-64 RootFS [Creating a RootFS](#RootFS-Generation).
|
||||
|
||||
|
||||
@@ -83,9 +83,8 @@ def IsSupportedDistro():
|
||||
if Distro[0] == "ubuntu":
|
||||
# We only support what is available in ppa:fex-emu/fex
|
||||
return Distro[1] == "22.04" or \
|
||||
Distro[1] == "23.04" or \
|
||||
Distro[1] == "23.10" or \
|
||||
Distro[1] == "24.04"
|
||||
Distro[1] == "24.04" or \
|
||||
Distro[1] == "24.10"
|
||||
|
||||
return False
|
||||
|
||||
|
||||
@@ -337,7 +337,7 @@ fextl::string RecoverGuestProgramFilename(fextl::string Program, bool ExecFDInte
|
||||
return Program;
|
||||
}
|
||||
|
||||
ApplicationNames GetApplicationNames(fextl::vector<fextl::string> Args, bool ExecFDInterp, int ProgramFDFromEnv) {
|
||||
ApplicationNames GetApplicationNames(const fextl::vector<fextl::string>& Args, bool ExecFDInterp, int ProgramFDFromEnv) {
|
||||
if (Args.empty()) {
|
||||
// Early exit if we weren't passed an argument
|
||||
return {};
|
||||
@@ -346,8 +346,7 @@ ApplicationNames GetApplicationNames(fextl::vector<fextl::string> Args, bool Exe
|
||||
fextl::string Program {};
|
||||
fextl::string ProgramName {};
|
||||
|
||||
Args[0] = RecoverGuestProgramFilename(std::move(Args[0]), ExecFDInterp, ProgramFDFromEnv);
|
||||
Program = Args[0];
|
||||
Program = RecoverGuestProgramFilename(Args[0], ExecFDInterp, ProgramFDFromEnv);
|
||||
|
||||
bool Wine = false;
|
||||
for (size_t CurrentProgramNameIndex = 0; CurrentProgramNameIndex < Args.size(); ++CurrentProgramNameIndex) {
|
||||
@@ -440,17 +439,17 @@ const char* GetHomeDirectory() {
|
||||
const char* HomeDir = getenv("HOME");
|
||||
|
||||
// Try to get home directory from uid
|
||||
if (!HomeDir) {
|
||||
if (!HomeDir || !FHU::Filesystem::Exists(HomeDir)) {
|
||||
HomeDir = FindUserHomeThroughUID();
|
||||
}
|
||||
|
||||
// try the PWD
|
||||
if (!HomeDir) {
|
||||
if (!HomeDir || !FHU::Filesystem::Exists(HomeDir)) {
|
||||
HomeDir = getenv("PWD");
|
||||
}
|
||||
|
||||
// Still doesn't exit? You get local
|
||||
if (!HomeDir) {
|
||||
if (!HomeDir || !FHU::Filesystem::Exists(HomeDir)) {
|
||||
HomeDir = ".";
|
||||
}
|
||||
|
||||
|
||||
@@ -40,7 +40,7 @@ struct PortableInformation {
|
||||
*
|
||||
* @return The application name and path structure
|
||||
*/
|
||||
ApplicationNames GetApplicationNames(fextl::vector<fextl::string> Args, bool ExecFDInterp, int ProgramFDFromEnv);
|
||||
ApplicationNames GetApplicationNames(const fextl::vector<fextl::string>& Args, bool ExecFDInterp, int ProgramFDFromEnv);
|
||||
|
||||
/**
|
||||
* @brief Loads the FEX and application configurations for the application that is getting ready to run.
|
||||
|
||||
@@ -96,14 +96,21 @@ fextl::string GetServerRootFSLockFile() {
|
||||
}
|
||||
|
||||
fextl::string GetTempFolder() {
|
||||
auto XDGRuntimeEnv = getenv("XDG_RUNTIME_DIR");
|
||||
if (XDGRuntimeEnv) {
|
||||
// If the XDG runtime directory works then use that.
|
||||
return XDGRuntimeEnv;
|
||||
const std::array<const char*, 5> Vars = {
|
||||
"XDG_RUNTIME_DIR", "TMPDIR", "TMP", "TEMP", "TEMPDIR",
|
||||
};
|
||||
|
||||
for (auto& Var : Vars) {
|
||||
auto Path = getenv(Var);
|
||||
if (Path) {
|
||||
// If one of the env variable-driven paths works then use that.
|
||||
return Path;
|
||||
}
|
||||
}
|
||||
// Fallback to `/tmp/` if XDG_RUNTIME_DIR doesn't exist.
|
||||
|
||||
// Fallback to `/tmp/` if no env vars are set.
|
||||
// Might not be ideal but we don't have much of a choice.
|
||||
return fextl::string {std::filesystem::temp_directory_path().string()};
|
||||
return fextl::string {"/tmp"};
|
||||
}
|
||||
|
||||
fextl::string GetServerMountFolder() {
|
||||
@@ -143,6 +150,24 @@ fextl::string GetServerSocketName() {
|
||||
return ServerSocketPath;
|
||||
}
|
||||
|
||||
fextl::string GetServerSocketPath() {
|
||||
FEX_CONFIG_OPT(ServerSocketPath, SERVERSOCKETPATH);
|
||||
|
||||
auto name = ServerSocketPath();
|
||||
|
||||
if (name.starts_with("/")) {
|
||||
return name;
|
||||
}
|
||||
|
||||
auto Folder = GetTempFolder();
|
||||
|
||||
if (name.empty()) {
|
||||
return fextl::fmt::format("{}/{}.FEXServer.Socket", Folder, ::geteuid());
|
||||
} else {
|
||||
return fextl::fmt::format("{}/{}", Folder, name);
|
||||
}
|
||||
}
|
||||
|
||||
int GetServerFD() {
|
||||
return ServerFD;
|
||||
}
|
||||
@@ -153,7 +178,7 @@ int ConnectToServer(ConnectionOption ConnectionOption) {
|
||||
// Create the initial unix socket
|
||||
int SocketFD = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (SocketFD == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't open AF_UNIX socket {} {}", errno, strerror(errno));
|
||||
LogMan::Msg::EFmt("Couldn't open AF_UNIX socket {}", errno);
|
||||
return -1;
|
||||
}
|
||||
|
||||
@@ -170,13 +195,29 @@ int ConnectToServer(ConnectionOption ConnectionOption) {
|
||||
|
||||
if (connect(SocketFD, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr) == -1) {
|
||||
if (ConnectionOption == ConnectionOption::Default || errno != ECONNREFUSED) {
|
||||
LogMan::Msg::EFmt("Couldn't connect to FEXServer socket {} {} {}", ServerSocketName, errno, strerror(errno));
|
||||
LogMan::Msg::EFmt("Couldn't connect to FEXServer socket {} {}", ServerSocketName, errno);
|
||||
}
|
||||
close(SocketFD);
|
||||
return -1;
|
||||
} else {
|
||||
return SocketFD;
|
||||
}
|
||||
|
||||
return SocketFD;
|
||||
// Try again with a path-based socket, since abstract sockets will fail if we have been
|
||||
// placed in a new netns as part of a sandbox.
|
||||
auto ServerSocketPath = GetServerSocketPath();
|
||||
|
||||
SizeOfSocketString = std::min(ServerSocketPath.size(), sizeof(addr.sun_path) - 1);
|
||||
strncpy(addr.sun_path, ServerSocketPath.data(), SizeOfSocketString);
|
||||
SizeOfAddr = sizeof(addr.sun_family) + SizeOfSocketString;
|
||||
if (connect(SocketFD, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr) == -1) {
|
||||
if (ConnectionOption == ConnectionOption::Default || (errno != ECONNREFUSED && errno != ENOENT)) {
|
||||
LogMan::Msg::EFmt("Couldn't connect to FEXServer socket {} {}", ServerSocketPath, errno);
|
||||
}
|
||||
} else {
|
||||
return SocketFD;
|
||||
}
|
||||
|
||||
close(SocketFD);
|
||||
return -1;
|
||||
}
|
||||
|
||||
bool SetupClient(char* InterpreterPath) {
|
||||
|
||||
@@ -53,6 +53,7 @@ fextl::string GetServerRootFSLockFile();
|
||||
fextl::string GetTempFolder();
|
||||
fextl::string GetServerMountFolder();
|
||||
fextl::string GetServerSocketName();
|
||||
fextl::string GetServerSocketPath();
|
||||
int GetServerFD();
|
||||
|
||||
bool SetupClient(char* InterpreterPath);
|
||||
|
||||
@@ -630,6 +630,8 @@ FEXCore::HostFeatures FetchHostFeatures() {
|
||||
|
||||
auto HostFeatures = FetchHostFeatures(Features, true, CTR, MIDR);
|
||||
FillMIDRInformationViaLinux(&HostFeatures);
|
||||
|
||||
HostFeatures.SupportsCPUIndexInTPIDRRO = false;
|
||||
return HostFeatures;
|
||||
}
|
||||
} // namespace FEX
|
||||
@@ -3,7 +3,7 @@
|
||||
|
||||
namespace FEX::StringUtil {
|
||||
void ltrim(fextl::string& s) {
|
||||
s.erase(std::find_if(s.begin(), s.end(), [](int ch) { return !std::isspace(ch); }));
|
||||
s.erase(s.begin(), std::find_if(s.begin(), s.end(), [](int ch) { return !std::isspace(ch); }));
|
||||
}
|
||||
|
||||
void rtrim(fextl::string& s) {
|
||||
|
||||
@@ -25,8 +25,6 @@ public:
|
||||
|
||||
class DummySignalDelegator final : public FEXCore::SignalDelegator, public FEXCore::Allocator::FEXAllocOperators {
|
||||
public:
|
||||
void SignalThread(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::SignalEvent Event) override {}
|
||||
|
||||
FEXCore::Core::InternalThreadState* GetBackingTLSThread() {
|
||||
return GetTLSThread();
|
||||
}
|
||||
|
||||
@@ -725,6 +725,9 @@ public:
|
||||
uint64_t ExecFNLocation = TotalArgumentMemSize;
|
||||
TotalArgumentMemSize += Args[0].size() + 1;
|
||||
|
||||
// Align the argument block to 16 bytes to keep the stack aligned
|
||||
TotalArgumentMemSize = FEXCore::AlignUp(TotalArgumentMemSize, 16);
|
||||
|
||||
// Offset the stack by how much memory we need
|
||||
StackPointer -= TotalArgumentMemSize;
|
||||
|
||||
|
||||
@@ -38,6 +38,7 @@ $end_info$
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
#include <FEXHeaderUtils/StringArgumentParser.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cerrno>
|
||||
@@ -135,63 +136,44 @@ private:
|
||||
};
|
||||
} // namespace AOTIR
|
||||
|
||||
void InterpreterHandler(fextl::string* Filename, const fextl::string& RootFS, fextl::vector<fextl::string>* args) {
|
||||
// Open the Filename to determine if it is a shebang file.
|
||||
int FD = open(Filename->c_str(), O_RDONLY | O_CLOEXEC);
|
||||
bool InterpreterHandler(fextl::string* Filename, const fextl::string& RootFS, fextl::vector<fextl::string>* args) {
|
||||
int FD {-1};
|
||||
|
||||
// Attempt to open the filename from the rootfs first.
|
||||
FD = open(fextl::fmt::format("{}{}", RootFS, *Filename).c_str(), O_RDONLY | O_CLOEXEC);
|
||||
if (FD == -1) {
|
||||
return;
|
||||
// Failing that, attempt to open the filename directly.
|
||||
FD = open(Filename->c_str(), O_RDONLY | O_CLOEXEC);
|
||||
if (FD == -1) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
std::array<char, 257> Header;
|
||||
const auto ChunkSize = 257l;
|
||||
const auto ReadSize = pread(FD, &Header.at(0), ChunkSize, 0);
|
||||
close(FD);
|
||||
|
||||
const auto Data = std::span<char>(Header.data(), ReadSize);
|
||||
|
||||
// Is the file large enough for shebang
|
||||
if (ReadSize <= 2) {
|
||||
close(FD);
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
|
||||
// Handle shebang files
|
||||
if (Data[0] == '#' && Data[1] == '!') {
|
||||
fextl::string InterpreterLine {Data.begin() + 2, // strip off "#!" prefix
|
||||
std::find(Data.begin(), Data.end(), '\n')};
|
||||
fextl::vector<fextl::string> ShebangArguments {};
|
||||
|
||||
// Shebang line can have a single argument
|
||||
fextl::istringstream InterpreterSS(InterpreterLine);
|
||||
fextl::string Argument;
|
||||
while (std::getline(InterpreterSS, Argument, ' ')) {
|
||||
if (Argument.empty()) {
|
||||
continue;
|
||||
}
|
||||
ShebangArguments.push_back(std::move(Argument));
|
||||
}
|
||||
std::string_view InterpreterLine {Data.begin() + 2, // strip off "#!" prefix
|
||||
std::find(Data.begin(), Data.end(), '\n')};
|
||||
const auto ShebangArguments = FHU::ParseArgumentsFromString(InterpreterLine);
|
||||
|
||||
// Executable argument
|
||||
fextl::string& ShebangProgram = ShebangArguments[0];
|
||||
|
||||
// If the filename is absolute then prepend the rootfs
|
||||
// If it is relative then don't append the rootfs
|
||||
if (ShebangProgram[0] == '/') {
|
||||
ShebangProgram = RootFS + ShebangProgram;
|
||||
}
|
||||
*Filename = ShebangProgram;
|
||||
*Filename = ShebangArguments.at(0);
|
||||
|
||||
// Insert all the arguments at the start
|
||||
args->insert(args->begin(), ShebangArguments.begin(), ShebangArguments.end());
|
||||
}
|
||||
close(FD);
|
||||
}
|
||||
|
||||
void RootFSRedirect(fextl::string* Filename, const fextl::string& RootFS) {
|
||||
auto RootFSLink = ELFCodeLoader::ResolveRootfsFile(*Filename, RootFS);
|
||||
|
||||
if (FHU::Filesystem::Exists(RootFSLink)) {
|
||||
*Filename = RootFSLink;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
FEX::Config::PortableInformation ReadPortabilityInformation() {
|
||||
@@ -287,6 +269,36 @@ void SetupTSOEmulation(FEXCore::Context::Context* CTX) {
|
||||
}
|
||||
} // namespace FEX::TSO
|
||||
|
||||
namespace FEX::CompatInput {
|
||||
void SetupCompatInput(bool enable) {
|
||||
// We need to check if these are defined or not. This is a very fresh feature.
|
||||
#ifndef PR_GET_COMPAT_INPUT
|
||||
#define PR_GET_COMPAT_INPUT 0x63494e50
|
||||
#endif
|
||||
#ifndef PR_SET_COMPAT_INPUT
|
||||
#define PR_SET_COMPAT_INPUT 0x43494e50
|
||||
#endif
|
||||
#ifndef PR_SET_COMPAT_INPUT_DISABLE
|
||||
#define PR_SET_COMPAT_INPUT_DISABLE 0
|
||||
#endif
|
||||
#ifndef PR_SET_COMPAT_INPUT_ENABLE
|
||||
#define PR_SET_COMPAT_INPUT_ENABLE 1
|
||||
#endif
|
||||
// Check to see if this is supported.
|
||||
auto Result = prctl(PR_GET_COMPAT_INPUT, 0, 0, 0, 0);
|
||||
if (Result == -1) {
|
||||
// Unsupported, early exit.
|
||||
return;
|
||||
}
|
||||
|
||||
if (enable) {
|
||||
prctl(PR_SET_COMPAT_INPUT, PR_SET_COMPAT_INPUT_ENABLE, 0, 0, 0);
|
||||
} else {
|
||||
prctl(PR_SET_COMPAT_INPUT, PR_SET_COMPAT_INPUT_DISABLE, 0, 0, 0);
|
||||
}
|
||||
}
|
||||
} // namespace FEX::CompatInput
|
||||
|
||||
/**
|
||||
* @brief Get an FD from an environment variable and then unset the environment variable.
|
||||
*
|
||||
@@ -405,10 +417,21 @@ int main(int argc, char** argv, char** const envp) {
|
||||
FEXCore::Profiler::Init();
|
||||
FEXCore::Telemetry::Initialize();
|
||||
|
||||
RootFSRedirect(&Program.ProgramPath, LDPath());
|
||||
InterpreterHandler(&Program.ProgramPath, LDPath(), &Args);
|
||||
if (!LDPath().empty() && Program.ProgramPath.starts_with(LDPath())) {
|
||||
// From this point on, ProgramPath needs to not have the LDPath prefixed on to it.
|
||||
auto RootFSLength = LDPath().size();
|
||||
if (Program.ProgramPath.at(RootFSLength) != '/') {
|
||||
// Ensure the modified path starts as an absolute path.
|
||||
// This edge case can occur when ROOTFS ends with '/' and passed a path like `<ROOTFS>usr/bin/true`.
|
||||
--RootFSLength;
|
||||
}
|
||||
|
||||
if (!ExecutedWithFD && FEXFD == -1 && !FHU::Filesystem::Exists(Program.ProgramPath)) {
|
||||
Program.ProgramPath.erase(0, RootFSLength);
|
||||
}
|
||||
|
||||
bool ProgramExists = InterpreterHandler(&Program.ProgramPath, LDPath(), &Args);
|
||||
|
||||
if (!ExecutedWithFD && FEXFD == -1 && !ProgramExists) {
|
||||
// Early exit if the program passed in doesn't exist
|
||||
// Will prevent a crash later
|
||||
fextl::fmt::print(stderr, "{}: command not found\n", Program.ProgramPath);
|
||||
@@ -510,6 +533,16 @@ int main(int argc, char** argv, char** const envp) {
|
||||
// Setup TSO hardware emulation immediately after initializing the context.
|
||||
FEX::TSO::SetupTSOEmulation(CTX.get());
|
||||
|
||||
if (!Loader.Is64BitMode()) {
|
||||
// Tell the kernel we want to use the compat input syscalls even though we're
|
||||
// a 64 bit process.
|
||||
FEX::CompatInput::SetupCompatInput(true);
|
||||
} else {
|
||||
// Our parent could be an instance running a 32 bit application, so we need
|
||||
// to disable compat input if we're running a 64 bit one ourselves.
|
||||
FEX::CompatInput::SetupCompatInput(false);
|
||||
}
|
||||
|
||||
auto SignalDelegation = FEX::HLE::CreateSignalDelegator(CTX.get(), Program.ProgramName, SupportsAVX);
|
||||
auto ThunkHandler = FEX::HLE::CreateThunkHandler();
|
||||
|
||||
@@ -571,18 +604,6 @@ int main(int argc, char** argv, char** const envp) {
|
||||
|
||||
SyscallHandler->DeserializeSeccompFD(ParentThread, FEXSeccompFD);
|
||||
|
||||
FEXCore::Context::ExitReason ShutdownReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN;
|
||||
|
||||
// There might already be an exit handler, leave it installed
|
||||
if (!CTX->GetExitHandler()) {
|
||||
CTX->SetExitHandler([&](FEXCore::Core::InternalThreadState* Thread, FEXCore::Context::ExitReason reason) {
|
||||
if (reason != FEXCore::Context::ExitReason::EXIT_DEBUG) {
|
||||
ShutdownReason = reason;
|
||||
SyscallHandler->TM.Stop();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
const bool AOTEnabled = AOTIRLoad() || AOTIRCapture() || AOTIRGenerate();
|
||||
if (AOTEnabled) {
|
||||
LogMan::Msg::IFmt("Warning: AOTIR is experimental, and might lead to crashes. "
|
||||
@@ -620,9 +641,12 @@ int main(int argc, char** argv, char** const envp) {
|
||||
FEX::AOT::AOTGenSection(CTX.get(), Section);
|
||||
}
|
||||
} else {
|
||||
CTX->RunUntilExit(ParentThread->Thread);
|
||||
CTX->ExecuteThread(ParentThread->Thread);
|
||||
}
|
||||
|
||||
DebugServer.reset();
|
||||
SyscallHandler->TM.Stop();
|
||||
|
||||
if (AOTEnabled) {
|
||||
if (FHU::Filesystem::CreateDirectories(fextl::fmt::format("{}/aotir", FEXCore::Config::GetDataDirectory()))) {
|
||||
CTX->WriteFilesWithCode([](const fextl::string& fileid, const fextl::string& filename) {
|
||||
@@ -641,7 +665,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
}
|
||||
|
||||
auto ProgramStatus = ParentThread->Thread->StatusCode;
|
||||
auto ProgramStatus = ParentThread->StatusCode;
|
||||
|
||||
SignalDelegation->UninstallTLSState(ParentThread);
|
||||
SyscallHandler->TM.DestroyThread(ParentThread);
|
||||
@@ -671,9 +695,5 @@ int main(int argc, char** argv, char** const envp) {
|
||||
|
||||
FEXCore::Allocator::ReenableSBRKAllocations(SBRKPointer);
|
||||
|
||||
if (ShutdownReason == FEXCore::Context::ExitReason::EXIT_SHUTDOWN) {
|
||||
return ProgramStatus;
|
||||
} else {
|
||||
return -64 | ShutdownReason;
|
||||
}
|
||||
return ProgramStatus;
|
||||
}
|
||||
@@ -163,7 +163,13 @@ int main(int argc, char** argv, char** const envp) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
if (!ProcessPipe::InitializeServerSocket()) {
|
||||
if (!ProcessPipe::InitializeServerSocket(true)) {
|
||||
// Couldn't create server socket for some reason
|
||||
PipeScanner::ClosePipes();
|
||||
return -1;
|
||||
}
|
||||
|
||||
if (!ProcessPipe::InitializeServerSocket(false)) {
|
||||
// Couldn't create server socket for some reason
|
||||
PipeScanner::ClosePipes();
|
||||
return -1;
|
||||
|
||||
@@ -19,6 +19,7 @@ namespace ProcessPipe {
|
||||
constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
int ServerLockFD {-1};
|
||||
int ServerSocketFD {-1};
|
||||
int ServerFSSocketFD {-1};
|
||||
std::atomic<bool> ShouldShutdown {false};
|
||||
time_t RequestTimeout {10};
|
||||
bool Foreground {false};
|
||||
@@ -175,40 +176,58 @@ bool InitializeServerPipe() {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool InitializeServerSocket() {
|
||||
auto ServerSocketName = FEXServerClient::GetServerSocketName();
|
||||
bool InitializeServerSocket(bool abstract) {
|
||||
|
||||
// Create the initial unix socket
|
||||
ServerSocketFD = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (ServerSocketFD == -1) {
|
||||
int fd = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0);
|
||||
if (fd == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't create AF_UNIX socket: {} {}\n", errno, strerror(errno));
|
||||
return false;
|
||||
}
|
||||
|
||||
struct sockaddr_un addr {};
|
||||
addr.sun_family = AF_UNIX;
|
||||
size_t SizeOfSocketString = std::min(ServerSocketName.size() + 1, sizeof(addr.sun_path) - 1);
|
||||
addr.sun_path[0] = 0; // Abstract AF_UNIX sockets start with \0
|
||||
strncpy(addr.sun_path + 1, ServerSocketName.data(), SizeOfSocketString);
|
||||
|
||||
size_t SizeOfSocketString;
|
||||
if (abstract) {
|
||||
auto ServerSocketName = FEXServerClient::GetServerSocketName();
|
||||
SizeOfSocketString = std::min(ServerSocketName.size() + 1, sizeof(addr.sun_path) - 1);
|
||||
addr.sun_path[0] = 0; // Abstract AF_UNIX sockets start with \0
|
||||
strncpy(addr.sun_path + 1, ServerSocketName.data(), SizeOfSocketString);
|
||||
} else {
|
||||
auto ServerSocketPath = FEXServerClient::GetServerSocketPath();
|
||||
// Unlink the socket file if it exists
|
||||
// We are being asked to create a daemon, not error check
|
||||
// We don't care if this failed or not
|
||||
unlink(ServerSocketPath.c_str());
|
||||
|
||||
SizeOfSocketString = std::min(ServerSocketPath.size(), sizeof(addr.sun_path) - 1);
|
||||
strncpy(addr.sun_path, ServerSocketPath.data(), SizeOfSocketString);
|
||||
}
|
||||
// Include final null character.
|
||||
size_t SizeOfAddr = sizeof(addr.sun_family) + SizeOfSocketString;
|
||||
|
||||
// Bind the socket to the path
|
||||
int Result = bind(ServerSocketFD, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr);
|
||||
int Result = bind(fd, reinterpret_cast<struct sockaddr*>(&addr), SizeOfAddr);
|
||||
if (Result == -1) {
|
||||
LogMan::Msg::EFmt("Couldn't bind AF_UNIX socket '{}': {} {}\n", addr.sun_path, errno, strerror(errno));
|
||||
close(ServerSocketFD);
|
||||
ServerSocketFD = -1;
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
|
||||
listen(ServerSocketFD, 16);
|
||||
listen(fd, 16);
|
||||
PollFDs.emplace_back(pollfd {
|
||||
.fd = ServerSocketFD,
|
||||
.fd = fd,
|
||||
.events = POLLIN,
|
||||
.revents = 0,
|
||||
});
|
||||
|
||||
if (abstract) {
|
||||
ServerSocketFD = fd;
|
||||
} else {
|
||||
ServerFSSocketFD = fd;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -422,6 +441,7 @@ void CloseConnections() {
|
||||
|
||||
// Close the server socket so no more connections can be started
|
||||
close(ServerSocketFD);
|
||||
close(ServerFSSocketFD);
|
||||
}
|
||||
|
||||
void WaitForRequests() {
|
||||
@@ -441,12 +461,12 @@ void WaitForRequests() {
|
||||
bool Erase {};
|
||||
|
||||
if (Event.revents != 0) {
|
||||
if (Event.fd == ServerSocketFD) {
|
||||
if (Event.fd == ServerSocketFD || Event.fd == ServerFSSocketFD) {
|
||||
if (Event.revents & POLLIN) {
|
||||
// If it is the listen socket then we have a new connection
|
||||
struct sockaddr_storage Addr {};
|
||||
socklen_t AddrSize {};
|
||||
int NewFD = accept(ServerSocketFD, reinterpret_cast<struct sockaddr*>(&Addr), &AddrSize);
|
||||
int NewFD = accept(Event.fd, reinterpret_cast<struct sockaddr*>(&Addr), &AddrSize);
|
||||
|
||||
// Add the new client to the temporary array
|
||||
NewPollFDs.emplace_back(pollfd {
|
||||
@@ -494,7 +514,7 @@ void WaitForRequests() {
|
||||
} else {
|
||||
auto Now = std::chrono::system_clock::now();
|
||||
auto Diff = Now - LastDataTime;
|
||||
if (Diff >= std::chrono::seconds(RequestTimeout) && !Foreground && PollFDs.size() == 1) {
|
||||
if (Diff >= std::chrono::seconds(RequestTimeout) && !Foreground && PollFDs.size() == 2) {
|
||||
// If we aren't running in the foreground and we have no connections after a timeout
|
||||
// Then we can just go ahead and leave
|
||||
ShouldShutdown = true;
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
|
||||
namespace ProcessPipe {
|
||||
bool InitializeServerPipe();
|
||||
bool InitializeServerSocket();
|
||||
bool InitializeServerSocket(bool abstract);
|
||||
void WaitForRequests();
|
||||
void SetConfiguration(bool Foreground, uint32_t PersistentTimeout);
|
||||
void Shutdown();
|
||||
|
||||
@@ -3,6 +3,7 @@ add_compile_options(-fno-operator-names)
|
||||
set (SRCS
|
||||
VDSO_Emulation.cpp
|
||||
Thunks.cpp
|
||||
GdbServer/Info.cpp
|
||||
LinuxSyscalls/GdbServer.cpp
|
||||
LinuxSyscalls/EmulatedFiles/EmulatedFiles.cpp
|
||||
LinuxSyscalls/FaultSafeUserMemAccess.cpp
|
||||
|
||||
@@ -0,0 +1,205 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: glue|gdbserver
|
||||
desc: Provides a gdb interface to the guest state
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "GdbServer/Info.h"
|
||||
|
||||
#include <Common/StringUtil.h>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
#include <string_view>
|
||||
|
||||
namespace FEX::GDB::Info {
|
||||
constexpr std::array<std::string_view, 22> FlagNames = {
|
||||
"CF", "", "PF", "", "AF", "", "ZF", "SF", "TF", "IF", "DF", "OF", "IOPL", "", "NT", "", "RF", "VM", "AC", "VIF", "VIP", "ID",
|
||||
};
|
||||
|
||||
const std::string_view& GetFlagName(unsigned Bit) {
|
||||
LOGMAN_THROW_A_FMT(Bit < 22, "Bit position too large");
|
||||
return FlagNames[Bit];
|
||||
}
|
||||
|
||||
std::string_view GetGRegName(unsigned Reg) {
|
||||
switch (Reg) {
|
||||
case FEXCore::X86State::REG_RAX: return "rax";
|
||||
case FEXCore::X86State::REG_RBX: return "rbx";
|
||||
case FEXCore::X86State::REG_RCX: return "rcx";
|
||||
case FEXCore::X86State::REG_RDX: return "rdx";
|
||||
case FEXCore::X86State::REG_RSP: return "rsp";
|
||||
case FEXCore::X86State::REG_RBP: return "rbp";
|
||||
case FEXCore::X86State::REG_RSI: return "rsi";
|
||||
case FEXCore::X86State::REG_RDI: return "rdi";
|
||||
case FEXCore::X86State::REG_R8: return "r8";
|
||||
case FEXCore::X86State::REG_R9: return "r9";
|
||||
case FEXCore::X86State::REG_R10: return "r10";
|
||||
case FEXCore::X86State::REG_R11: return "r11";
|
||||
case FEXCore::X86State::REG_R12: return "r12";
|
||||
case FEXCore::X86State::REG_R13: return "r13";
|
||||
case FEXCore::X86State::REG_R14: return "r14";
|
||||
case FEXCore::X86State::REG_R15: return "r15";
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
fextl::string GetThreadName(uint32_t PID, uint32_t ThreadID) {
|
||||
const auto ThreadFile = fextl::fmt::format("/proc/{}/task/{}/comm", PID, ThreadID);
|
||||
fextl::string ThreadName;
|
||||
FEXCore::FileLoading::LoadFile(ThreadName, ThreadFile);
|
||||
// Trim out the potential newline, breaks GDB if it exists.
|
||||
FEX::StringUtil::trim(ThreadName);
|
||||
return ThreadName;
|
||||
}
|
||||
|
||||
fextl::string BuildOSXML() {
|
||||
fextl::ostringstream xml;
|
||||
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
|
||||
xml << "<!DOCTYPE target SYSTEM \"osdata.dtd\">\n";
|
||||
xml << "<osdata type=\"processes\">";
|
||||
// XXX
|
||||
xml << "</osdata>";
|
||||
|
||||
xml << std::flush;
|
||||
|
||||
return xml.str();
|
||||
}
|
||||
|
||||
fextl::string BuildTargetXML() {
|
||||
fextl::ostringstream xml;
|
||||
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
xml << "<!DOCTYPE target SYSTEM 'gdb-target.dtd'>\n";
|
||||
xml << "<target>\n";
|
||||
xml << "<architecture>i386:x86-64</architecture>\n";
|
||||
xml << "<osabi>GNU/Linux</osabi>\n";
|
||||
xml << "<feature name='org.gnu.gdb.i386.core'>\n";
|
||||
|
||||
xml << "<flags id='fex_eflags' size='4'>\n";
|
||||
// flags register
|
||||
for (int i = 0; i < 22; i++) {
|
||||
auto name = GDB::Info::GetFlagName(i);
|
||||
if (name.empty()) {
|
||||
continue;
|
||||
}
|
||||
xml << "\t<field name='" << name << "' start='" << i << "' end='" << i << "' />\n";
|
||||
}
|
||||
xml << "</flags>\n";
|
||||
|
||||
int32_t TargetSize {};
|
||||
auto reg = [&](std::string_view name, std::string_view type, int size) {
|
||||
TargetSize += size;
|
||||
xml << "<reg name='" << name << "' bitsize='" << size << "' type='" << type << "' />" << std::endl;
|
||||
};
|
||||
|
||||
// Register ordering.
|
||||
// We want to just memcpy our x86 state to gdb, so we tell it the ordering.
|
||||
|
||||
// GPRs
|
||||
for (uint32_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; i++) {
|
||||
reg(GDB::Info::GetGRegName(i), "int64", 64);
|
||||
}
|
||||
|
||||
reg("rip", "code_ptr", 64);
|
||||
|
||||
reg("eflags", "fex_eflags", 32);
|
||||
|
||||
// Fake registers which GDB requires, but we don't support;
|
||||
// We stick them past the end of our cpu state.
|
||||
|
||||
// non-userspace segment registers
|
||||
reg("cs", "int32", 32);
|
||||
reg("ss", "int32", 32);
|
||||
reg("ds", "int32", 32);
|
||||
reg("es", "int32", 32);
|
||||
|
||||
reg("fs", "int32", 32);
|
||||
reg("gs", "int32", 32);
|
||||
|
||||
// x87 stack
|
||||
for (int i = 0; i < 8; i++) {
|
||||
reg(fextl::fmt::format("st{}", i), "i387_ext", 80);
|
||||
}
|
||||
|
||||
// x87 control
|
||||
reg("fctrl", "int32", 32);
|
||||
reg("fstat", "int32", 32);
|
||||
reg("ftag", "int32", 32);
|
||||
reg("fiseg", "int32", 32);
|
||||
reg("fioff", "int32", 32);
|
||||
reg("foseg", "int32", 32);
|
||||
reg("fooff", "int32", 32);
|
||||
reg("fop", "int32", 32);
|
||||
|
||||
|
||||
xml << "</feature>\n";
|
||||
xml << "<feature name='org.gnu.gdb.i386.sse'>\n";
|
||||
xml <<
|
||||
R"(<vector id="v4f" type="ieee_single" count="4"/>
|
||||
<vector id="v2d" type="ieee_double" count="2"/>
|
||||
<vector id="v16i8" type="int8" count="16"/>
|
||||
<vector id="v8i16" type="int16" count="8"/>
|
||||
<vector id="v4i32" type="int32" count="4"/>
|
||||
<vector id="v2i64" type="int64" count="2"/>
|
||||
<union id="vec128">
|
||||
<field name="v4_float" type="v4f"/>
|
||||
<field name="v2_double" type="v2d"/>
|
||||
<field name="v16_int8" type="v16i8"/>
|
||||
<field name="v8_int16" type="v8i16"/>
|
||||
<field name="v4_int32" type="v4i32"/>
|
||||
<field name="v2_int64" type="v2i64"/>
|
||||
<field name="uint128" type="uint128"/>
|
||||
</union>
|
||||
)";
|
||||
|
||||
// SSE regs
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
reg(fextl::fmt::format("xmm{}", i), "vec128", 128);
|
||||
}
|
||||
|
||||
reg("mxcsr", "int", 32);
|
||||
|
||||
xml << "</feature>\n";
|
||||
|
||||
xml << "<feature name='org.gnu.gdb.i386.avx'>";
|
||||
xml <<
|
||||
R"(<vector id="v4f" type="ieee_single" count="4"/>
|
||||
<vector id="v2d" type="ieee_double" count="2"/>
|
||||
<vector id="v16i8" type="int8" count="16"/>
|
||||
<vector id="v8i16" type="int16" count="8"/>
|
||||
<vector id="v4i32" type="int32" count="4"/>
|
||||
<vector id="v2i64" type="int64" count="2"/>
|
||||
<union id="vec128">
|
||||
<field name="v4_float" type="v4f"/>
|
||||
<field name="v2_double" type="v2d"/>
|
||||
<field name="v16_int8" type="v16i8"/>
|
||||
<field name="v8_int16" type="v8i16"/>
|
||||
<field name="v4_int32" type="v4i32"/>
|
||||
<field name="v2_int64" type="v2i64"/>
|
||||
<field name="uint128" type="uint128"/>
|
||||
</union>
|
||||
)";
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; i++) {
|
||||
reg(fmt::format("ymm{}h", i), "vec128", 128);
|
||||
}
|
||||
xml << "</feature>\n";
|
||||
|
||||
xml << "</target>";
|
||||
xml << std::flush;
|
||||
|
||||
return xml.str();
|
||||
}
|
||||
|
||||
} // namespace FEX::GDB::Info
|
||||
@@ -0,0 +1,51 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: glue|gdbserver
|
||||
desc: Provides a gdb interface to the guest state
|
||||
$end_info$
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <string_view>
|
||||
|
||||
namespace FEXCore::X86State {
|
||||
enum X86Reg : uint32_t;
|
||||
}
|
||||
|
||||
namespace FEX::GDB::Info {
|
||||
/**
|
||||
* @brief Returns textual name of bit location from EFLAGs register.
|
||||
*
|
||||
* @param Bit Which bit of EFLAG to query
|
||||
*/
|
||||
const std::string_view& GetFlagName(unsigned Bit);
|
||||
|
||||
/**
|
||||
* @brief Returns the textual name of a GPR register
|
||||
*
|
||||
* @param Reg Index of the register to fetch
|
||||
*/
|
||||
std::string_view GetGRegName(unsigned Reg);
|
||||
|
||||
/**
|
||||
* @brief Fetches the thread's name
|
||||
*
|
||||
* @param PID The program id of the application
|
||||
* @param ThreadID The thread id of the program
|
||||
*/
|
||||
fextl::string GetThreadName(uint32_t PID, uint32_t ThreadID);
|
||||
|
||||
/**
|
||||
* @brief Returns the GDB specific construct of OS describing XML.
|
||||
*/
|
||||
fextl::string BuildOSXML();
|
||||
|
||||
/**
|
||||
* @brief Returns the GDB specific construct of target describing XML.
|
||||
*/
|
||||
fextl::string BuildTargetXML();
|
||||
} // namespace FEX::GDB::Info
|
||||
@@ -322,6 +322,16 @@ FileManager::FileManager(FEXCore::Context::Context* ctx)
|
||||
}
|
||||
}
|
||||
|
||||
// Keep an fd open for /proc, to bypass chroot-style sandboxes
|
||||
ProcFD = open("/proc", O_RDONLY | O_CLOEXEC);
|
||||
|
||||
// Track the st_dev of /proc, to check for inode equality
|
||||
struct stat Buffer;
|
||||
auto Result = fstat(ProcFD, &Buffer);
|
||||
if (Result >= 0) {
|
||||
ProcFSDev = Buffer.st_dev;
|
||||
}
|
||||
|
||||
UpdatePID(::getpid());
|
||||
}
|
||||
|
||||
@@ -994,4 +1004,47 @@ uint64_t FileManager::LRemovexattr(const char* path, const char* name) {
|
||||
return ::lremovexattr(SelfPath, name);
|
||||
}
|
||||
|
||||
void FileManager::UpdatePID(uint32_t PID) {
|
||||
CurrentPID = PID;
|
||||
|
||||
// Track the inode of /proc/self/fd/<RootFSFD>, to be able to hide it
|
||||
auto FDpath = fextl::fmt::format("self/fd/{}", RootFSFD);
|
||||
struct stat Buffer {};
|
||||
int Result = fstatat(ProcFD, FDpath.c_str(), &Buffer, AT_SYMLINK_NOFOLLOW);
|
||||
if (Result >= 0) {
|
||||
RootFSFDInode = Buffer.st_ino;
|
||||
} else {
|
||||
// Probably in a strict sandbox
|
||||
RootFSFDInode = 0;
|
||||
ProcFDInode = 0;
|
||||
return;
|
||||
}
|
||||
|
||||
// And track the ProcFSFD itself
|
||||
FDpath = fextl::fmt::format("self/fd/{}", ProcFD);
|
||||
Result = fstatat(ProcFD, FDpath.c_str(), &Buffer, AT_SYMLINK_NOFOLLOW);
|
||||
if (Result >= 0) {
|
||||
ProcFDInode = Buffer.st_ino;
|
||||
} else {
|
||||
// ??
|
||||
ProcFDInode = 0;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
bool FileManager::IsRootFSFD(int dirfd, uint64_t inode) {
|
||||
|
||||
// Check if we have to hide this entry
|
||||
if (inode == RootFSFDInode || inode == ProcFDInode) {
|
||||
struct stat Buffer;
|
||||
if (fstat(dirfd, &Buffer) >= 0) {
|
||||
if (Buffer.st_dev == ProcFSDev) {
|
||||
LogMan::Msg::DFmt("Hiding directory entry for RootFSFD");
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
} // namespace FEX::HLE
|
||||
@@ -81,9 +81,8 @@ public:
|
||||
std::optional<std::string_view> GetSelf(const char* Pathname);
|
||||
bool IsSelfNoFollow(const char* Pathname, int flags) const;
|
||||
|
||||
void UpdatePID(uint32_t PID) {
|
||||
CurrentPID = PID;
|
||||
}
|
||||
void UpdatePID(uint32_t PID);
|
||||
bool IsRootFSFD(int dirfd, uint64_t inode);
|
||||
|
||||
fextl::string GetEmulatedPath(const char* pathname, bool FollowSymlink = false);
|
||||
using FDPathTmpData = std::array<char[PATH_MAX], 2>;
|
||||
@@ -162,5 +161,9 @@ private:
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
uint32_t CurrentPID {};
|
||||
int RootFSFD {AT_FDCWD};
|
||||
int ProcFD {0};
|
||||
int64_t RootFSFDInode = 0;
|
||||
int64_t ProcFDInode = 0;
|
||||
dev_t ProcFSDev;
|
||||
};
|
||||
} // namespace FEX::HLE
|
||||
File diff suppressed because it is too large.
Load diff
@@ -36,7 +36,7 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
void Break(int signal);
|
||||
void Break(FEXCore::Core::InternalThreadState* Thread, int signal);
|
||||
|
||||
void OpenListenSocket();
|
||||
void CloseListenSocket();
|
||||
@@ -70,18 +70,78 @@ private:
|
||||
|
||||
void SendPacketPair(const HandledPacketType& packetPair);
|
||||
HandledPacketType ProcessPacket(const fextl::string& packet);
|
||||
HandledPacketType handleQuery(const fextl::string& packet);
|
||||
HandledPacketType handleXfer(const fextl::string& packet);
|
||||
HandledPacketType handleMemory(const fextl::string& packet);
|
||||
HandledPacketType handleV(const fextl::string& packet);
|
||||
HandledPacketType handleThreadOp(const fextl::string& packet);
|
||||
HandledPacketType handleBreakpoint(const fextl::string& packet);
|
||||
HandledPacketType handleProgramOffsets();
|
||||
|
||||
HandledPacketType ThreadAction(char action, uint32_t tid);
|
||||
|
||||
fextl::string readRegs();
|
||||
HandledPacketType readReg(const fextl::string& packet);
|
||||
// Binary data transfer handlers
|
||||
// XFer function to correctly encode any reply
|
||||
static fextl::string EncodeXferString(const fextl::string& data, int offset, int length) {
|
||||
if (offset == data.size()) {
|
||||
return "l";
|
||||
}
|
||||
if (offset >= data.size()) {
|
||||
return "E34"; // ERANGE
|
||||
}
|
||||
if ((data.size() - offset) > length) {
|
||||
return "m" + data.substr(offset, length);
|
||||
}
|
||||
return "l" + data.substr(offset);
|
||||
};
|
||||
|
||||
HandledPacketType XferCommandExecFile(const fextl::string& annex, int offset, int length);
|
||||
HandledPacketType XferCommandFeatures(const fextl::string& annex, int offset, int length);
|
||||
HandledPacketType XferCommandThreads(const fextl::string& annex, int offset, int length);
|
||||
HandledPacketType XferCommandOSData(const fextl::string& annex, int offset, int length);
|
||||
HandledPacketType XferCommandLibraries(const fextl::string& annex, int offset, int length);
|
||||
HandledPacketType XferCommandAuxv(const fextl::string& annex, int offset, int length);
|
||||
HandledPacketType handleXfer(const fextl::string& packet);
|
||||
|
||||
HandledPacketType HandlevFile(const fextl::string& packet);
|
||||
HandledPacketType HandlevCont(const fextl::string& packet);
|
||||
|
||||
// Command handlers
|
||||
HandledPacketType CommandEnableExtendedMode(const fextl::string& packet);
|
||||
HandledPacketType CommandQueryHalted(const fextl::string& packet);
|
||||
HandledPacketType CommandContinue(const fextl::string& packet);
|
||||
HandledPacketType CommandDetach(const fextl::string& packet);
|
||||
HandledPacketType CommandReadRegisters(const fextl::string& packet);
|
||||
HandledPacketType CommandThreadOp(const fextl::string& packet);
|
||||
HandledPacketType CommandKill(const fextl::string& packet);
|
||||
HandledPacketType CommandMemory(const fextl::string& packet);
|
||||
HandledPacketType CommandReadReg(const fextl::string& packet);
|
||||
HandledPacketType CommandQuery(const fextl::string& packet);
|
||||
HandledPacketType CommandSingleStep(const fextl::string& packet);
|
||||
HandledPacketType CommandQueryThreadAlive(const fextl::string& packet);
|
||||
HandledPacketType CommandMultiLetterV(const fextl::string& packet);
|
||||
HandledPacketType CommandBreakpoint(const fextl::string& packet);
|
||||
HandledPacketType CommandUnknown(const fextl::string& packet);
|
||||
|
||||
/**
|
||||
* @brief Returns the ThreadStateObject for the matching TID, or parent thread if TID isn't found
|
||||
*
|
||||
* @param TID Which TID to search for
|
||||
*/
|
||||
const FEX::HLE::ThreadStateObject* FindThreadByTID(uint32_t TID);
|
||||
|
||||
struct X80Float {
|
||||
uint8_t Data[10];
|
||||
};
|
||||
|
||||
struct FEX_PACKED GDBContextDefinition {
|
||||
uint64_t gregs[FEXCore::Core::CPUState::NUM_GPRS];
|
||||
uint64_t rip;
|
||||
uint32_t eflags;
|
||||
uint32_t cs, ss, ds, es, fs, gs;
|
||||
X80Float mm[FEXCore::Core::CPUState::NUM_MMS];
|
||||
uint32_t fctrl;
|
||||
uint32_t fstat;
|
||||
uint32_t dummies[6];
|
||||
uint64_t xmm[FEXCore::Core::CPUState::NUM_XMMS][4];
|
||||
uint32_t mxcsr;
|
||||
};
|
||||
|
||||
GDBContextDefinition GenerateContextDefinition(const FEX::HLE::ThreadStateObject* ThreadObject);
|
||||
|
||||
FEXCore::Context::Context* CTX;
|
||||
FEX::HLE::SyscallHandler* const SyscallHandler;
|
||||
|
||||
@@ -392,6 +392,7 @@ bool SignalDelegator::HandleDispatcherGuestSignal(FEXCore::Core::InternalThreadS
|
||||
bool SignalDelegator::HandleSIGILL(FEXCore::Core::InternalThreadState* Thread, int Signal, void* info, void* ucontext) {
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == Config.SignalHandlerReturnAddress ||
|
||||
ArchHelpers::Context::GetPc(ucontext) == Config.SignalHandlerReturnAddressRT) {
|
||||
auto ThreadObject = FEX::HLE::ThreadManager::GetStateObjectFromFEXCoreThread(Thread);
|
||||
RestoreThreadState(Thread, ucontext,
|
||||
ArchHelpers::Context::GetPc(ucontext) == Config.SignalHandlerReturnAddressRT ? RestoreType::TYPE_REALTIME :
|
||||
RestoreType::TYPE_NONREALTIME);
|
||||
@@ -400,7 +401,7 @@ bool SignalDelegator::HandleSIGILL(FEXCore::Core::InternalThreadState* Thread, i
|
||||
// We use this to track if it is safe to clear cache
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
if (Thread->DeferredSignalFrames.size() != 0) {
|
||||
if (ThreadObject->SignalInfo.DeferredSignalFrames.size() != 0) {
|
||||
// If we have more deferred frames to process then mprotect back to PROT_NONE.
|
||||
// It will have been RW coming in to this sigreturn and now we need to remove permissions
|
||||
// to ensure FEX trampolines back to the SIGSEGV deferred handler.
|
||||
@@ -422,10 +423,11 @@ bool SignalDelegator::HandleSIGILL(FEXCore::Core::InternalThreadState* Thread, i
|
||||
}
|
||||
|
||||
bool SignalDelegator::HandleSignalPause(FEXCore::Core::InternalThreadState* Thread, int Signal, void* info, void* ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = Thread->SignalReason.load();
|
||||
auto ThreadObject = FEX::HLE::ThreadManager::GetStateObjectFromFEXCoreThread(Thread);
|
||||
SignalEvent SignalReason = ThreadObject->SignalReason.load();
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
|
||||
if (SignalReason == SignalEvent::Pause) {
|
||||
// Store our thread state so we can come back to this
|
||||
StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
@@ -446,11 +448,11 @@ bool SignalDelegator::HandleSignalPause(FEXCore::Core::InternalThreadState* Thre
|
||||
// We use this to track if it is safe to clear cache
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
ThreadObject->SignalReason.store(SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Stop) {
|
||||
if (SignalReason == SignalEvent::Stop) {
|
||||
// Our thread is stopping
|
||||
// We don't care about anything at this point
|
||||
// Set the stack to our starting location when we entered the core and get out safely
|
||||
@@ -473,37 +475,32 @@ bool SignalDelegator::HandleSignalPause(FEXCore::Core::InternalThreadState* Thre
|
||||
// We need to be a little bit careful here
|
||||
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
|
||||
// Then we need to ensure we don't double decrement our idle thread counter
|
||||
if (Thread->RunningEvents.ThreadSleeping) {
|
||||
if (ThreadObject->ThreadSleeping) {
|
||||
// If the thread was sleeping then its idle counter was decremented
|
||||
// Reincrement it here to not break logic
|
||||
FEX::HLE::_SyscallHandler->TM.IncrementIdleRefCount();
|
||||
}
|
||||
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
ThreadObject->SignalReason.store(SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Return || SignalReason == FEXCore::Core::SignalEvent::ReturnRT) {
|
||||
RestoreThreadState(Thread, ucontext,
|
||||
SignalReason == FEXCore::Core::SignalEvent::ReturnRT ? RestoreType::TYPE_REALTIME : RestoreType::TYPE_NONREALTIME);
|
||||
if (SignalReason == SignalEvent::Return || SignalReason == SignalEvent::ReturnRT) {
|
||||
RestoreThreadState(Thread, ucontext, SignalReason == SignalEvent::ReturnRT ? RestoreType::TYPE_REALTIME : RestoreType::TYPE_NONREALTIME);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
ThreadObject->SignalReason.store(SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void SignalDelegator::SignalThread(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::SignalEvent Event) {
|
||||
auto ThreadObject = static_cast<const FEX::HLE::ThreadStateObject*>(Thread->FrontendPtr);
|
||||
if (Event == FEXCore::Core::SignalEvent::Pause && Thread->RunningEvents.Running.load() == false) {
|
||||
// Skip signaling a thread if it is already paused.
|
||||
return;
|
||||
}
|
||||
Thread->SignalReason.store(Event);
|
||||
void SignalDelegator::SignalThread(FEXCore::Core::InternalThreadState* Thread, SignalEvent Event) {
|
||||
auto ThreadObject = FEX::HLE::ThreadManager::GetStateObjectFromFEXCoreThread(Thread);
|
||||
ThreadObject->SignalReason.store(Event);
|
||||
FHU::Syscalls::tgkill(ThreadObject->ThreadInfo.PID, ThreadObject->ThreadInfo.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
|
||||
@@ -548,16 +545,16 @@ void SignalDelegator::HandleGuestSignal(FEX::HLE::ThreadStateObject* ThreadObjec
|
||||
|
||||
mprotect(reinterpret_cast<void*>(&Thread->InterruptFaultPage), sizeof(Thread->InterruptFaultPage), PROT_READ | PROT_WRITE);
|
||||
|
||||
if (Thread->DeferredSignalFrames.empty()) {
|
||||
if (ThreadObject->SignalInfo.DeferredSignalFrames.empty()) {
|
||||
// No signals to defer. Just set the fault page back to RW and continue execution.
|
||||
// This occurs as a minor race condition between the refcount decrement and the access to the fault page.
|
||||
return;
|
||||
}
|
||||
|
||||
const auto& Top = Thread->DeferredSignalFrames.back();
|
||||
const auto& Top = ThreadObject->SignalInfo.DeferredSignalFrames.back();
|
||||
Signal = Top.Signal;
|
||||
SigInfo = Top.Info;
|
||||
Thread->DeferredSignalFrames.pop_back();
|
||||
ThreadObject->SignalInfo.DeferredSignalFrames.pop_back();
|
||||
|
||||
// Until we re-protect the page to PROT_NONE, FEX will now *permanently* defer signals and /not/ check them.
|
||||
//
|
||||
@@ -596,10 +593,11 @@ void SignalDelegator::HandleGuestSignal(FEX::HLE::ThreadStateObject* ThreadObjec
|
||||
if (IsAsyncSignal(&SigInfo, Signal) && MustDeferSignal) {
|
||||
// If the signal is asynchronous (as determined by si_code) and FEX is in a state of needing
|
||||
// to defer the signal, then add the signal to the thread's signal queue.
|
||||
LOGMAN_THROW_A_FMT(Thread->DeferredSignalFrames.size() != Thread->DeferredSignalFrames.capacity(), "Deferred signals vector hit "
|
||||
"capacity size. This will "
|
||||
"likely crash! Asserting now!");
|
||||
Thread->DeferredSignalFrames.emplace_back(FEXCore::Core::InternalThreadState::DeferredSignalState {
|
||||
LOGMAN_THROW_A_FMT(ThreadObject->SignalInfo.DeferredSignalFrames.size() != ThreadObject->SignalInfo.DeferredSignalFrames.capacity(),
|
||||
"Deferred signals vector hit "
|
||||
"capacity size. This will "
|
||||
"likely crash! Asserting now!");
|
||||
ThreadObject->SignalInfo.DeferredSignalFrames.emplace_back(ThreadStateObject::DeferredSignalState {
|
||||
.Info = SigInfo,
|
||||
.Signal = Signal,
|
||||
});
|
||||
@@ -956,6 +954,8 @@ SignalDelegator::~SignalDelegator() {
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterTLSState(FEX::HLE::ThreadStateObject* Thread) {
|
||||
FEXCore::Allocator::RegisterTLSData(Thread->Thread);
|
||||
|
||||
Thread->SignalInfo.Delegator = this;
|
||||
|
||||
// Set up our signal alternative stack
|
||||
@@ -982,7 +982,7 @@ void SignalDelegator::RegisterTLSState(FEX::HLE::ThreadStateObject* Thread) {
|
||||
if (Thread->Thread) {
|
||||
// Reserve a small amount of deferred signal frames. Usually the stack won't be utilized beyond
|
||||
// 1 or 2 signals but add a few more just in case.
|
||||
Thread->Thread->DeferredSignalFrames.reserve(8);
|
||||
Thread->SignalInfo.DeferredSignalFrames.reserve(8);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -999,6 +999,8 @@ void SignalDelegator::UninstallTLSState(FEX::HLE::ThreadStateObject* Thread) {
|
||||
if (Result == -1) {
|
||||
LogMan::Msg::EFmt("Failed to uninstall alternative signal stack {}", strerror(errno));
|
||||
}
|
||||
|
||||
FEXCore::Allocator::UninstallTLSData(Thread->Thread);
|
||||
}
|
||||
|
||||
void SignalDelegator::FrontendRegisterHostSignalHandler(int Signal, bool Required) {
|
||||
|
||||
@@ -124,7 +124,13 @@ public:
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
void SignalThread(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::SignalEvent Event) override;
|
||||
/**
|
||||
* @brief Signals a thread with a specific core event.
|
||||
*
|
||||
* @param Thread Which thread to signal.
|
||||
* @param Event Which event to signal the event with.
|
||||
*/
|
||||
void SignalThread(FEXCore::Core::InternalThreadState* Thread, SignalEvent Event);
|
||||
|
||||
FEXCore::ArchHelpers::Arm64::UnalignedHandlerType GetUnalignedHandlerType() const {
|
||||
return UnalignedHandlerType;
|
||||
|
||||
@@ -120,6 +120,11 @@ uint64_t GetDentsEmulation(int fd, T* dirp, uint32_t count) {
|
||||
Outgoing->d_name[Outgoing->d_reclen - offsetof(T, d_name) - 1] = Tmp->d_type;
|
||||
|
||||
TmpOffset += Tmp->d_reclen;
|
||||
|
||||
if (FEX::HLE::_SyscallHandler->FM.IsRootFSFD(fd, Outgoing->d_ino)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Outgoing is 5 bytes smaller
|
||||
Offset += NewRecLen;
|
||||
|
||||
@@ -578,7 +583,7 @@ uint64_t CloneHandler(FEXCore::Core::CpuStateFrame* Frame, FEX::HLE::clone3_args
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
if (AnyFlagsSet(args->args.flags, CLONE_SYSVSEM | CLONE_FS | CLONE_FILES | CLONE_SIGHAND | CLONE_VM)) {
|
||||
if (AnyFlagsSet(args->args.flags, CLONE_SYSVSEM | CLONE_SIGHAND | CLONE_VM)) {
|
||||
// CLONE_VM is particularly nasty here
|
||||
// Memory regions at the point of clone(More similar to a fork) are shared
|
||||
LogMan::Msg::IFmt("clone: Unsupported flags w/o CLONE_THREAD (Shared Resources), {:X}", args->args.flags);
|
||||
@@ -654,12 +659,9 @@ uint64_t CloneHandler(FEXCore::Core::CpuStateFrame* Frame, FEX::HLE::clone3_args
|
||||
// Return the new threads TID
|
||||
uint64_t Result = NewThread->ThreadInfo.TID;
|
||||
|
||||
// Actually start the thread
|
||||
FEX::HLE::_SyscallHandler->TM.RunThread(NewThread);
|
||||
|
||||
if (flags & CLONE_VFORK) {
|
||||
// If VFORK is set then the calling process is suspended until the thread exits with execve or exit
|
||||
NewThread->Thread->ExecutionThread->join(nullptr);
|
||||
NewThread->ExecutionThread->join(nullptr);
|
||||
|
||||
// Normally a thread cleans itself up on exit. But because we need to join, we are now responsible
|
||||
FEX::HLE::_SyscallHandler->TM.DestroyThread(NewThread);
|
||||
|
||||
@@ -24,6 +24,7 @@ $end_info$
|
||||
#include <sys/syscall.h>
|
||||
#include <sys/utsname.h>
|
||||
#include <sys/klog.h>
|
||||
#include <sys/personality.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include <git_version.h>
|
||||
@@ -37,6 +38,8 @@ void RegisterInfo(FEX::HLE::SyscallHandler* Handler) {
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(
|
||||
uname, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY, [](FEXCore::Core::CpuStateFrame* Frame, struct utsname* buf) -> uint64_t {
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
struct utsname Local {};
|
||||
if (::uname(&Local) == 0) {
|
||||
memcpy(buf->nodename, Local.nodename, sizeof(Local.nodename));
|
||||
@@ -49,17 +52,48 @@ void RegisterInfo(FEX::HLE::SyscallHandler* Handler) {
|
||||
}
|
||||
strcpy(buf->sysname, "Linux");
|
||||
uint32_t GuestVersion = FEX::HLE::_SyscallHandler->GetGuestKernelVersion();
|
||||
if (Thread->persona & UNAME26) {
|
||||
// Kernel version converts from 6.x.y to 2.6.60+x.
|
||||
GuestVersion = FEX::HLE::SyscallHandler::KernelVersion(2, 6, 60 + FEX::HLE::SyscallHandler::KernelMinor(GuestVersion));
|
||||
}
|
||||
snprintf(buf->release, sizeof(buf->release), "%d.%d.%d", FEX::HLE::SyscallHandler::KernelMajor(GuestVersion),
|
||||
FEX::HLE::SyscallHandler::KernelMinor(GuestVersion), FEX::HLE::SyscallHandler::KernelPatch(GuestVersion));
|
||||
|
||||
const char version[] = "#" GIT_DESCRIBE_STRING " SMP " __DATE__ " " __TIME__;
|
||||
strcpy(buf->version, version);
|
||||
static_assert(sizeof(version) <= sizeof(buf->version), "uname version define became too large!");
|
||||
// Tell the guest that we are a 64bit kernel
|
||||
strcpy(buf->machine, "x86_64");
|
||||
if (Thread->persona & PER_LINUX32) {
|
||||
// Tell the guest that we are a 32bit kernel
|
||||
strcpy(buf->machine, "i686");
|
||||
} else {
|
||||
// Tell the guest that we are a 64bit kernel
|
||||
strcpy(buf->machine, "x86_64");
|
||||
}
|
||||
return 0;
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(personality, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, uint32_t persona) -> uint64_t {
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
if (persona == ~0U) {
|
||||
// Special case, only queries the persona.
|
||||
return Thread->persona;
|
||||
}
|
||||
|
||||
// Mask off `PER_LINUX32` because AArch64 doesn't support it.
|
||||
uint32_t NewPersona = persona & ~PER_LINUX32;
|
||||
|
||||
// This syscall can not physically fail with PER_LINUX32 masked off.
|
||||
// It also can not fail on a real x86 kernel.
|
||||
(void)::syscall(SYSCALL_DEF(personality), NewPersona);
|
||||
|
||||
// Return the old persona while setting the new one.
|
||||
auto OldPersona = Thread->persona;
|
||||
Thread->persona = persona;
|
||||
return OldPersona;
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(seccomp, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, unsigned int operation, unsigned int flags, void* args) -> uint64_t {
|
||||
return FEX::HLE::_SyscallHandler->SeccompEmulator.Handle(Frame, operation, flags, args);
|
||||
|
||||
@@ -330,8 +330,6 @@ void RegisterCommon(FEX::HLE::SyscallHandler* Handler) {
|
||||
SyscallPassthrough2<SYSCALL_DEF(capget)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(capset, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(capset)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(personality, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough1<SYSCALL_DEF(personality)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(getpriority, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough2<SYSCALL_DEF(getpriority)>);
|
||||
REGISTER_SYSCALL_IMPL_PASS_FLAGS(setpriority, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
@@ -696,8 +694,6 @@ namespace x64 {
|
||||
SyscallPassthrough6<SYSCALL_DEF(futex)>);
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(io_getevents, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough5<SYSCALL_DEF(io_getevents)>);
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(getdents64, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough3<SYSCALL_DEF(getdents64)>);
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(semtimedop, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
SyscallPassthrough4<SYSCALL_DEF(semtimedop)>);
|
||||
REGISTER_SYSCALL_IMPL_X64_PASS_FLAGS(timer_create, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY,
|
||||
|
||||
@@ -26,6 +26,7 @@ $end_info$
|
||||
#include <limits.h>
|
||||
#include <linux/futex.h>
|
||||
#include <linux/seccomp.h>
|
||||
#include <linux/sched.h>
|
||||
#include <stdint.h>
|
||||
#include <sched.h>
|
||||
#include <sys/personality.h>
|
||||
@@ -46,19 +47,33 @@ namespace FEX::HLE {
|
||||
struct ExecutionThreadHandler {
|
||||
FEXCore::Context::Context* CTX;
|
||||
FEX::HLE::ThreadStateObject* Thread;
|
||||
Event ThreadWaiting {};
|
||||
|
||||
// Pause on thread start handling.
|
||||
FEXCore::InterruptableConditionVariable StartRunningCV {};
|
||||
FEXCore::InterruptableConditionVariable StartRunningResponse {};
|
||||
};
|
||||
|
||||
static void* ThreadHandler(void* Data) {
|
||||
ExecutionThreadHandler* Handler = reinterpret_cast<ExecutionThreadHandler*>(Data);
|
||||
auto CTX = Handler->CTX;
|
||||
auto Thread = Handler->Thread;
|
||||
FEXCore::Allocator::free(Handler);
|
||||
|
||||
Thread->ThreadInfo.PID = ::getpid();
|
||||
Thread->ThreadInfo.TID = FHU::Syscalls::gettid();
|
||||
|
||||
FEX::HLE::_SyscallHandler->RegisterTLSState(Thread);
|
||||
CTX->ExecutionThread(Thread->Thread);
|
||||
|
||||
// Now notify the thread that we are initialized
|
||||
Handler->ThreadWaiting.NotifyOne();
|
||||
|
||||
Handler->StartRunningCV.Wait();
|
||||
|
||||
// Notify the parent thread that it can continue.
|
||||
// Handler is a stack object on the parent thread, and will be invalid after notification.
|
||||
Handler->StartRunningResponse.NotifyOne();
|
||||
|
||||
CTX->ExecuteThread(Thread->Thread);
|
||||
FEX::HLE::_SyscallHandler->UninstallTLSState(Thread);
|
||||
FEX::HLE::_SyscallHandler->TM.DestroyThread(Thread);
|
||||
return nullptr;
|
||||
@@ -91,17 +106,15 @@ FEX::HLE::ThreadStateObject* CreateNewThread(FEXCore::Context::Context* CTX, FEX
|
||||
x32::AdjustRipForNewThread(NewThread->Thread->CurrentFrame);
|
||||
}
|
||||
|
||||
// We need to do some post-thread creation setup.
|
||||
NewThread->Thread->StartPaused = true;
|
||||
|
||||
// Initialize a new thread for execution.
|
||||
ExecutionThreadHandler* Arg = reinterpret_cast<ExecutionThreadHandler*>(FEXCore::Allocator::malloc(sizeof(ExecutionThreadHandler)));
|
||||
Arg->CTX = CTX;
|
||||
Arg->Thread = NewThread;
|
||||
NewThread->Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
ExecutionThreadHandler Arg {
|
||||
.CTX = CTX,
|
||||
.Thread = NewThread,
|
||||
};
|
||||
NewThread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, &Arg);
|
||||
|
||||
// Wait for the thread to have started.
|
||||
NewThread->Thread->ThreadWaiting.Wait();
|
||||
Arg.ThreadWaiting.Wait();
|
||||
|
||||
if (FEX::HLE::_SyscallHandler->NeedXIDCheck()) {
|
||||
// The first time an application creates a thread, GLIBC installs their SETXID signal handler.
|
||||
@@ -146,6 +159,12 @@ FEX::HLE::ThreadStateObject* CreateNewThread(FEXCore::Context::Context* CTX, FEX
|
||||
|
||||
FEX::HLE::_SyscallHandler->TM.TrackThread(NewThread);
|
||||
|
||||
// Start running the thread
|
||||
Arg.StartRunningCV.NotifyOne();
|
||||
|
||||
// Wait for the thread to start running.
|
||||
Arg.StartRunningResponse.Wait();
|
||||
|
||||
return NewThread;
|
||||
}
|
||||
|
||||
@@ -170,7 +189,6 @@ uint64_t HandleNewClone(FEX::HLE::ThreadStateObject* Thread, FEXCore::Context::C
|
||||
|
||||
// CLONE_PARENT_SETTID, CLONE_CHILD_SETTID, CLONE_CHILD_CLEARTID, CLONE_PIDFD will be handled by kernel
|
||||
// Call execution thread directly since we already are on the new thread
|
||||
NewThread->Thread->StartRunning.NotifyAll(); // Clear the start running flag
|
||||
CreatedNewThreadObject = true;
|
||||
} else {
|
||||
// If we don't have CLONE_THREAD then we are effectively a fork
|
||||
@@ -220,12 +238,21 @@ uint64_t HandleNewClone(FEX::HLE::ThreadStateObject* Thread, FEXCore::Context::C
|
||||
|
||||
// Start exuting the thread directly
|
||||
// Our host clone starts in a new stack space, so it can't return back to the JIT space
|
||||
CTX->ExecutionThread(Thread->Thread);
|
||||
CTX->ExecuteThread(Thread->Thread);
|
||||
|
||||
FEX::HLE::_SyscallHandler->UninstallTLSState(Thread);
|
||||
|
||||
// The rest of the context remains as is and the thread will continue executing
|
||||
return Thread->Thread->StatusCode;
|
||||
return Thread->StatusCode;
|
||||
}
|
||||
|
||||
static int Clone3Fork(uint32_t flags) {
|
||||
struct clone_args cl_args = {
|
||||
.flags = (flags & (CLONE_FS | CLONE_FILES)),
|
||||
.exit_signal = SIGCHLD,
|
||||
};
|
||||
|
||||
return syscall(SYS_clone3, cl_args, sizeof(cl_args));
|
||||
}
|
||||
|
||||
uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::CpuStateFrame* Frame, uint32_t flags, void* stack,
|
||||
@@ -248,7 +275,7 @@ uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::Cp
|
||||
|
||||
// XXX: We don't currently support a real `vfork` as it causes problems.
|
||||
// Currently behaves like a fork (with wait after the fact), which isn't correct. Need to find where the problem is
|
||||
Result = fork();
|
||||
Result = Clone3Fork(flags);
|
||||
|
||||
if (Result == 0) {
|
||||
// Close the read end of the pipe.
|
||||
@@ -259,7 +286,7 @@ uint64_t ForkGuest(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::Cp
|
||||
close(VForkFDs[1]);
|
||||
}
|
||||
} else {
|
||||
Result = fork();
|
||||
Result = Clone3Fork(flags);
|
||||
}
|
||||
const bool IsChild = Result == 0;
|
||||
|
||||
@@ -380,8 +407,6 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
|
||||
REGISTER_SYSCALL_IMPL_FLAGS(exit, SyscallFlags::OPTIMIZETHROUGH | SyscallFlags::NOSYNCSTATEONENTRY | SyscallFlags::NORETURN,
|
||||
[](FEXCore::Core::CpuStateFrame* Frame, int status) -> uint64_t {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// TLS/DTV teardown is something FEX can't control. Disable glibc checking when we leave a pthread.
|
||||
// Since this thread is hard stopping, we can't track the TLS/DTV teardown in FEX's thread handling.
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator::HardDisable();
|
||||
@@ -393,7 +418,7 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
syscall(SYSCALL_DEF(futex), ThreadObject->ThreadInfo.clear_child_tid, FUTEX_WAKE, ~0ULL, 0, 0, 0);
|
||||
}
|
||||
|
||||
Thread->StatusCode = status;
|
||||
ThreadObject->StatusCode = status;
|
||||
FEX::HLE::_SyscallHandler->TM.StopThread(ThreadObject);
|
||||
|
||||
return 0;
|
||||
|
||||
@@ -22,6 +22,7 @@ FEX::HLE::ThreadStateObject* ThreadManager::CreateThread(uint64_t InitialRIP, ui
|
||||
|
||||
if (InheritThread) {
|
||||
FEX::HLE::_SyscallHandler->SeccompEmulator.InheritSeccompFilters(InheritThread, ThreadStateObject);
|
||||
ThreadStateObject->persona = InheritThread->persona;
|
||||
}
|
||||
|
||||
++IdleWaitRefCount;
|
||||
@@ -40,28 +41,25 @@ void ThreadManager::DestroyThread(FEX::HLE::ThreadStateObject* Thread, bool Need
|
||||
}
|
||||
|
||||
void ThreadManager::StopThread(FEX::HLE::ThreadStateObject* Thread) {
|
||||
if (Thread->Thread->RunningEvents.Running.exchange(false)) {
|
||||
SignalDelegation->SignalThread(Thread->Thread, FEXCore::Core::SignalEvent::Stop);
|
||||
}
|
||||
}
|
||||
|
||||
void ThreadManager::RunThread(FEX::HLE::ThreadStateObject* Thread) {
|
||||
// Tell the thread to start executing
|
||||
Thread->Thread->StartRunning.NotifyAll();
|
||||
SignalDelegation->SignalThread(Thread->Thread, SignalEvent::Stop);
|
||||
}
|
||||
|
||||
void ThreadManager::HandleThreadDeletion(FEX::HLE::ThreadStateObject* Thread, bool NeedsTLSUninstall) {
|
||||
if (Thread->Thread->ExecutionThread) {
|
||||
if (Thread->Thread->ExecutionThread->joinable()) {
|
||||
Thread->Thread->ExecutionThread->join(nullptr);
|
||||
if (Thread->ExecutionThread) {
|
||||
if (Thread->ExecutionThread->joinable()) {
|
||||
Thread->ExecutionThread->join(nullptr);
|
||||
}
|
||||
|
||||
if (Thread->Thread->ExecutionThread->IsSelf()) {
|
||||
Thread->Thread->ExecutionThread->detach();
|
||||
if (Thread->ExecutionThread->IsSelf()) {
|
||||
Thread->ExecutionThread->detach();
|
||||
}
|
||||
}
|
||||
|
||||
CTX->DestroyThread(Thread->Thread, NeedsTLSUninstall);
|
||||
if (NeedsTLSUninstall) {
|
||||
FEXCore::Allocator::UninstallTLSData(Thread->Thread);
|
||||
}
|
||||
|
||||
CTX->DestroyThread(Thread->Thread);
|
||||
FEX::HLE::_SyscallHandler->SeccompEmulator.FreeSeccompFilters(Thread);
|
||||
|
||||
delete Thread;
|
||||
@@ -73,7 +71,7 @@ void ThreadManager::NotifyPause() {
|
||||
// Tell all the threads that they should pause
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
for (auto& Thread : Threads) {
|
||||
SignalDelegation->SignalThread(Thread->Thread, FEXCore::Core::SignalEvent::Pause);
|
||||
SignalDelegation->SignalThread(Thread->Thread, SignalEvent::Pause);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -86,11 +84,7 @@ void ThreadManager::Run() {
|
||||
// Spin up all the threads
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
for (auto& Thread : Threads) {
|
||||
Thread->Thread->SignalReason.store(FEXCore::Core::SignalEvent::Return);
|
||||
}
|
||||
|
||||
for (auto& Thread : Threads) {
|
||||
Thread->Thread->StartRunning.NotifyAll();
|
||||
Thread->SignalReason.store(SignalEvent::Return);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -158,16 +152,7 @@ void ThreadManager::Stop(bool IgnoreCurrentThread) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Thread->Thread->RunningEvents.Running.load()) {
|
||||
StopThread(Thread);
|
||||
}
|
||||
|
||||
// If the thread is waiting to start but immediately killed then there can be a hang
|
||||
// This occurs in the case of gdb attach with immediate kill
|
||||
if (Thread->Thread->RunningEvents.WaitingToStart.load()) {
|
||||
Thread->Thread->RunningEvents.EarlyExit = true;
|
||||
Thread->Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
StopThread(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -178,23 +163,26 @@ void ThreadManager::Stop(bool IgnoreCurrentThread) {
|
||||
}
|
||||
|
||||
void ThreadManager::SleepThread(FEXCore::Context::Context* CTX, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto ThreadObject = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
|
||||
--IdleWaitRefCount;
|
||||
IdleWaitCV.notify_all();
|
||||
|
||||
Thread->RunningEvents.ThreadSleeping = true;
|
||||
ThreadObject->ThreadSleeping = true;
|
||||
|
||||
// Go to sleep
|
||||
Thread->StartRunning.Wait();
|
||||
ThreadObject->ThreadPaused.Wait();
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
++IdleWaitRefCount;
|
||||
Thread->RunningEvents.ThreadSleeping = false;
|
||||
ThreadObject->ThreadSleeping = false;
|
||||
|
||||
IdleWaitCV.notify_all();
|
||||
}
|
||||
|
||||
void ThreadManager::UnpauseThread(FEX::HLE::ThreadStateObject* Thread) {
|
||||
Thread->ThreadPaused.NotifyOne();
|
||||
}
|
||||
|
||||
void ThreadManager::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread, bool Child) {
|
||||
if (!Child) {
|
||||
return;
|
||||
@@ -207,9 +195,6 @@ void ThreadManager::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThre
|
||||
continue;
|
||||
}
|
||||
|
||||
// Setting running to false ensures that when they are shutdown we won't send signals to kill them
|
||||
DeadThread->Thread->RunningEvents.Running = false;
|
||||
|
||||
// Despite what google searches may susgest, glibc actually has special code to handle forks
|
||||
// with multiple active threads.
|
||||
// It cleans up the stacks of dead threads and marks them as terminated.
|
||||
|
||||
@@ -22,7 +22,20 @@ namespace FEX::HLE {
|
||||
class SyscallHandler;
|
||||
class SignalDelegator;
|
||||
|
||||
enum class SignalEvent : uint32_t {
|
||||
Nothing, // If the guest uses our signal we need to know it was errant on our end
|
||||
Pause,
|
||||
Stop,
|
||||
Return,
|
||||
ReturnRT,
|
||||
};
|
||||
|
||||
struct ThreadStateObject : public FEXCore::Allocator::FEXAllocOperators {
|
||||
struct DeferredSignalState {
|
||||
siginfo_t Info;
|
||||
int Signal;
|
||||
};
|
||||
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
|
||||
struct {
|
||||
@@ -50,11 +63,28 @@ struct ThreadStateObject : public FEXCore::Allocator::FEXAllocOperators {
|
||||
|
||||
uint64_t PendingSignals {};
|
||||
|
||||
// Queue of thread local signal frames that have been deferred.
|
||||
// Async signals aren't guaranteed to be delivered in any particular order, but FEX treats them as FILO.
|
||||
fextl::vector<DeferredSignalState> DeferredSignalFrames;
|
||||
} SignalInfo {};
|
||||
|
||||
// Seccomp thread specific data.
|
||||
uint32_t SeccompMode {SECCOMP_MODE_DISABLED};
|
||||
fextl::vector<FEX::HLE::SeccompEmulator::FilterInformation*> Filters {};
|
||||
|
||||
// personality emulation.
|
||||
uint32_t persona {};
|
||||
|
||||
FEXCore::Core::NonMovableUniquePtr<FEXCore::Threads::Thread> ExecutionThread;
|
||||
|
||||
// Thread signaling information
|
||||
std::atomic<SignalEvent> SignalReason {SignalEvent::Nothing};
|
||||
|
||||
// Thread pause handling
|
||||
std::atomic_bool ThreadSleeping {false};
|
||||
FEXCore::InterruptableConditionVariable ThreadPaused;
|
||||
|
||||
int StatusCode {};
|
||||
};
|
||||
|
||||
class ThreadManager final {
|
||||
@@ -84,7 +114,7 @@ public:
|
||||
|
||||
void DestroyThread(FEX::HLE::ThreadStateObject* Thread, bool NeedsTLSUninstall = false);
|
||||
void StopThread(FEX::HLE::ThreadStateObject* Thread);
|
||||
void RunThread(FEX::HLE::ThreadStateObject* Thread);
|
||||
void UnpauseThread(FEX::HLE::ThreadStateObject* Thread);
|
||||
|
||||
void Pause();
|
||||
void Run();
|
||||
|
||||
@@ -600,6 +600,11 @@ void RegisterFD(FEX::HLE::SyscallHandler* Handler) {
|
||||
for (size_t i = 0, num = 0; i < Result; ++num) {
|
||||
linux_dirent_64* Incoming = (linux_dirent_64*)(reinterpret_cast<uint64_t>(dirp) + i);
|
||||
Incoming->d_off = num;
|
||||
if (FEX::HLE::_SyscallHandler->FM.IsRootFSFD(fd, Incoming->d_ino)) {
|
||||
Result -= Incoming->d_reclen;
|
||||
memmove(Incoming, (linux_dirent_64*)(reinterpret_cast<uint64_t>(Incoming) + Incoming->d_reclen), Result - i);
|
||||
continue;
|
||||
}
|
||||
i += Incoming->d_reclen;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -112,6 +112,23 @@ void RegisterFD(FEX::HLE::SyscallHandler* Handler) {
|
||||
return GetDentsEmulation<false>(fd, reinterpret_cast<FEX::HLE::x64::linux_dirent*>(dirp), count);
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64(getdents64, [](FEXCore::Core::CpuStateFrame* Frame, int fd, void* dirp, uint32_t count) -> uint64_t {
|
||||
uint64_t Result = ::syscall(SYSCALL_DEF(getdents64), static_cast<uint64_t>(fd), dirp, static_cast<uint64_t>(count));
|
||||
if (Result != -1) {
|
||||
// Check for and hide the RootFS FD
|
||||
for (size_t i = 0; i < Result;) {
|
||||
linux_dirent_64* Incoming = (linux_dirent_64*)(reinterpret_cast<uint64_t>(dirp) + i);
|
||||
if (FEX::HLE::_SyscallHandler->FM.IsRootFSFD(fd, Incoming->d_ino)) {
|
||||
Result -= Incoming->d_reclen;
|
||||
memmove(Incoming, (linux_dirent_64*)(reinterpret_cast<uint64_t>(Incoming) + Incoming->d_reclen), Result - i);
|
||||
continue;
|
||||
}
|
||||
i += Incoming->d_reclen;
|
||||
}
|
||||
}
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64(dup2, [](FEXCore::Core::CpuStateFrame* Frame, int oldfd, int newfd) -> uint64_t {
|
||||
uint64_t Result = ::dup2(oldfd, newfd);
|
||||
SYSCALL_ERRNO();
|
||||
|
||||
@@ -746,7 +746,7 @@ VDSOMapping LoadVDSOThunks(bool Is64Bit, FEX::HLE::SyscallHandler* const Handler
|
||||
Mapping.VDSOSize = FEXCore::AlignUp(Mapping.VDSOSize, 4096);
|
||||
|
||||
// Map the VDSO file to memory
|
||||
Mapping.VDSOBase = Handler->GuestMmap(nullptr, nullptr, Mapping.VDSOSize, PROT_READ, MAP_PRIVATE, VDSOFD, 0);
|
||||
Mapping.VDSOBase = Handler->GuestMmap(nullptr, nullptr, Mapping.VDSOSize, PROT_READ, MAP_SHARED, VDSOFD, 0);
|
||||
|
||||
// Since we found our VDSO thunk library, find our host VDSO function implementations.
|
||||
LoadHostVDSO();
|
||||
|
||||
@@ -332,7 +332,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
|
||||
int LongJumpVal = setjmp(LongJumpHandler::LongJump);
|
||||
if (!LongJumpVal) {
|
||||
CTX->RunUntilExit(ParentThread->Thread);
|
||||
CTX->ExecuteThread(ParentThread->Thread);
|
||||
}
|
||||
|
||||
// Just re-use compare state. It also checks against the expected values in config.
|
||||
|
||||
@@ -51,6 +51,10 @@ public:
|
||||
push(r13);
|
||||
push(r14);
|
||||
push(r15);
|
||||
rdfsbase(rbx);
|
||||
push(rbx);
|
||||
rdgsbase(rbx);
|
||||
push(rbx);
|
||||
sub(rsp, 8);
|
||||
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
@@ -105,6 +109,10 @@ public:
|
||||
|
||||
add(rsp, 8);
|
||||
|
||||
pop(rbx);
|
||||
wrgsbase(rbx);
|
||||
pop(rbx);
|
||||
wrfsbase(rbx);
|
||||
pop(r15);
|
||||
pop(r14);
|
||||
pop(r13);
|
||||
|
||||
@@ -449,6 +449,9 @@ static void RethrowGuestException(const EXCEPTION_RECORD& Rec, ARM64_NT_CONTEXT&
|
||||
Args->Rec = FEX::Windows::HandleGuestException(Fault, Rec, Args->Context.Pc, Args->Context.X8);
|
||||
if (Args->Rec.ExceptionCode == EXCEPTION_SINGLE_STEP) {
|
||||
Args->Context.Cpsr &= ~(1 << 21); // PSTATE.SS
|
||||
} else if (Args->Rec.ExceptionCode == EXCEPTION_BREAKPOINT) {
|
||||
// INT3 will set RIP to the instruction following it, undo this (any edge cases with multibyte instructions that trigger breakpoints are bugs present in Windows also)
|
||||
Args->Context.Pc -= 1;
|
||||
}
|
||||
|
||||
Context.Sp = reinterpret_cast<uint64_t>(Args);
|
||||
|
||||
@@ -65,6 +65,8 @@ FEXCore::HostFeatures CPUFeatures::FetchHostFeatures(bool IsWine) {
|
||||
HostFeatures.CPUMIDRs.push_back(static_cast<uint32_t>(ReadRegU64(Key, "CP 4000")));
|
||||
RegCloseKey(Key);
|
||||
}
|
||||
|
||||
HostFeatures.SupportsCPUIndexInTPIDRRO = !IsWine;
|
||||
return HostFeatures;
|
||||
}
|
||||
|
||||
|
||||
@@ -29,8 +29,7 @@ HandleGuestException(FEXCore::Core::CpuStateFrame::SynchronousFaultDataStruct& F
|
||||
switch (Fault.TrapNo) {
|
||||
case FEXCore::X86State::X86_TRAPNO_DB: Dst.ExceptionCode = EXCEPTION_SINGLE_STEP; return Dst;
|
||||
case FEXCore::X86State::X86_TRAPNO_BP:
|
||||
Rip -= 1;
|
||||
Dst.ExceptionAddress = reinterpret_cast<void*>(Rip);
|
||||
Dst.ExceptionAddress = reinterpret_cast<void*>(Rip - 1);
|
||||
Dst.ExceptionCode = EXCEPTION_BREAKPOINT;
|
||||
Dst.NumberParameters = 1;
|
||||
Dst.ExceptionInformation[0] = 0;
|
||||
@@ -44,9 +43,9 @@ HandleGuestException(FEXCore::Core::CpuStateFrame::SynchronousFaultDataStruct& F
|
||||
if ((Fault.err_code & 0b111) == 0b010) {
|
||||
switch (Fault.err_code >> 3) {
|
||||
case 0x2d:
|
||||
Rip += 2;
|
||||
Rip += 3;
|
||||
Dst.ExceptionCode = EXCEPTION_BREAKPOINT;
|
||||
Dst.ExceptionAddress = reinterpret_cast<void*>(Rip + 1);
|
||||
Dst.ExceptionAddress = reinterpret_cast<void*>(Rip);
|
||||
Dst.NumberParameters = 1;
|
||||
Dst.ExceptionInformation[0] = Rax; // RAX
|
||||
// Note that ExceptionAddress doesn't equal the reported context RIP here, this discrepancy expected and not having it can trigger anti-debug logic.
|
||||
|
||||
@@ -74,6 +74,10 @@ struct TLS {
|
||||
explicit TLS(_TEB* TEB)
|
||||
: TEB(TEB) {}
|
||||
|
||||
WOW64INFO& Wow64Info() const {
|
||||
return *reinterpret_cast<WOW64INFO*>(TEB->TlsSlots[WOW64_TLS_WOW64INFO]);
|
||||
}
|
||||
|
||||
std::atomic<uint32_t>& ControlWord() const {
|
||||
// TODO: Change this when libc++ gains std::atomic_ref support
|
||||
return reinterpret_cast<std::atomic<uint32_t>&>(TEB->TlsSlots[FEXCore::ToUnderlying(Slot::CONTROL_WORD)]);
|
||||
@@ -479,6 +483,9 @@ void BTCpuProcessInit() {
|
||||
if (Sym) {
|
||||
WineUnixCall = *reinterpret_cast<decltype(WineUnixCall)*>(Sym);
|
||||
}
|
||||
|
||||
// wow64.dll will only initialise the cross-process queue if this is set
|
||||
GetTLS().Wow64Info().CpuFlags = WOW64_CPUFLAGS_SOFTWARE;
|
||||
}
|
||||
|
||||
void BTCpuProcessTerm(HANDLE Handle, BOOL After, ULONG Status) {}
|
||||
@@ -705,6 +712,7 @@ bool BTCpuResetToConsistentStateImpl(EXCEPTION_POINTERS* Ptrs) {
|
||||
if (Exception->ExceptionCode == EXCEPTION_SINGLE_STEP) {
|
||||
WowContext.EFlags &= ~(1 << FEXCore::X86State::RFLAG_TF_LOC);
|
||||
}
|
||||
// wow64.dll will handle adjusting PC in the dispatched context after a breakpoint
|
||||
|
||||
BTCpuSetContext(GetCurrentThread(), GetCurrentProcess(), nullptr, &WowContext);
|
||||
Context::UnlockJITContext();
|
||||
|
||||
@@ -13,8 +13,11 @@ extern "C" {
|
||||
#define NtCurrentProcess() ((HANDLE) ~(ULONG_PTR)0)
|
||||
#define NtCurrentThread() ((HANDLE) ~(ULONG_PTR)1)
|
||||
|
||||
#define WOW64_TLS_WOW64INFO 10
|
||||
#define WOW64_TLS_MAX_NUMBER 19
|
||||
|
||||
#define WOW64_CPUFLAGS_SOFTWARE 0x02
|
||||
|
||||
#define STATUS_EMULATION_SYSCALL ((NTSTATUS)0x40000039)
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
@@ -342,6 +345,16 @@ typedef struct __TEB { /* win32/win64 */
|
||||
GUID EffectiveContainerId; /* ff0/1828 */
|
||||
} __TEB, *__PTEB;
|
||||
|
||||
typedef struct _WOW64INFO {
|
||||
ULONG NativeSystemPageSize;
|
||||
ULONG CpuFlags;
|
||||
ULONG Wow64ExecuteFlags;
|
||||
ULONG unknown;
|
||||
ULONGLONG SectionHandle;
|
||||
ULONGLONG CrossProcessWorkList;
|
||||
USHORT NativeMachineType;
|
||||
USHORT EmulatedMachineType;
|
||||
} WOW64INFO;
|
||||
|
||||
typedef struct _THREAD_BASIC_INFORMATION {
|
||||
NTSTATUS ExitStatus;
|
||||
|
||||
@@ -79,6 +79,9 @@ Use `FHU::Filesystem::GetFilename` instead.
|
||||
#### std::filesystem::copy_file
|
||||
Use `FHU::Filesystem::CopyFile` instead.
|
||||
|
||||
#### std::filesystem::temp_directory_path
|
||||
See `GetTempFolder()` in `FEXServerClient.cpp` (split/move to `FHU::Filesystem` if needed by other users).
|
||||
|
||||
### `std::fstream`
|
||||
This API always allocates memory and should be avoided.
|
||||
Use a combination of open and fextl::string APIs instead of fstream.
|
||||
|
||||
+14
-14
@@ -1,4 +1,4 @@
|
||||
# FEX-2410
|
||||
# FEX-2412
|
||||
|
||||
## FEXCore
|
||||
See [FEXCore/Readme.md](../FEXCore/Readme.md) for more details
|
||||
@@ -16,18 +16,18 @@ See [FEXCore/Readme.md](../FEXCore/Readme.md) for more details
|
||||
IR to host code generation
|
||||
|
||||
#### arm64
|
||||
- [ALUOps.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp)
|
||||
- [Arm64Relocations.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/Arm64Relocations.cpp): relocation logic of the arm64 splatter backend
|
||||
- [AtomicOps.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/AtomicOps.cpp)
|
||||
- [BranchOps.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/BranchOps.cpp)
|
||||
- [ConversionOps.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/ConversionOps.cpp)
|
||||
- [EncryptionOps.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/EncryptionOps.cpp)
|
||||
- [JIT.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp): Main glue logic of the arm64 splatter backend
|
||||
- [JITClass.h](../FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h)
|
||||
- [MemoryOps.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/MemoryOps.cpp)
|
||||
- [MiscOps.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/MiscOps.cpp)
|
||||
- [MoveOps.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/MoveOps.cpp)
|
||||
- [VectorOps.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp)
|
||||
- [ALUOps.cpp](../FEXCore/Source/Interface/Core/JIT/ALUOps.cpp)
|
||||
- [Arm64Relocations.cpp](../FEXCore/Source/Interface/Core/JIT/Arm64Relocations.cpp): relocation logic of the arm64 splatter backend
|
||||
- [AtomicOps.cpp](../FEXCore/Source/Interface/Core/JIT/AtomicOps.cpp)
|
||||
- [BranchOps.cpp](../FEXCore/Source/Interface/Core/JIT/BranchOps.cpp)
|
||||
- [ConversionOps.cpp](../FEXCore/Source/Interface/Core/JIT/ConversionOps.cpp)
|
||||
- [EncryptionOps.cpp](../FEXCore/Source/Interface/Core/JIT/EncryptionOps.cpp)
|
||||
- [JIT.cpp](../FEXCore/Source/Interface/Core/JIT/JIT.cpp): Main glue logic of the arm64 splatter backend
|
||||
- [JITClass.h](../FEXCore/Source/Interface/Core/JIT/JITClass.h)
|
||||
- [MemoryOps.cpp](../FEXCore/Source/Interface/Core/JIT/MemoryOps.cpp)
|
||||
- [MiscOps.cpp](../FEXCore/Source/Interface/Core/JIT/MiscOps.cpp)
|
||||
- [MoveOps.cpp](../FEXCore/Source/Interface/Core/JIT/MoveOps.cpp)
|
||||
- [VectorOps.cpp](../FEXCore/Source/Interface/Core/JIT/VectorOps.cpp)
|
||||
|
||||
#### shared
|
||||
- [CPUBackend.h](../FEXCore/Source/Interface/Core/CPUBackend.h)
|
||||
@@ -50,9 +50,9 @@ Metadata that drives the frontend x86/64 decoding
|
||||
- [SecondaryModRMTables.cpp](../FEXCore/Source/Interface/Core/X86Tables/SecondaryModRMTables.cpp)
|
||||
- [SecondaryTables.cpp](../FEXCore/Source/Interface/Core/X86Tables/SecondaryTables.cpp)
|
||||
- [VEXTables.cpp](../FEXCore/Source/Interface/Core/X86Tables/VEXTables.cpp)
|
||||
- [X86TableGen.h](../FEXCore/Source/Interface/Core/X86Tables/X86TableGen.h)
|
||||
- [X86Tables.h](../FEXCore/Source/Interface/Core/X86Tables/X86Tables.h)
|
||||
- [X87Tables.cpp](../FEXCore/Source/Interface/Core/X86Tables/X87Tables.cpp)
|
||||
- [XOPTables.cpp](../FEXCore/Source/Interface/Core/X86Tables/XOPTables.cpp)
|
||||
- [X86Tables.cpp](../FEXCore/Source/Interface/Core/X86Tables.cpp)
|
||||
|
||||
#### x86-to-ir
|
||||
|
||||
@@ -1,3 +1,2 @@
|
||||
# Simulator can't handle all rounding modes
|
||||
Test_X87/RoundingPos.asm
|
||||
Test_X87/RoundingNeg.asm
|
||||
# Simulator can't handle `mrs x0, nzcv`
|
||||
Test_32Bit_SecondaryModRM/Reg_7_1.asm
|
||||
Loaded 100 of 1099 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user