mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 01:00:17 +02:00
Compare commits
210
Commits
FEX-2310
...
FEX-2311.1
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4a7839b5ac | ||
|
|
d8efcb39b8 | ||
|
|
fa8c35feba | ||
|
|
65b7d4007e | ||
|
|
2c0444e846 | ||
|
|
b92e716d0c | ||
|
|
a7a1365cf7 | ||
|
|
8ee5b5cf50 | ||
|
|
b45023bedf | ||
|
|
3702e513f5 | ||
|
|
829384e488 | ||
|
|
ed23fbe932 | ||
|
|
0af0427efd | ||
|
|
a499272d81 | ||
|
|
e91c5ff906 | ||
|
|
190f7c27e0 | ||
|
|
b15f0b5d36 | ||
|
|
e2c65189ff | ||
|
|
03f63f99a8 | ||
|
|
e305a9a0d5 | ||
|
|
3a90dbbb35 | ||
|
|
5103f2d92b | ||
|
|
d4a6b031ea | ||
|
|
319cf4bf3d | ||
|
|
d75c0f2c50 | ||
|
|
a586d3823d | ||
|
|
5522c6db9c | ||
|
|
367e1658ad | ||
|
|
bbad06f81a | ||
|
|
9612b2fe4b | ||
|
|
ef5503f0b7 | ||
|
|
26bf67ca76 | ||
|
|
e5df636efd | ||
|
|
f34f4a0227 | ||
|
|
f45722d2cd | ||
|
|
db7ef0e4b0 | ||
|
|
de10cbad98 | ||
|
|
e4d9c264d8 | ||
|
|
97b3efa90a | ||
|
|
18065199e3 | ||
|
|
77d92872bc | ||
|
|
460f13be71 | ||
|
|
47c9463217 | ||
|
|
15c825f362 | ||
|
|
bbd20b47ba | ||
|
|
5b70209728 | ||
|
|
a287f2a189 | ||
|
|
3a240e3b61 | ||
|
|
de1e593ec2 | ||
|
|
13cd8b33a2 | ||
|
|
0f26bc20a3 | ||
|
|
eacab3cc22 | ||
|
|
09e3371a0d | ||
|
|
ff3f7345b6 | ||
|
|
8181e53727 | ||
|
|
5431aa5a28 | ||
|
|
1a293cc542 | ||
|
|
b1e78934ad | ||
|
|
61f22911c7 | ||
|
|
fe8778bb96 | ||
|
|
14e5ea1e22 | ||
|
|
74f1205f33 | ||
|
|
4045bfd187 | ||
|
|
9db93a43dd | ||
|
|
a379d50729 | ||
|
|
f5822f83b0 | ||
|
|
5028434292 | ||
|
|
8f0461cac8 | ||
|
|
0b4cb23411 | ||
|
|
bab96b9441 | ||
|
|
11e2f14185 | ||
|
|
6177290e9d | ||
|
|
20a54913bd | ||
|
|
dd9ed89a7a | ||
|
|
a4e1e0a1fb | ||
|
|
5e9f69001d | ||
|
|
39c5ab1c81 | ||
|
|
5bf790324c | ||
|
|
adcdb32d49 | ||
|
|
f264578f12 | ||
|
|
7149da387a | ||
|
|
0f3d14e7c0 | ||
|
|
aad5080224 | ||
|
|
c77a3d673c | ||
|
|
0ff2e6e1e3 | ||
|
|
8538f5bac4 | ||
|
|
a305baf6e5 | ||
|
|
807619aa02 | ||
|
|
6db2125b41 | ||
|
|
9f6d80fe5d | ||
|
|
423ce12001 | ||
|
|
4edd72fc33 | ||
|
|
978f607dd9 | ||
|
|
9ba78c9771 | ||
|
|
e2144345c0 | ||
|
|
63e4c3682d | ||
|
|
2956e84ead | ||
|
|
4466c50c2b | ||
|
|
dd5ca1d349 | ||
|
|
95c756b466 | ||
|
|
99465faf63 | ||
|
|
e018917f76 | ||
|
|
a261d9909e | ||
|
|
6de8bc6848 | ||
|
|
d87155e4ee | ||
|
|
7484cacaf9 | ||
|
|
42259974c4 | ||
|
|
e455996dbd | ||
|
|
b5dd1d05e9 | ||
|
|
bbaf70da15 | ||
|
|
65e8d094ef | ||
|
|
4c801d594a | ||
|
|
d4403edea9 | ||
|
|
06ef012fb2 | ||
|
|
8f8f37684a | ||
|
|
7140b8d901 | ||
|
|
d5beba9423 | ||
|
|
826e15aea9 | ||
|
|
14e80ce228 | ||
|
|
165d3d3d4d | ||
|
|
887200e571 | ||
|
|
2c0bc0654d | ||
|
|
b3d76bd2f1 | ||
|
|
2e694412f4 | ||
|
|
d84577c36c | ||
|
|
cf9c2aa72c | ||
|
|
1cb8e4891c | ||
|
|
24f2796141 | ||
|
|
cb215b5f21 | ||
|
|
0cf2695772 | ||
|
|
6a6886305e | ||
|
|
5ef7537e61 | ||
|
|
167fe85cc3 | ||
|
|
cf65747667 | ||
|
|
27bb28b47f | ||
|
|
a00da800e7 | ||
|
|
bf835e80ac | ||
|
|
8f246b206b | ||
|
|
3c5c23bf36 | ||
|
|
5bcfaf4b9f | ||
|
|
1f6c6345d9 | ||
|
|
93792577eb | ||
|
|
3d23cd5765 | ||
|
|
fcc239552c | ||
|
|
8238de024f | ||
|
|
39e658f02a | ||
|
|
e89dd27f2a | ||
|
|
f85fae0041 | ||
|
|
65eec673fc | ||
|
|
5c93a085d2 | ||
|
|
4b356a7c2c | ||
|
|
2b67f87054 | ||
|
|
1ea40ae676 | ||
|
|
e0ef32e0bf | ||
|
|
a2b53c8eb0 | ||
|
|
21b6cccb4e | ||
|
|
d539829251 | ||
|
|
ef321e4bf8 | ||
|
|
47a0f14537 | ||
|
|
6d39f369b0 | ||
|
|
2304cfc530 | ||
|
|
1a39de4509 | ||
|
|
efb479f88f | ||
|
|
b27bf43901 | ||
|
|
c612fa8f2f | ||
|
|
4ccc40f697 | ||
|
|
cb53a704ba | ||
|
|
483423674a | ||
|
|
180d16af7a | ||
|
|
1acc038826 | ||
|
|
cc558fd5dc | ||
|
|
8f04223193 | ||
|
|
4be649c44e | ||
|
|
6253f4f708 | ||
|
|
cc2eef619c | ||
|
|
3bff42e6a7 | ||
|
|
2671246fef | ||
|
|
8cb8f090dd | ||
|
|
a37d89a7d5 | ||
|
|
252d7712ea | ||
|
|
c548625fbe | ||
|
|
f036a0b84f | ||
|
|
2e1389b25e | ||
|
|
a5f82a57fa | ||
|
|
cd83d3eb24 | ||
|
|
93ab8ab23c | ||
|
|
462fff2c67 | ||
|
|
8dab35cbf8 | ||
|
|
a1a479e69f | ||
|
|
6403290019 | ||
|
|
580bd50a00 | ||
|
|
b2a8b0ca12 | ||
|
|
f78bdf0852 | ||
|
|
5652eb4c5d | ||
|
|
a52bb47551 | ||
|
|
22590dde77 | ||
|
|
6543a80ff9 | ||
|
|
9c36d1061b | ||
|
|
5a3cc7b469 | ||
|
|
4cff3e5f1f | ||
|
|
a1eb571630 | ||
|
|
559cf6491a | ||
|
|
4bdda1eeb5 | ||
|
|
fc70fc3506 | ||
|
|
26ee63cc24 | ||
|
|
0092ea7c0b | ||
|
|
439a3b9c3a | ||
|
|
b4ddf36582 | ||
|
|
12c44f26e5 | ||
|
|
5b7ba06d5c |
No files matched your search
@@ -56,6 +56,16 @@ jobs:
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Set vixl_sim x86
|
||||
if: matrix.arch[1] == 'x64'
|
||||
run: |
|
||||
echo "VIXL_SIM_ENABLED=True" >> $GITHUB_ENV
|
||||
|
||||
- name: Set vixl_sim Arm64
|
||||
if: matrix.arch[1] == 'ARM64'
|
||||
run: |
|
||||
echo "VIXL_SIM_ENABLED=False" >> $GITHUB_ENV
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
@@ -64,7 +74,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=False -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=$VIXL_SIM_ENABLED -DENABLE_VIXL_DISASSEMBLER=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
+6
-5
@@ -293,10 +293,11 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
set(FEX_TUNE_COMPILE_FLAGS)
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
add_compile_options("-march=${TUNE_ARCH}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
endif()
|
||||
@@ -309,7 +310,7 @@ if (TUNE_CPU STREQUAL "native")
|
||||
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
|
||||
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=native")
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
@@ -323,19 +324,19 @@ if (TUNE_CPU STREQUAL "native")
|
||||
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${AARCH64_CPU}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${AARCH64_CPU}")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
add_compile_options("-march=native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=native")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${TUNE_CPU}")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${TUNE_CPU}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
|
||||
endif()
|
||||
|
||||
Vendored
-13
@@ -1,13 +0,0 @@
|
||||
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
|
||||
Version 2, December 2004
|
||||
|
||||
Copyright (C) 2018 Ryan Houdek <Sonicadvance1@gmail.com>
|
||||
|
||||
Everyone is permitted to copy and distribute verbatim or modified
|
||||
copies of this license document, and changing it is allowed as long
|
||||
as the name is changed.
|
||||
|
||||
DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE
|
||||
TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION
|
||||
|
||||
0. You just DO WHAT THE FUCK YOU WANT TO.
|
||||
Vendored
+1
-1
Submodule External/fmt updated: e57ca2e368...f5e54359df.
@@ -46,11 +46,15 @@ class OpDefinition:
|
||||
NumElements: str
|
||||
OpClass: str
|
||||
HasSideEffects: bool
|
||||
ImplicitFlagClobber: bool
|
||||
RAOverride: int
|
||||
SwitchGen: bool
|
||||
ArgPrinter: bool
|
||||
SSAArgNum: int
|
||||
NonSSAArgNum: int
|
||||
DynamicDispatch: bool
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
@@ -64,11 +68,15 @@ class OpDefinition:
|
||||
self.OpClass = None
|
||||
self.OpSize = 0
|
||||
self.HasSideEffects = False
|
||||
self.ImplicitFlagClobber = False
|
||||
self.RAOverride = -1
|
||||
self.SwitchGen = True
|
||||
self.ArgPrinter = True
|
||||
self.SSAArgNum = 0
|
||||
self.NonSSAArgNum = 0
|
||||
self.DynamicDispatch = False
|
||||
self.JITDispatch = True
|
||||
self.JITDispatchOverride = None
|
||||
self.Arguments = []
|
||||
self.EmitValidation = []
|
||||
self.Desc = []
|
||||
@@ -213,6 +221,9 @@ def parse_ops(ops):
|
||||
if "HasSideEffects" in op_val:
|
||||
OpDef.HasSideEffects = bool(op_val["HasSideEffects"])
|
||||
|
||||
if "ImplicitFlagClobber" in op_val:
|
||||
OpDef.ImplicitFlagClobber = bool(op_val["ImplicitFlagClobber"])
|
||||
|
||||
if "ArgPrinter" in op_val:
|
||||
OpDef.ArgPrinter = bool(op_val["ArgPrinter"])
|
||||
|
||||
@@ -228,6 +239,15 @@ def parse_ops(ops):
|
||||
if "Desc" in op_val:
|
||||
OpDef.Desc = op_val["Desc"]
|
||||
|
||||
if "DynamicDispatch" in op_val:
|
||||
OpDef.DynamicDispatch = bool(op_val["DynamicDispatch"])
|
||||
|
||||
if "JITDispatch" in op_val:
|
||||
OpDef.JITDispatch = bool(op_val["JITDispatch"])
|
||||
|
||||
if "JITDispatchOverride" in op_val:
|
||||
OpDef.JITDispatchOverride = op_val["JITDispatchOverride"]
|
||||
|
||||
# Do some fixups of the data here
|
||||
if len(OpDef.EmitValidation) != 0:
|
||||
for i in range(len(OpDef.EmitValidation)):
|
||||
@@ -357,6 +377,7 @@ def print_ir_sizes():
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool ImplicitFlagClobber(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
|
||||
|
||||
output_file.write("#undef IROP_SIZES\n")
|
||||
@@ -450,15 +471,17 @@ def print_ir_getraargs():
|
||||
def print_ir_hassideeffects():
|
||||
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> SideEffects = {\n")
|
||||
for op in IROps:
|
||||
output_file.write("\t{},\n".format(("true" if op.HasSideEffects else "false")))
|
||||
for array, prop in [("SideEffects", "HasSideEffects"),
|
||||
("ImplicitFlagClobbers", "ImplicitFlagClobber")]:
|
||||
output_file.write(f"constexpr std::array<uint8_t, OP_LAST + 1> {array} = {{\n")
|
||||
for op in IROps:
|
||||
output_file.write("\t{},\n".format(("true" if getattr(op, prop) else "false")))
|
||||
|
||||
output_file.write("};\n\n")
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("bool HasSideEffects(IROps Op) {\n")
|
||||
output_file.write(" return SideEffects[Op];\n")
|
||||
output_file.write("}\n")
|
||||
output_file.write(f"bool {prop}(IROps Op) {{\n")
|
||||
output_file.write(f" return {array}[Op];\n")
|
||||
output_file.write("}\n")
|
||||
|
||||
output_file.write("#undef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -627,6 +650,10 @@ def print_ir_allocator_helpers():
|
||||
|
||||
output_file.write(") {\n")
|
||||
|
||||
# Save NZCV if needed before clobbering NZCV
|
||||
if op.ImplicitFlagClobber:
|
||||
output_file.write("\t\tSaveNZCV();")
|
||||
|
||||
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
@@ -675,7 +702,8 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"\");\n".format(Validation))
|
||||
Sanitized = Validation.replace("\"", "\\\"")
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
|
||||
output_file.write("\t\treturn Op;\n")
|
||||
@@ -730,10 +758,38 @@ def print_ir_parser_switch_helper():
|
||||
output_file.write("#undef IROP_PARSER_SWITCH_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
|
||||
if (len(sys.argv) < 3):
|
||||
def print_ir_dispatcher_defs():
|
||||
output_dispatch_file.write("#ifdef IROP_DISPATCH_DEFS\n")
|
||||
for op in IROps:
|
||||
if op.Name != "Last" and op.SwitchGen and op.JITDispatch and op.JITDispatchOverride == None:
|
||||
output_dispatch_file.write("DEF_OP({});\n".format(op.Name))
|
||||
|
||||
output_dispatch_file.write("#undef IROP_DISPATCH_DEFS\n")
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
def print_ir_dispatcher_dispatch():
|
||||
output_dispatch_file.write("#ifdef IROP_DISPATCH_DISPATCH\n")
|
||||
for op in IROps:
|
||||
if op.Name != "Last" and op.JITDispatch:
|
||||
DispatchName = op.Name
|
||||
if op.JITDispatchOverride != None:
|
||||
DispatchName = op.JITDispatchOverride
|
||||
|
||||
if (op.DynamicDispatch):
|
||||
output_dispatch_file.write("REGISTER_OP_RT({}, {});\n".format(op.Name.upper(), DispatchName))
|
||||
else:
|
||||
output_dispatch_file.write("REGISTER_OP({}, {});\n".format(op.Name.upper(), DispatchName))
|
||||
|
||||
output_dispatch_file.write("#undef IROP_DISPATCH_DISPATCH\n")
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
|
||||
if (len(sys.argv) < 4):
|
||||
ExitError()
|
||||
|
||||
output_filename = sys.argv[2]
|
||||
output_dispatcher_filename = sys.argv[3]
|
||||
|
||||
json_file = open(sys.argv[1], "r")
|
||||
json_text = json_file.read()
|
||||
json_file.close()
|
||||
@@ -763,3 +819,10 @@ print_ir_allocator_helpers()
|
||||
print_ir_parser_switch_helper()
|
||||
|
||||
output_file.close()
|
||||
|
||||
output_dispatch_file = open(output_dispatcher_filename, "w")
|
||||
print_ir_dispatcher_defs()
|
||||
print_ir_dispatcher_dispatch()
|
||||
|
||||
output_dispatch_file.close()
|
||||
|
||||
@@ -221,15 +221,16 @@ configure_file(
|
||||
# Generate IR include file
|
||||
set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
|
||||
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
|
||||
set(OUTPUT_DISPATCHER_NAME "${OUTPUT_IR_FOLDER}/IRDefines_Dispatch.inc")
|
||||
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
|
||||
|
||||
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_NAME}"
|
||||
OUTPUT "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}" "${OUTPUT_DISPATCHER_NAME}"
|
||||
)
|
||||
|
||||
set_source_files_properties(${OUTPUT_NAME} PROPERTIES
|
||||
@@ -361,6 +362,7 @@ function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
@@ -369,6 +371,7 @@ endfunction()
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
target_compile_options(${Name} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
if (MINGW_BUILD)
|
||||
# Mingw build isn't building a linux shared library, so it can't have a SONAME.
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -8,6 +7,7 @@
|
||||
#include <FEXCore/Utils/CPUInfo.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
|
||||
@@ -82,7 +82,11 @@
|
||||
"ENABLEFLAGM": "enableflagm",
|
||||
"DISABLEFLAGM": "disableflagm",
|
||||
"ENABLEFLAGM2": "enableflagm2",
|
||||
"DISABLEFLAGM2": "disableflagm2"
|
||||
"DISABLEFLAGM2": "disableflagm2",
|
||||
"ENABLECRYPTO": "enablecrypto",
|
||||
"DISABLECRYPTO": "disablecrypto",
|
||||
"ENABLERPRES": "enablerpres",
|
||||
"DISABLERPRES": "disablerpres"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -100,7 +104,9 @@
|
||||
"\t{enable,disable}atomics: Will force enable or disable ARMv8.1 LSE atomics even if the host doesn't support it",
|
||||
"\t{enable,disable}fcma: Will force enable or disable fcma even if the host doesn't support it",
|
||||
"\t{enable,disable}flagm: Will force enable or disable flagm even if the host doesn't support it",
|
||||
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it"
|
||||
"\t{enable,disable}flagm2: Will force enable or disable flagm2 even if the host doesn't support it",
|
||||
"\t{enable,disable}crypto: Will force enable or disable crypto extensions even if the host doesn't support it",
|
||||
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it"
|
||||
]
|
||||
}
|
||||
},
|
||||
|
||||
@@ -48,6 +48,10 @@ namespace FEXCore::Context {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
CompileBlock(Thread->CurrentFrame, GuestRIP, MaxInst);
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason FEXCore::Context::ContextImpl::GetExitReason() {
|
||||
return ParentThread->ExitReason;
|
||||
}
|
||||
|
||||
@@ -90,6 +90,7 @@ namespace FEXCore::Context {
|
||||
void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) override;
|
||||
void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) override;
|
||||
|
||||
int GetProgramStatus() const override;
|
||||
|
||||
@@ -215,6 +216,9 @@ namespace FEXCore::Context {
|
||||
// this is for internal use
|
||||
bool ValidateIRarser { false };
|
||||
|
||||
// Used if the JIT needs to have its interrupt fault code emitted.
|
||||
bool NeedsPendingInterruptFaultCheck { false };
|
||||
|
||||
FEX_CONFIG_OPT(Multiblock, MULTIBLOCK);
|
||||
FEX_CONFIG_OPT(SingleStepConfig, SINGLESTEP);
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
@@ -281,9 +285,9 @@ namespace FEXCore::Context {
|
||||
~ContextImpl();
|
||||
|
||||
bool IsPaused() const { return !Running; }
|
||||
void WaitForThreadsToRun();
|
||||
void WaitForThreadsToRun() override;
|
||||
void Stop(bool IgnoreCurrentThread);
|
||||
void WaitForIdle();
|
||||
void WaitForIdle() override;
|
||||
void SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event);
|
||||
|
||||
bool GetGdbServerStatus() const { return DebugServer != nullptr; }
|
||||
@@ -322,7 +326,7 @@ namespace FEXCore::Context {
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
@@ -333,8 +337,8 @@ namespace FEXCore::Context {
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
[[nodiscard]] CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
uintptr_t CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
@@ -349,8 +353,6 @@ namespace FEXCore::Context {
|
||||
|
||||
void CopyMemoryMapping(FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread);
|
||||
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*>* GetThreads() { return &Threads; }
|
||||
|
||||
uint8_t GetGPRSize() const { return Config.Is64BitMode ? 8 : 4; }
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
@@ -398,6 +400,13 @@ namespace FEXCore::Context {
|
||||
|
||||
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
|
||||
|
||||
ThreadsState GetThreads() override {
|
||||
return ThreadsState {
|
||||
.ParentThread = ParentThread,
|
||||
.Threads = &Threads,
|
||||
};
|
||||
}
|
||||
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
|
||||
protected:
|
||||
|
||||
@@ -335,8 +335,8 @@ namespace x32 {
|
||||
}
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
: Emitter(size ? (uint8_t*)FEXCore::Allocator::VirtualAlloc(size, true) : nullptr, size)
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr, size_t size)
|
||||
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
|
||||
, EmitterCTX {ctx}
|
||||
#ifdef VIXL_SIMULATOR
|
||||
, Simulator {&SimDecoder}
|
||||
@@ -379,13 +379,6 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
}
|
||||
}
|
||||
|
||||
Arm64Emitter::~Arm64Emitter() {
|
||||
auto BufferSize = GetBufferSize();
|
||||
if (BufferSize) {
|
||||
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
@@ -581,6 +574,24 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Disable AFP features when spilling registers.
|
||||
//
|
||||
// Disable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
@@ -645,6 +656,38 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
bool FoundRegister{};
|
||||
for (auto Reg : StaticRegisters) {
|
||||
if (((1U << Reg.Idx()) & GPRFillMask)) {
|
||||
TmpReg = Reg;
|
||||
FoundRegister = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(FoundRegister, "Didn't have an SRA register to use as a temporary while spilling!");
|
||||
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Enable AFP features when filling JIT state.
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
|
||||
// Enable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
@@ -655,11 +698,11 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
// since all that matters is we restore them on a fill.
|
||||
// It's not a concern if they get trounced by something else.
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE) {
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
@@ -672,8 +715,6 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
|
||||
@@ -57,8 +57,7 @@ constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PRe
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size);
|
||||
~Arm64Emitter();
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl *ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
@@ -766,6 +766,11 @@ public:
|
||||
EvaluateIntoFlags(Op, 1, rn);
|
||||
}
|
||||
|
||||
void cfinv() {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'0000'0100'0000'0001'1111;
|
||||
dc32(Op);
|
||||
}
|
||||
|
||||
// Conditional compare - register
|
||||
void ccmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::StatusFlags flags, FEXCore::ARMEmitter::Condition Cond) {
|
||||
constexpr uint32_t Op = 0b0011'1010'010 << 21;
|
||||
|
||||
@@ -1384,12 +1384,10 @@ public:
|
||||
}
|
||||
|
||||
// SVE predicate initialize
|
||||
template <SubRegSize size>
|
||||
void ptrue(PRegister pd, PredicatePattern pattern) {
|
||||
void ptrue(SubRegSize size, PRegister pd, PredicatePattern pattern) {
|
||||
SVEPredicateMisc(0b1000, 0b10000, FEXCore::ToUnderlying(pattern), size, pd);
|
||||
}
|
||||
template <SubRegSize size>
|
||||
void ptrues(PRegister pd, PredicatePattern pattern) {
|
||||
void ptrues(SubRegSize size, PRegister pd, PredicatePattern pattern) {
|
||||
SVEPredicateMisc(0b1001, 0b10000, FEXCore::ToUnderlying(pattern), size, pd);
|
||||
}
|
||||
|
||||
|
||||
@@ -798,6 +798,52 @@ public:
|
||||
// XXX:
|
||||
//
|
||||
// Floating-point data-processing (1 source)
|
||||
void fmov(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000000, rd, rn);
|
||||
}
|
||||
void fabs(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000001, rd, rn);
|
||||
}
|
||||
void fneg(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000010, rd, rn);
|
||||
}
|
||||
void fsqrt(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b000011, rd, rn);
|
||||
}
|
||||
void frintn(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001000, rd, rn);
|
||||
}
|
||||
void frintp(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001001, rd, rn);
|
||||
}
|
||||
void frintm(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001010, rd, rn);
|
||||
}
|
||||
void frintz(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001011, rd, rn);
|
||||
}
|
||||
void frinta(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001100, rd, rn);
|
||||
}
|
||||
void frintx(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001110, rd, rn);
|
||||
}
|
||||
void frinti(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b001111, rd, rn);
|
||||
}
|
||||
void frint32z(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010000, rd, rn);
|
||||
}
|
||||
void frint32x(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010001, rd, rn);
|
||||
}
|
||||
void frint64z(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010010, rd, rn);
|
||||
}
|
||||
void frint64x(ScalarRegSize size, VRegister rd, VRegister rn) {
|
||||
Float1Source(size, 0, 0, 0b010011, rd, rn);
|
||||
}
|
||||
|
||||
void fmov(SRegister rd, SRegister rn) {
|
||||
Float1Source(0, 0, 0b00, 0b000000, rd.V(), rn.V());
|
||||
}
|
||||
@@ -1065,6 +1111,34 @@ public:
|
||||
}
|
||||
|
||||
// Floating-point data-processing (2 source)
|
||||
void fmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0000, rd, rn, rm);
|
||||
}
|
||||
void fdiv(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0001, rd, rn, rm);
|
||||
}
|
||||
void fadd(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0010, rd, rn, rm);
|
||||
}
|
||||
void fsub(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0011, rd, rn, rm);
|
||||
}
|
||||
void fmax(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0100, rd, rn, rm);
|
||||
}
|
||||
void fmin(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0101, rd, rn, rm);
|
||||
}
|
||||
void fmaxnm(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0110, rd, rn, rm);
|
||||
}
|
||||
void fminnm(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b0111, rd, rn, rm);
|
||||
}
|
||||
void fnmul(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm) {
|
||||
Float2Source(size, 0, 0, 0b1000, rd, rn, rm);
|
||||
}
|
||||
|
||||
void fmul(SRegister rd, SRegister rn, SRegister rm) {
|
||||
Float2Source(0, 0, 0b00, 0b0000, rd.V(), rn.V(), rm.V());
|
||||
}
|
||||
@@ -1150,6 +1224,16 @@ public:
|
||||
}
|
||||
|
||||
// Floating-point conditional select
|
||||
void fcsel(ScalarRegSize size, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
FloatConditionalSelect(0, 0, ConvertedSize, rd, rn, rm, Cond);
|
||||
}
|
||||
|
||||
void fcsel(SRegister rd, SRegister rn, SRegister rm, Condition Cond) {
|
||||
FloatConditionalSelect(0, 0, 0b00, rd.V(), rn.V(), rm.V(), Cond);
|
||||
}
|
||||
@@ -1305,6 +1389,16 @@ private:
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
void Float1Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float1Source(M, S, ConvertedSize, opcode, rd, rn);
|
||||
}
|
||||
|
||||
// Floating-point compare
|
||||
void FloatCompare(uint32_t M, uint32_t S, uint32_t ftype, uint32_t op, uint32_t opcode2, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0010'0000'0000'0000;
|
||||
@@ -1337,6 +1431,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
// Floating-point data-processing (2 source)
|
||||
|
||||
void Float2Source(uint32_t M, uint32_t S, uint32_t ptype, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1000'0000'0000;
|
||||
|
||||
@@ -1351,6 +1446,16 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void Float2Source(ScalarRegSize size, uint32_t M, uint32_t S, uint32_t opcode, VRegister rd, VRegister rn, VRegister rm) {
|
||||
LOGMAN_THROW_AA_FMT(size == ScalarRegSize::i16Bit || size == ScalarRegSize::i64Bit || size == ScalarRegSize::i32Bit, "Invalid size selected for {}", __func__);
|
||||
|
||||
const uint32_t ConvertedSize =
|
||||
size == ScalarRegSize::i64Bit ? 0b01 :
|
||||
size == ScalarRegSize::i32Bit ? 0b00 : 0b11;
|
||||
|
||||
Float2Source(M, S, ConvertedSize, opcode, rd, rn, rm);
|
||||
}
|
||||
|
||||
// Floating-point conditional select
|
||||
void FloatConditionalSelect(uint32_t M, uint32_t S, uint32_t ptype, VRegister rd, VRegister rn, VRegister rm, Condition Cond) {
|
||||
uint32_t Instr = 0b0001'1110'0010'0000'0000'1100'0000'0000;
|
||||
|
||||
@@ -17,6 +17,12 @@ constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant:
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT_UPPER
|
||||
{0x0000'0001'0000'0000ULL, 0x0000'0003'0000'0002ULL}, // NAMED_VECTOR_MOVMSKPS_SHIFT
|
||||
{0x040B'0E01'0B0E'0104ULL, 0x0C03'0609'0306'090CULL}, // NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE
|
||||
{0x0706'0504'FFFF'FFFFULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0110B
|
||||
{0x0706'0504'0302'0100ULL, 0xFFFF'FFFF'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_0111B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1001B
|
||||
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'FFFF'FFFFULL}, // NAMED_VECTOR_BLENDPS_1011B
|
||||
{0xFFFF'FFFF'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1101B
|
||||
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {
|
||||
@@ -176,6 +182,96 @@ constexpr static auto SHUFPS_LUT {
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto DPPS_MASK {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint32_t Val[4];
|
||||
};
|
||||
|
||||
std::array<LUTType, 16> TotalLUT{};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1U;
|
||||
}
|
||||
return 0U;
|
||||
};
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
LUT.Val[2] = GetLUT(i, 2);
|
||||
LUT.Val[3] = GetLUT(i, 3);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto DPPD_MASK {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint64_t Val[2];
|
||||
};
|
||||
|
||||
std::array<LUTType, 4> TotalLUT{};
|
||||
for (size_t i = 0; i < TotalLUT.size(); ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
constexpr auto GetLUT = [](size_t i, size_t Index) {
|
||||
if (i & (1U << Index)) {
|
||||
return -1ULL;
|
||||
}
|
||||
return 0ULL;
|
||||
};
|
||||
|
||||
LUT.Val[0] = GetLUT(i, 0);
|
||||
LUT.Val[1] = GetLUT(i, 1);
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
constexpr static auto PBLENDW_LUT {
|
||||
[]() consteval {
|
||||
struct LUTType {
|
||||
uint16_t Val[8];
|
||||
};
|
||||
// 16-bit words in [127:112], [111:96], [95:80], [79:64], [63:48], [47:32], [31:16], [15:0] are selected using 8-bit swizzle.
|
||||
// Expectation for this LUT is to simulate PBLENDW with ARM's TBX (one register) instruction.
|
||||
// PBLENDW behaviour:
|
||||
// 16-bit words from the source is moved in to the destination based on the bit in the swizzle.
|
||||
// Dest[15:0] = Swizzle[0] ? Src[15:0] : Dest[15:0]
|
||||
// Dest[31:16] = Swizzle[1] ? Src[31:16] : Dest[31:16]
|
||||
// Dest[47:32] = Swizzle[2] ? Src[47:32] : Dest[47:32]
|
||||
// Dest[63:48] = Swizzle[3] ? Src[63:48] : Dest[63:48]
|
||||
// Dest[79:64] = Swizzle[4] ? Src[79:64] : Dest[79:64]
|
||||
// Dest[95:80] = Swizzle[5] ? Src[95:80] : Dest[95:80]
|
||||
// Dest[111:96] = Swizzle[6] ? Src[111:96] : Dest[111:96]
|
||||
// Dest[127:112] = Swizzle[7] ? Src[127:112] : Dest[127:112]
|
||||
|
||||
std::array<LUTType, 256> TotalLUT{};
|
||||
const uint16_t WordSelectionSrc[8] = {
|
||||
0x01'00,
|
||||
0x03'02,
|
||||
0x05'04,
|
||||
0x07'06,
|
||||
0x09'08,
|
||||
0x0B'0A,
|
||||
0x0D'0C,
|
||||
0x0F'0E,
|
||||
};
|
||||
|
||||
constexpr uint16_t OriginalDest = 0xFF'FF;
|
||||
|
||||
for (size_t i = 0; i < 256; ++i) {
|
||||
auto &LUT = TotalLUT[i];
|
||||
for (size_t j = 0; j < 8; ++j) {
|
||||
LUT.Val[j] = ((i >> j) & 1) ? WordSelectionSrc[j] : OriginalDest;
|
||||
}
|
||||
}
|
||||
return TotalLUT;
|
||||
}()
|
||||
};
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {
|
||||
|
||||
@@ -194,6 +290,9 @@ CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t I
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast<uint64_t>(PSHUFHW_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast<uint64_t>(PSHUFD_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast<uint64_t>(SHUFPS_LUT.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] = reinterpret_cast<uint64_t>(DPPS_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] = reinterpret_cast<uint64_t>(DPPD_MASK.data());
|
||||
Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW] = reinterpret_cast<uint64_t>(PBLENDW_LUT.data());
|
||||
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
// Fill in telemetry values
|
||||
|
||||
@@ -33,14 +33,19 @@ namespace ProductNames {
|
||||
static const char ARM_A76[] = "Cortex-A76";
|
||||
static const char ARM_A76AE[] = "Cortex-A76AE";
|
||||
static const char ARM_V1[] = "Neoverse V1";
|
||||
static const char ARM_V2[] = "Neoverse V2";
|
||||
static const char ARM_A77[] = "Cortex-A77";
|
||||
static const char ARM_A78[] = "Cortex-A78";
|
||||
static const char ARM_A78AE[] = "Cortex-A78AE";
|
||||
static const char ARM_A78C[] = "Cortex-A78C";
|
||||
static const char ARM_A710[] = "Cortex-A710";
|
||||
static const char ARM_A715[] = "Cortex-A715";
|
||||
static const char ARM_A720[] = "Cortex-A720";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_X3[] = "Cortex-X3";
|
||||
static const char ARM_X4[] = "Cortex-X4";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
static const char ARM_E1[] = "Neoverse E1";
|
||||
@@ -49,6 +54,7 @@ namespace ProductNames {
|
||||
static const char ARM_A55[] = "Cortex-A55";
|
||||
static const char ARM_A65[] = "Cortex-A65";
|
||||
static const char ARM_A510[] = "Cortex-A510";
|
||||
static const char ARM_A520[] = "Cortex-A520";
|
||||
|
||||
static const char ARM_Kryo200[] = "Kryo 2xx";
|
||||
static const char ARM_Kryo300[] = "Kryo 3xx";
|
||||
@@ -138,10 +144,15 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 36> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 42> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
@@ -173,6 +184,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
|
||||
// Typically Little CPU cores
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm}, // Apple M1 Icestorm
|
||||
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
{0x41, 0xd05, 0, ProductNames::ARM_A55}, // A55
|
||||
|
||||
@@ -227,39 +227,46 @@ namespace FEXCore::Context {
|
||||
|
||||
// Currently these flags just map 1:1 inside of the resulting value.
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
if (i == X86State::RFLAG_PF_LOC || i == X86State::RFLAG_AF_LOC) {
|
||||
// Intentionally do nothing.
|
||||
// These contain multiple bits which can corrupt other members when compacted.
|
||||
continue;
|
||||
switch (i) {
|
||||
case X86State::RFLAG_CF_RAW_LOC:
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
case X86State::RFLAG_AF_RAW_LOC:
|
||||
case X86State::RFLAG_ZF_RAW_LOC:
|
||||
case X86State::RFLAG_SF_RAW_LOC:
|
||||
case X86State::RFLAG_OF_RAW_LOC:
|
||||
// Intentionally do nothing.
|
||||
// These contain multiple bits which can corrupt other members when compacted.
|
||||
break;
|
||||
default:
|
||||
EFLAGS |= uint32_t{Frame->State.flags[i]} << i;
|
||||
break;
|
||||
}
|
||||
|
||||
EFLAGS |= uint32_t{Frame->State.flags[i]} << i;
|
||||
}
|
||||
|
||||
// SF/ZF/CF/OF are packed in a 32-bit value in RFLAG_NZCV_LOC.
|
||||
uint32_t Packed_NZCV{};
|
||||
memcpy(&Packed_NZCV, &Frame->State.flags[X86State::RFLAG_NZCV_LOC], sizeof(Packed_NZCV));
|
||||
uint32_t OF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_LOC)) & 1;
|
||||
uint32_t CF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_LOC)) & 1;
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_LOC)) & 1;
|
||||
uint32_t SF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_LOC)) & 1;
|
||||
uint32_t OF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC)) & 1;
|
||||
uint32_t CF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC)) & 1;
|
||||
uint32_t ZF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC)) & 1;
|
||||
uint32_t SF = (Packed_NZCV >> IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC)) & 1;
|
||||
|
||||
// Pack in to EFLAGS
|
||||
EFLAGS |= OF << X86State::RFLAG_OF_LOC;
|
||||
EFLAGS |= CF << X86State::RFLAG_CF_LOC;
|
||||
EFLAGS |= ZF << X86State::RFLAG_ZF_LOC;
|
||||
EFLAGS |= SF << X86State::RFLAG_SF_LOC;
|
||||
EFLAGS |= OF << X86State::RFLAG_OF_RAW_LOC;
|
||||
EFLAGS |= CF << X86State::RFLAG_CF_RAW_LOC;
|
||||
EFLAGS |= ZF << X86State::RFLAG_ZF_RAW_LOC;
|
||||
EFLAGS |= SF << X86State::RFLAG_SF_RAW_LOC;
|
||||
|
||||
// PF calculation is deferred, calculate it now.
|
||||
// Popcount the 8-bit flag and then extract the lower bit.
|
||||
uint32_t PFByte = Frame->State.flags[X86State::RFLAG_PF_LOC];
|
||||
uint32_t PFByte = Frame->State.flags[X86State::RFLAG_PF_RAW_LOC];
|
||||
uint32_t PF = std::popcount(PFByte ^ 1) & 1;
|
||||
EFLAGS |= PF << X86State::RFLAG_PF_LOC;
|
||||
EFLAGS |= PF << X86State::RFLAG_PF_RAW_LOC;
|
||||
|
||||
// AF calculation is deferred, calculate it now.
|
||||
// XOR with PF byte and extract bit 4.
|
||||
uint32_t AF = ((Frame->State.flags[X86State::RFLAG_AF_LOC] ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_LOC;
|
||||
uint32_t AF = ((Frame->State.flags[X86State::RFLAG_AF_RAW_LOC] ^ PFByte) & (1 << 4)) ? 1 : 0;
|
||||
EFLAGS |= AF << X86State::RFLAG_AF_RAW_LOC;
|
||||
|
||||
return EFLAGS;
|
||||
}
|
||||
@@ -268,19 +275,19 @@ namespace FEXCore::Context {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
switch (i) {
|
||||
case X86State::RFLAG_OF_LOC:
|
||||
case X86State::RFLAG_CF_LOC:
|
||||
case X86State::RFLAG_ZF_LOC:
|
||||
case X86State::RFLAG_SF_LOC:
|
||||
case X86State::RFLAG_OF_RAW_LOC:
|
||||
case X86State::RFLAG_CF_RAW_LOC:
|
||||
case X86State::RFLAG_ZF_RAW_LOC:
|
||||
case X86State::RFLAG_SF_RAW_LOC:
|
||||
// Intentionally do nothing.
|
||||
break;
|
||||
case X86State::RFLAG_AF_LOC:
|
||||
case X86State::RFLAG_AF_RAW_LOC:
|
||||
// AF stored in bit 4 in our internal representation. It is also
|
||||
// XORed with byte 4 of the PF byte, but we write that as zero here so
|
||||
// we don't need any special handling for that.
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
|
||||
break;
|
||||
case X86State::RFLAG_PF_LOC:
|
||||
case X86State::RFLAG_PF_RAW_LOC:
|
||||
// PF is inverted in our internal representation.
|
||||
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? 0 : 1;
|
||||
break;
|
||||
@@ -292,10 +299,10 @@ namespace FEXCore::Context {
|
||||
|
||||
// Calculate packed NZCV
|
||||
uint32_t Packed_NZCV{};
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_OF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_ZF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_SF_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_OF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_OF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_CF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_CF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_ZF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_ZF_RAW_LOC) : 0;
|
||||
Packed_NZCV |= (EFLAGS & (1U << X86State::RFLAG_SF_RAW_LOC)) ? 1U << IR::OpDispatchBuilder::IndexNZCV(X86State::RFLAG_SF_RAW_LOC) : 0;
|
||||
memcpy(&Frame->State.flags[X86State::RFLAG_NZCV_LOC], &Packed_NZCV, sizeof(Packed_NZCV));
|
||||
|
||||
// Reserved, Read-As-1, Write-as-1
|
||||
@@ -361,6 +368,9 @@ namespace FEXCore::Context {
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#else
|
||||
// WIN32 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
|
||||
using namespace FEXCore::Core;
|
||||
@@ -381,7 +391,7 @@ namespace FEXCore::Context {
|
||||
void ContextImpl::StartGdbServer() {
|
||||
#ifndef _WIN32
|
||||
if (!DebugServer) {
|
||||
DebugServer = fextl::make_unique<GdbServer>(this);
|
||||
DebugServer = fextl::make_unique<GdbServer>(this, SignalDelegation, SyscallHandler);
|
||||
StartPaused = true;
|
||||
}
|
||||
#endif
|
||||
@@ -826,7 +836,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
ContextImpl::GenerateIRResult ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
ContextImpl::GenerateIRResult ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
@@ -851,7 +861,7 @@ namespace FEXCore::Context {
|
||||
|
||||
bool HadDispatchError {false};
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, MaxInst, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Thread, Start, Length);
|
||||
}
|
||||
@@ -1005,7 +1015,7 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
@@ -1056,7 +1066,7 @@ namespace FEXCore::Context {
|
||||
|
||||
if (IRList == nullptr) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols());
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
@@ -1077,7 +1087,7 @@ namespace FEXCore::Context {
|
||||
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
|
||||
// In the future with code caching getting wired up, we will pass the rest of the data forward.
|
||||
// TODO: Pass the data forward when code caching is wired up to this.
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get(), GetGdbServerStatus()).BlockEntry,
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get()).BlockEntry,
|
||||
.IRData = IRList,
|
||||
.DebugData = DebugData,
|
||||
.RAData = std::move(RAData),
|
||||
@@ -1098,7 +1108,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
@@ -1118,7 +1128,7 @@ namespace FEXCore::Context {
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {}, Length {};
|
||||
|
||||
auto [Code, IR, Data, RAData, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP);
|
||||
auto [Code, IR, Data, RAData, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
CodePtr = Code;
|
||||
IRList = IR;
|
||||
DebugData = Data;
|
||||
|
||||
@@ -53,15 +53,22 @@ uint64_t Dispatcher::GetCompileBlockPtr() {
|
||||
return CompileBlockPtr.Data;
|
||||
}
|
||||
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
|
||||
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
|
||||
: Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
|
||||
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
|
||||
, CTX {ctx}
|
||||
, config {config} {
|
||||
EmitDispatcher();
|
||||
}
|
||||
|
||||
Dispatcher::~Dispatcher() {
|
||||
auto BufferSize = GetBufferSize();
|
||||
if (BufferSize) {
|
||||
FEXCore::Allocator::VirtualFree(GetBufferBase(), BufferSize);
|
||||
}
|
||||
}
|
||||
|
||||
void Dispatcher::EmitDispatcher() {
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
@@ -121,8 +128,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
|
||||
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, RipReg.R());
|
||||
b(ARMEmitter::Condition::CC_NE, &FullLookup);
|
||||
sub(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &FullLookup);
|
||||
|
||||
br(ARMEmitter::Reg::r3);
|
||||
|
||||
@@ -167,8 +174,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
cmp(ARMEmitter::XReg::x1, RipReg);
|
||||
b(ARMEmitter::Condition::CC_NE, &NoBlock);
|
||||
sub(ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
|
||||
@@ -225,7 +232,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
@@ -262,7 +269,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
@@ -393,7 +400,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->X86CodeGen.CallbackReturn);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, 16);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, CTX->Config.Is64BitMode ? 16 : 12);
|
||||
str(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
@@ -549,40 +556,6 @@ void Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_
|
||||
|
||||
#endif
|
||||
|
||||
size_t Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
FEXCore::ARMEmitter::Emitter emit{CodeBuffer, MaxGDBPauseCheckSize};
|
||||
|
||||
ARMEmitter::ForwardLabel RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(FEXCore::Context::ContextImpl::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Thread));
|
||||
emit.ldr(ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, offsetof(FEXCore::Core::InternalThreadState, CTX)); // Get Context
|
||||
emit.ldr(ARMEmitter::WReg::w0, ARMEmitter::Reg::r0, offsetof(FEXCore::Context::ContextImpl, Config.RunningMode));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cbz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, &RunBlock);
|
||||
{
|
||||
ARMEmitter::ForwardLabel l_GuestRIP;
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.ldr(ARMEmitter::XReg::x0, &l_GuestRIP);
|
||||
emit.str(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Stop the thread
|
||||
emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.br(ARMEmitter::Reg::r0);
|
||||
emit.Bind(&l_GuestRIP);
|
||||
emit.dc64(GuestRIP);
|
||||
}
|
||||
emit.Bind(&RunBlock);
|
||||
|
||||
auto UsedBytes = emit.GetCursorOffset();
|
||||
emit.ClearICache(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
|
||||
@@ -43,7 +43,7 @@ public:
|
||||
static fextl::unique_ptr<Dispatcher> Create(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config);
|
||||
|
||||
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config);
|
||||
~Dispatcher() = default;
|
||||
~Dispatcher();
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
@@ -71,11 +71,6 @@ public:
|
||||
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
// These are across all arches for now
|
||||
static constexpr size_t MaxGDBPauseCheckSize = 128;
|
||||
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) ;
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
|
||||
@@ -1095,7 +1095,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, uint64_t MaxInst, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
@@ -1133,6 +1133,10 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
if (MaxInst == 0) {
|
||||
MaxInst = CTX->Config.MaxInstPerBlock;
|
||||
}
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1195,9 +1199,9 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
CanContinue = true;
|
||||
}
|
||||
|
||||
bool FinalInstruction = DecodedSize >= CTX->Config.MaxInstPerBlock ||
|
||||
bool FinalInstruction = DecodedSize >= MaxInst ||
|
||||
DecodedSize >= DefaultDecodedBufferSize ||
|
||||
TotalInstructions >= CTX->Config.MaxInstPerBlock;
|
||||
TotalInstructions >= MaxInst;
|
||||
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SETS_RIP) {
|
||||
// If we have multiblock enabled
|
||||
|
||||
@@ -34,7 +34,7 @@ public:
|
||||
|
||||
Decoder(FEXCore::Context::ContextImpl *ctx);
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, uint64_t MaxInst, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
|
||||
DecodedBlockInformation const *GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
|
||||
@@ -11,9 +11,6 @@ $end_info$
|
||||
#include <iomanip>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
@@ -28,6 +25,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/NetStream.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
@@ -72,7 +70,9 @@ void GdbServer::WaitForThreadWakeup() {
|
||||
ThreadBreakEvent.Wait();
|
||||
}
|
||||
|
||||
GdbServer::GdbServer(FEXCore::Context::ContextImpl *ctx) : CTX(ctx) {
|
||||
GdbServer::GdbServer(FEXCore::Context::Context *ctx, SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler)
|
||||
: CTX(ctx)
|
||||
, SyscallHandler {SyscallHandler} {
|
||||
// Pass all signals by default
|
||||
std::fill(PassSignals.begin(), PassSignals.end(), true);
|
||||
|
||||
@@ -80,19 +80,21 @@ GdbServer::GdbServer(FEXCore::Context::ContextImpl *ctx) : CTX(ctx) {
|
||||
if (ExitReason == FEXCore::Context::ExitReason::EXIT_DEBUG) {
|
||||
this->Break(SIGTRAP);
|
||||
}
|
||||
|
||||
if (ExitReason == FEXCore::Context::ExitReason::EXIT_SHUTDOWN) {
|
||||
CoreShuttingDown = true;
|
||||
}
|
||||
});
|
||||
|
||||
// This is a total hack as there is currently no way to resume once hitting a segfault
|
||||
// But it's semi-useful for debugging.
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
ctx->SignalDelegation->RegisterHostSignalHandler(Signal, [this] (FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
SignalDelegation->RegisterHostSignalHandler(Signal, [this] (FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
if (PassSignals[Signal]) {
|
||||
// Pass signal to the guest
|
||||
return false;
|
||||
}
|
||||
|
||||
this->CTX->Config.RunningMode = FEXCore::Context::CoreRunningMode::MODE_SINGLESTEP;
|
||||
|
||||
// Let GDB know that we have a signal
|
||||
this->Break(Signal);
|
||||
|
||||
@@ -140,7 +142,7 @@ static fextl::string encodeHex(const unsigned char *data, size_t length) {
|
||||
|
||||
static fextl::string getThreadName(uint32_t ThreadID) {
|
||||
const auto ThreadFile = fextl::fmt::format("/proc/{}/task/{}/comm", getpid(), ThreadID);
|
||||
fextl::string ThreadName {"<No Name>"};
|
||||
fextl::string ThreadName;
|
||||
FEXCore::FileLoading::LoadFile(ThreadName, ThreadFile);
|
||||
return ThreadName;
|
||||
}
|
||||
@@ -247,12 +249,16 @@ void GdbServer::SendACK(std::ostream &stream, bool NACK) {
|
||||
}
|
||||
}
|
||||
|
||||
struct X80Float {
|
||||
uint8_t Data[10];
|
||||
};
|
||||
|
||||
struct FEX_PACKED GDBContextDefinition {
|
||||
uint64_t gregs[Core::CPUState::NUM_GPRS];
|
||||
uint64_t rip;
|
||||
uint32_t eflags;
|
||||
uint32_t cs, ss, ds, es, fs, gs;
|
||||
X80SoftFloat mm[Core::CPUState::NUM_MMS];
|
||||
X80Float mm[Core::CPUState::NUM_MMS];
|
||||
uint32_t fctrl;
|
||||
uint32_t fstat;
|
||||
uint32_t dummies[6];
|
||||
@@ -265,10 +271,10 @@ fextl::string GdbServer::readRegs() {
|
||||
FEXCore::Core::CPUState state{};
|
||||
|
||||
auto Threads = CTX->GetThreads();
|
||||
FEXCore::Core::InternalThreadState *CurrentThread { CTX->ParentThread };
|
||||
FEXCore::Core::InternalThreadState *CurrentThread { Threads.ParentThread };
|
||||
bool Found = false;
|
||||
|
||||
for (auto &Thread : *Threads) {
|
||||
for (auto &Thread : *Threads.Threads) {
|
||||
if (Thread->ThreadManager.GetTID() != CurrentDebuggingThread) {
|
||||
continue;
|
||||
}
|
||||
@@ -280,7 +286,7 @@ fextl::string GdbServer::readRegs() {
|
||||
|
||||
if (!Found) {
|
||||
// If set to an invalid thread then just get the parent thread ID
|
||||
memcpy(&state, CTX->ParentThread->CurrentFrame, sizeof(state));
|
||||
memcpy(&state, Threads.ParentThread->CurrentFrame, sizeof(state));
|
||||
}
|
||||
|
||||
// Encode the GDB context definition
|
||||
@@ -316,10 +322,10 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
FEXCore::Core::CPUState state{};
|
||||
|
||||
auto Threads = CTX->GetThreads();
|
||||
FEXCore::Core::InternalThreadState *CurrentThread { CTX->ParentThread };
|
||||
FEXCore::Core::InternalThreadState *CurrentThread { Threads.ParentThread };
|
||||
bool Found = false;
|
||||
|
||||
for (auto &Thread : *Threads) {
|
||||
for (auto &Thread : *Threads.Threads) {
|
||||
if (Thread->ThreadManager.GetTID() != CurrentDebuggingThread) {
|
||||
continue;
|
||||
}
|
||||
@@ -331,7 +337,7 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
|
||||
if (!Found) {
|
||||
// If set to an invalid thread then just get the parent thread ID
|
||||
memcpy(&state, CTX->ParentThread->CurrentFrame, sizeof(state));
|
||||
memcpy(&state, Threads.ParentThread->CurrentFrame, sizeof(state));
|
||||
}
|
||||
|
||||
|
||||
@@ -354,7 +360,7 @@ GdbServer::HandledPacketType GdbServer::readReg(const fextl::string& packet) {
|
||||
}
|
||||
else if (addr >= offsetof(GDBContextDefinition, mm[0]) &&
|
||||
addr < offsetof(GDBContextDefinition, mm[8])) {
|
||||
return {encodeHex((unsigned char *)(&state.mm[(addr - offsetof(GDBContextDefinition, mm[0])) / sizeof(X80SoftFloat)]), sizeof(X80SoftFloat)), HandledPacketType::TYPE_ACK};
|
||||
return {encodeHex((unsigned char *)(&state.mm[(addr - offsetof(GDBContextDefinition, mm[0])) / sizeof(X80Float)]), sizeof(X80Float)), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
else if (addr == offsetof(GDBContextDefinition, fctrl)) {
|
||||
// XXX: We don't support this yet
|
||||
@@ -673,15 +679,18 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet)
|
||||
|
||||
ThreadString.clear();
|
||||
fextl::ostringstream ss;
|
||||
ss << "<?xml version=\"1.0\"?>\n";
|
||||
ss << "<threads>\n";
|
||||
for (auto &Thread : *Threads) {
|
||||
for (auto &Thread : *Threads.Threads) {
|
||||
// Thread id is in hex without 0x prefix
|
||||
ss << "\t<thread id=\"" << std::hex << Thread->ThreadManager.GetTID() << "\" name=\"" << getThreadName(Thread->ThreadManager.GetTID()) << "\">\n";
|
||||
ss << "\t</thread>\n";
|
||||
const auto ThreadName = getThreadName(Thread->ThreadManager.GetTID());
|
||||
ss << "<thread id=\"" << std::hex << Thread->ThreadManager.GetTID() << "\"";
|
||||
if (!ThreadName.empty()) {
|
||||
ss << " name=\"" << ThreadName << "\"";
|
||||
}
|
||||
ss << "/>\n";
|
||||
}
|
||||
|
||||
ss << "</threads>";
|
||||
ss << "</threads>\n";
|
||||
ss << std::flush;
|
||||
ThreadString = ss.str();
|
||||
}
|
||||
@@ -705,11 +714,11 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const fextl::string &packet)
|
||||
}
|
||||
|
||||
if (object == "auxv") {
|
||||
auto CodeLoader = CTX->SyscallHandler->GetCodeLoader();
|
||||
auto CodeLoader = SyscallHandler->GetCodeLoader();
|
||||
uint64_t auxv_ptr, auxv_size;
|
||||
CodeLoader->GetAuxv(auxv_ptr, auxv_size);
|
||||
fextl::string data;
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
if (Is64BitMode()) {
|
||||
data.resize(auxv_size);
|
||||
memcpy(data.data(), reinterpret_cast<void*>(auxv_ptr), data.size());
|
||||
}
|
||||
@@ -760,7 +769,7 @@ static size_t CheckMemMapping(uint64_t Address, size_t Size) {
|
||||
}
|
||||
|
||||
GdbServer::HandledPacketType GdbServer::handleProgramOffsets() {
|
||||
auto CodeLoader = CTX->SyscallHandler->GetCodeLoader();
|
||||
auto CodeLoader = SyscallHandler->GetCodeLoader();
|
||||
uint64_t BaseOffset = CodeLoader->GetBaseOffset();
|
||||
fextl::string str = fextl::fmt::format("Text={:x};Data={:x};Bss={:x}", BaseOffset, BaseOffset, BaseOffset);
|
||||
return {std::move(str), HandledPacketType::TYPE_ACK};
|
||||
@@ -904,10 +913,10 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
|
||||
fextl::ostringstream ss;
|
||||
ss << "m";
|
||||
for (size_t i = 0; i < Threads->size(); ++i) {
|
||||
auto Thread = Threads->at(i);
|
||||
for (size_t i = 0; i < Threads.Threads->size(); ++i) {
|
||||
auto Thread = Threads.Threads->at(i);
|
||||
ss << std::hex << Thread->ThreadManager.TID;
|
||||
if (i != (Threads->size() - 1)) {
|
||||
if (i != (Threads.Threads->size() - 1)) {
|
||||
ss << ",";
|
||||
}
|
||||
}
|
||||
@@ -928,7 +937,7 @@ GdbServer::HandledPacketType GdbServer::handleQuery(const fextl::string &packet)
|
||||
if (match("qC")) {
|
||||
// Returns the current Thread ID
|
||||
fextl::ostringstream ss;
|
||||
ss << "m" << std::hex << CTX->ParentThread->ThreadManager.TID;
|
||||
ss << "m" << std::hex << CTX->GetThreads().ParentThread->ThreadManager.TID;
|
||||
return {ss.str(), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (match("QStartNoAckMode")) {
|
||||
@@ -996,7 +1005,7 @@ GdbServer::HandledPacketType GdbServer::ThreadAction(char action, uint32_t tid)
|
||||
}
|
||||
case 't':
|
||||
// This thread isn't part of the thread pool
|
||||
CTX->Stop(false /* Ignore current thread */);
|
||||
CTX->Stop();
|
||||
return {"OK", HandledPacketType::TYPE_ACK};
|
||||
default:
|
||||
return {"E00", HandledPacketType::TYPE_ACK};
|
||||
@@ -1183,7 +1192,7 @@ GdbServer::HandledPacketType GdbServer::ProcessPacket(const fextl::string &packe
|
||||
case 'Z': // Inserts breakpoint or watchpoint
|
||||
return handleBreakpoint(packet);
|
||||
case 'k': // Kill the process
|
||||
CTX->Stop(false /* Ignore current thread */);
|
||||
CTX->Stop();
|
||||
CTX->WaitForIdle(); // Block until exit
|
||||
return {"", HandledPacketType::TYPE_NONE};
|
||||
default:
|
||||
@@ -1215,7 +1224,7 @@ void GdbServer::SendPacketPair(const HandledPacketType& response) {
|
||||
void GdbServer::GdbServerLoop() {
|
||||
OpenListenSocket();
|
||||
|
||||
while (!CTX->CoreShuttingDown.load()) {
|
||||
while (!CoreShuttingDown.load()) {
|
||||
CommsStream = OpenSocket();
|
||||
|
||||
HandledPacketType response{};
|
||||
|
||||
@@ -7,6 +7,7 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -20,13 +21,9 @@ $end_info$
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
class GdbServer {
|
||||
public:
|
||||
GdbServer(FEXCore::Context::ContextImpl *ctx);
|
||||
GdbServer(FEXCore::Context::Context *ctx, SignalDelegator *SignalDelegation, FEXCore::HLE::SyscallHandler *const SyscallHandler);
|
||||
|
||||
// Public for threading
|
||||
void GdbServerLoop();
|
||||
@@ -77,7 +74,8 @@ private:
|
||||
fextl::string readRegs();
|
||||
HandledPacketType readReg(const fextl::string& packet);
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::HLE::SyscallHandler *const SyscallHandler;
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> gdbServerThread;
|
||||
fextl::unique_ptr<std::iostream> CommsStream;
|
||||
std::mutex sendMutex;
|
||||
@@ -88,6 +86,7 @@ private:
|
||||
fextl::string OSDataString{};
|
||||
void buildLibraryMap();
|
||||
std::atomic<bool> LibraryMapChanged = true;
|
||||
std::atomic<bool> CoreShuttingDown{};
|
||||
fextl::string LibraryMapString{};
|
||||
|
||||
// Used to keep track of which signals to pass to the guest
|
||||
@@ -95,6 +94,7 @@ private:
|
||||
uint32_t CurrentDebuggingThread{};
|
||||
int ListenSocket{};
|
||||
FEX_CONFIG_OPT(Filename, APP_FILENAME);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
@@ -69,61 +69,29 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
return;
|
||||
}
|
||||
|
||||
const bool DisableAVX = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEAVX;
|
||||
const bool EnableAVX = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEAVX;
|
||||
LogMan::Throw::AFmt(!(DisableAVX && EnableAVX), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
#define ENABLE_DISABLE_OPTION(name, enum_name) \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
|
||||
|
||||
const bool DisableAVX2 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEAVX2;
|
||||
const bool EnableAVX2 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEAVX2;
|
||||
LogMan::Throw::AFmt(!(DisableAVX2 && EnableAVX2), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
ENABLE_DISABLE_OPTION(AVX, AVX);
|
||||
ENABLE_DISABLE_OPTION(AVX2, AVX2);
|
||||
ENABLE_DISABLE_OPTION(SVE, SVE);
|
||||
ENABLE_DISABLE_OPTION(AFP, AFP);
|
||||
ENABLE_DISABLE_OPTION(LRCPC, LRCPC);
|
||||
ENABLE_DISABLE_OPTION(LRCPC2, LRCPC2);
|
||||
ENABLE_DISABLE_OPTION(CSSC, CSSC);
|
||||
ENABLE_DISABLE_OPTION(PMULL128, PMULL128);
|
||||
ENABLE_DISABLE_OPTION(RNG, RNG);
|
||||
ENABLE_DISABLE_OPTION(CLZERO, CLZERO);
|
||||
ENABLE_DISABLE_OPTION(Atomics, ATOMICS);
|
||||
ENABLE_DISABLE_OPTION(FCMA, FCMA);
|
||||
ENABLE_DISABLE_OPTION(FlagM, FLAGM);
|
||||
ENABLE_DISABLE_OPTION(FlagM2, FLAGM2);
|
||||
ENABLE_DISABLE_OPTION(Crypto, CRYPTO);
|
||||
ENABLE_DISABLE_OPTION(RPRES, RPRES);
|
||||
|
||||
const bool DisableSVE = HostFeatures() & FEXCore::Config::HostFeatures::DISABLESVE;
|
||||
const bool EnableSVE = HostFeatures() & FEXCore::Config::HostFeatures::ENABLESVE;
|
||||
LogMan::Throw::AFmt(!(DisableSVE && EnableSVE), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableAFP = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEAFP;
|
||||
const bool EnableAFP = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEAFP;
|
||||
LogMan::Throw::AFmt(!(DisableAFP && EnableAFP), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableLRCPC = HostFeatures() & FEXCore::Config::HostFeatures::DISABLELRCPC;
|
||||
const bool EnableLRCPC = HostFeatures() & FEXCore::Config::HostFeatures::ENABLELRCPC;
|
||||
LogMan::Throw::AFmt(!(DisableLRCPC && EnableLRCPC), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableLRCPC2 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLELRCPC2;
|
||||
const bool EnableLRCPC2 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLELRCPC2;
|
||||
LogMan::Throw::AFmt(!(DisableLRCPC2 && EnableLRCPC2), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableCSSC = HostFeatures() & FEXCore::Config::HostFeatures::DISABLECSSC;
|
||||
const bool EnableCSSC = HostFeatures() & FEXCore::Config::HostFeatures::ENABLECSSC;
|
||||
LogMan::Throw::AFmt(!(DisableCSSC && EnableCSSC), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisablePMULL128 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEPMULL128;
|
||||
const bool EnablePMULL128 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEPMULL128;
|
||||
LogMan::Throw::AFmt(!(DisablePMULL128 && EnablePMULL128), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableRNG = HostFeatures() & FEXCore::Config::HostFeatures::DISABLERNG;
|
||||
const bool EnableRNG = HostFeatures() & FEXCore::Config::HostFeatures::ENABLERNG;
|
||||
LogMan::Throw::AFmt(!(DisableRNG && EnableRNG), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableCLZERO = HostFeatures() & FEXCore::Config::HostFeatures::DISABLECLZERO;
|
||||
const bool EnableCLZERO = HostFeatures() & FEXCore::Config::HostFeatures::ENABLECLZERO;
|
||||
LogMan::Throw::AFmt(!(DisableCLZERO && EnableCLZERO), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableAtomics = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEATOMICS;
|
||||
const bool EnableAtomics = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEATOMICS;
|
||||
LogMan::Throw::AFmt(!(DisableAtomics && EnableAtomics), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableFCMA = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEFCMA;
|
||||
const bool EnableFCMA = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEFCMA;
|
||||
LogMan::Throw::AFmt(!(DisableFCMA && EnableFCMA), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableFlagM = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEFLAGM;
|
||||
const bool EnableFlagM = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEFLAGM;
|
||||
LogMan::Throw::AFmt(!(DisableFlagM && EnableFlagM), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
|
||||
const bool DisableFlagM2 = HostFeatures() & FEXCore::Config::HostFeatures::DISABLEFLAGM2;
|
||||
const bool EnableFlagM2 = HostFeatures() & FEXCore::Config::HostFeatures::ENABLEFLAGM2;
|
||||
LogMan::Throw::AFmt(!(DisableFlagM2 && EnableFlagM2), "Disabling and Enabling CPU features are mutually exclusive");
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
|
||||
if (EnableAVX) {
|
||||
Features->SupportsAVX = true;
|
||||
@@ -144,10 +112,10 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
Features->SupportsSVE = false;
|
||||
}
|
||||
if (EnableAFP) {
|
||||
Features->SupportsFlushInputsToZero = true;
|
||||
Features->SupportsAFP = true;
|
||||
}
|
||||
else if (DisableAFP) {
|
||||
Features->SupportsFlushInputsToZero = false;
|
||||
Features->SupportsAFP = false;
|
||||
}
|
||||
if (EnableLRCPC) {
|
||||
Features->SupportsRCPC = true;
|
||||
@@ -209,11 +177,31 @@ static void OverrideFeatures(HostFeatures *Features) {
|
||||
else if (DisableFlagM2) {
|
||||
Features->SupportsFlagM2 = false;
|
||||
}
|
||||
if (EnableCrypto) {
|
||||
Features->SupportsAES = true;
|
||||
Features->SupportsCRC = true;
|
||||
Features->SupportsPMULL_128Bit = true;
|
||||
}
|
||||
else if (DisableCrypto) {
|
||||
Features->SupportsAES = false;
|
||||
Features->SupportsCRC = false;
|
||||
Features->SupportsPMULL_128Bit = false;
|
||||
}
|
||||
if (EnableRPRES) {
|
||||
Features->SupportsRPRES = true;
|
||||
}
|
||||
else if (DisableRPRES) {
|
||||
Features->SupportsRPRES = false;
|
||||
}
|
||||
}
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
// Vixl simulator doesn't support AFP.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kAFP);
|
||||
// Vixl simulator doesn't support RPRES.
|
||||
Features.Remove(vixl::CPUFeatures::Feature::kRPRES);
|
||||
#elif !defined(_WIN32)
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#else
|
||||
@@ -227,7 +215,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
|
||||
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsAFP = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
@@ -235,6 +223,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsFCMA = Features.Has(vixl::CPUFeatures::Feature::kFcma);
|
||||
SupportsFlagM = Features.Has(vixl::CPUFeatures::Feature::kFlagM);
|
||||
SupportsFlagM2 = Features.Has(vixl::CPUFeatures::Feature::kAXFlag);
|
||||
SupportsRPRES = Features.Has(vixl::CPUFeatures::Feature::kRPRES);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
@@ -254,6 +243,9 @@ HostFeatures::HostFeatures() {
|
||||
SupportsBMI2 = true;
|
||||
SupportsCLWB = true;
|
||||
|
||||
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
|
||||
SupportsAFP = false;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
@@ -336,7 +328,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsCLZERO = data[1] & 1;
|
||||
}
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsAFP = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
@@ -187,6 +187,16 @@ DEF_OP(Sub) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SubShift) {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(SubNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SubNZCV>();
|
||||
const IR::OpSize OpSize = Op->Size;
|
||||
@@ -1217,6 +1227,11 @@ DEF_OP(Bfi) {
|
||||
// If Dst and SrcDst match then this turns in to a simple BFI instruction.
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else if (Dst != Src) {
|
||||
// If the destination isn't the source then we can move the DstSrc and insert directly.
|
||||
mov(EmitSize, Dst, SrcDst);
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else {
|
||||
// Destination didn't match the dst source register.
|
||||
// TODO: Inefficient until FEX can have RA constraints here.
|
||||
@@ -1246,6 +1261,11 @@ DEF_OP(Bfxil) {
|
||||
// If Dst and SrcDst match then this turns in to a single instruction.
|
||||
bfxil(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else if (Dst != Src) {
|
||||
// If the destination isn't the source then we can move the DstSrc and insert directly.
|
||||
mov(EmitSize, Dst, SrcDst);
|
||||
bfxil(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else {
|
||||
// Destination didn't match the dst source register.
|
||||
// TODO: Inefficient until FEX can have RA constraints here.
|
||||
@@ -1328,7 +1348,7 @@ DEF_OP(Select) {
|
||||
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
|
||||
Op->Cond == FEXCore::IR::COND_ANDNZ;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!tests || IsGPR(Op->Cmp1.ID()), "Only GPRs can be tested");
|
||||
LOGMAN_THROW_A_FMT(!tests || IsGPR(Op->Cmp1.ID()), "Only GPRs can be tested");
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
|
||||
@@ -193,6 +193,33 @@ DEF_OP(AtomicAnd) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicCLR) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicCLR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stclrl(SubEmitSize, Src, MemSrc);
|
||||
}
|
||||
else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
bic(EmitSize, TMP2, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -247,6 +274,27 @@ DEF_OP(AtomicXor) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicNeg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
}
|
||||
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
@@ -75,9 +75,10 @@ DEF_OP(ExitFunction) {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::XReg::x3, ARMEmitter::ShiftType::LSL, 4);
|
||||
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x1, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
|
||||
cmp(ARMEmitter::XReg::x0, RipReg.X());
|
||||
b(ARMEmitter::Condition::CC_NE, &FullLookup);
|
||||
sub(TMP1, ARMEmitter::XReg::x0, RipReg.X());
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
|
||||
br(ARMEmitter::Reg::r1);
|
||||
|
||||
Bind(&FullLookup);
|
||||
@@ -467,7 +468,7 @@ DEF_OP(CPUID) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
DEF_OP(XGetBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -12,12 +12,12 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
DEF_OP(VAESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetVReg(Node), GetVReg(Op->Vector.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
DEF_OP(VAESEnc) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -44,7 +44,7 @@ DEF_OP(AESEnc) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
DEF_OP(VAESEncLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -69,7 +69,7 @@ DEF_OP(AESEncLast) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
DEF_OP(VAESDec) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -96,7 +96,7 @@ DEF_OP(AESDec) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
DEF_OP(VAESDecLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -121,7 +121,7 @@ DEF_OP(AESDecLast) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
DEF_OP(VAESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
|
||||
@@ -530,9 +530,11 @@ void Arm64JITCore::Op_NoOp(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx, 0)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128{ctx->HostFeatures.SupportsSVE}
|
||||
, HostSupportsSVE256{ctx->HostFeatures.SupportsAVX}
|
||||
, HostSupportsRPRES{ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP{ctx->HostFeatures.SupportsAFP}
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
@@ -687,8 +689,7 @@ bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
@@ -700,7 +701,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
@@ -744,16 +745,11 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
adr(TMP1, &JITCodeHeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
#ifdef _WIN32
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) -
|
||||
offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
#endif
|
||||
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(CodeData.BlockEntry, Entry);
|
||||
CursorIncrement(GDBSize);
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) -
|
||||
offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
@@ -800,272 +796,9 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
#define REGISTER_OP(op, x) case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
// ALU ops
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
REGISTER_OP(CONSTANT, Constant);
|
||||
REGISTER_OP(ENTRYPOINTOFFSET, EntrypointOffset);
|
||||
REGISTER_OP(INLINECONSTANT, InlineConstant);
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(ADDNZCV, AddNZCV);
|
||||
REGISTER_OP(ADCNZCV, AdcNZCV);
|
||||
REGISTER_OP(SBBNZCV, SbbNZCV);
|
||||
REGISTER_OP(TESTNZ, TestNZ);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(SUBNZCV, SubNZCV);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
REGISTER_OP(UDIV, UDiv);
|
||||
REGISTER_OP(REM, Rem);
|
||||
REGISTER_OP(UREM, URem);
|
||||
REGISTER_OP(MULH, MulH);
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(ORLSHL, Orlshl);
|
||||
REGISTER_OP(ORLSHR, Orlshr);
|
||||
REGISTER_OP(ORNROR, Ornror);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
REGISTER_OP(LSHL, Lshl);
|
||||
REGISTER_OP(LSHR, Lshr);
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
REGISTER_OP(LUREM, LURem);
|
||||
REGISTER_OP(NOT, Not);
|
||||
REGISTER_OP(POPCOUNT, Popcount);
|
||||
REGISTER_OP(FINDLSB, FindLSB);
|
||||
REGISTER_OP(FINDMSB, FindMSB);
|
||||
REGISTER_OP(FINDTRAILINGZEROES, FindTrailingZeroes);
|
||||
REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes);
|
||||
REGISTER_OP(REV, Rev);
|
||||
REGISTER_OP(BFI, Bfi);
|
||||
REGISTER_OP(BFXIL, Bfxil);
|
||||
REGISTER_OP(BFE, Bfe);
|
||||
REGISTER_OP(SBFE, Sbfe);
|
||||
REGISTER_OP(SELECT, Select);
|
||||
REGISTER_OP(VEXTRACTTOGPR, VExtractToGPR);
|
||||
REGISTER_OP(FLOAT_TOGPR_ZS, Float_ToGPR_ZS);
|
||||
REGISTER_OP(FLOAT_TOGPR_S, Float_ToGPR_S);
|
||||
REGISTER_OP(FCMP, FCmp);
|
||||
|
||||
// Atomic ops
|
||||
REGISTER_OP(CASPAIR, CASPair);
|
||||
REGISTER_OP(CAS, CAS);
|
||||
REGISTER_OP(ATOMICADD, AtomicAdd);
|
||||
REGISTER_OP(ATOMICSUB, AtomicSub);
|
||||
REGISTER_OP(ATOMICAND, AtomicAnd);
|
||||
REGISTER_OP(ATOMICOR, AtomicOr);
|
||||
REGISTER_OP(ATOMICXOR, AtomicXor);
|
||||
REGISTER_OP(ATOMICSWAP, AtomicSwap);
|
||||
REGISTER_OP(ATOMICFETCHADD, AtomicFetchAdd);
|
||||
REGISTER_OP(ATOMICFETCHSUB, AtomicFetchSub);
|
||||
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
|
||||
REGISTER_OP(ATOMICFETCHCLR, AtomicFetchCLR);
|
||||
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
|
||||
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
REGISTER_OP(TELEMETRYSETVALUE, TelemetrySetValue);
|
||||
|
||||
// Branch ops
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
REGISTER_OP(JUMP, Jump);
|
||||
REGISTER_OP(CONDJUMP, CondJump);
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
REGISTER_OP(VCASTFROMGPR, VCastFromGPR);
|
||||
REGISTER_OP(VDUPFROMGPR, VDupFromGPR);
|
||||
REGISTER_OP(FLOAT_FROMGPR_S, Float_FromGPR_S);
|
||||
REGISTER_OP(FLOAT_FTOF, Float_FToF);
|
||||
REGISTER_OP(VECTOR_STOF, Vector_SToF);
|
||||
REGISTER_OP(VECTOR_FTOZS, Vector_FToZS);
|
||||
REGISTER_OP(VECTOR_FTOS, Vector_FToS);
|
||||
REGISTER_OP(VECTOR_FTOF, Vector_FToF);
|
||||
REGISTER_OP(VECTOR_FTOI, Vector_FToI);
|
||||
|
||||
// Encryption ops
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
|
||||
// Flag ops
|
||||
REGISTER_OP(GETHOSTFLAG, GetHostFlag);
|
||||
|
||||
// Memory ops
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP_RT(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP_RT(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
REGISTER_OP(FILLREGISTER, FillRegister);
|
||||
REGISTER_OP(LOADFLAG, LoadFlag);
|
||||
REGISTER_OP(STOREFLAG, StoreFlag);
|
||||
REGISTER_OP(LOADMEM, LoadMem);
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
REGISTER_OP_RT(LOADMEMTSO, LoadMemTSO);
|
||||
REGISTER_OP_RT(STOREMEMTSO, StoreMemTSO);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
REGISTER_OP(VLOADVECTORELEMENT, VLoadVectorElement);
|
||||
REGISTER_OP(VSTOREVECTORELEMENT, VStoreVectorElement);
|
||||
REGISTER_OP(VBROADCASTFROMMEM, VBroadcastFromMem);
|
||||
REGISTER_OP(PUSH, Push);
|
||||
REGISTER_OP(MEMSET, MemSet);
|
||||
REGISTER_OP(MEMCPY, MemCpy);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINECLEAN, CacheLineClean);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
// Misc ops
|
||||
REGISTER_OP(DUMMY, NoOp);
|
||||
REGISTER_OP(IRHEADER, NoOp);
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PRINT, Print);
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
REGISTER_OP(RDRAND, RDRAND);
|
||||
REGISTER_OP(YIELD, Yield);
|
||||
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
|
||||
// Vector ops
|
||||
REGISTER_OP(VECTORZERO, VectorZero);
|
||||
REGISTER_OP(VECTORIMM, VectorImm);
|
||||
REGISTER_OP(LOADNAMEDVECTORCONSTANT, LoadNamedVectorConstant);
|
||||
REGISTER_OP(LOADNAMEDVECTORINDEXEDCONSTANT, LoadNamedVectorIndexedConstant);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
REGISTER_OP(VAND, VAnd);
|
||||
REGISTER_OP(VBIC, VBic);
|
||||
REGISTER_OP(VOR, VOr);
|
||||
REGISTER_OP(VXOR, VXor);
|
||||
REGISTER_OP(VADD, VAdd);
|
||||
REGISTER_OP(VSUB, VSub);
|
||||
REGISTER_OP(VUQADD, VUQAdd);
|
||||
REGISTER_OP(VUQSUB, VUQSub);
|
||||
REGISTER_OP(VSQADD, VSQAdd);
|
||||
REGISTER_OP(VSQSUB, VSQSub);
|
||||
REGISTER_OP(VADDP, VAddP);
|
||||
REGISTER_OP(VADDV, VAddV);
|
||||
REGISTER_OP(VUMINV, VUMinV);
|
||||
REGISTER_OP(VURAVG, VURAvg);
|
||||
REGISTER_OP(VABS, VAbs);
|
||||
REGISTER_OP(VFABS, VFAbs);
|
||||
REGISTER_OP(VPOPCOUNT, VPopcount);
|
||||
REGISTER_OP(VFADD, VFAdd);
|
||||
REGISTER_OP(VFADDP, VFAddP);
|
||||
REGISTER_OP(VFSUB, VFSub);
|
||||
REGISTER_OP(VFMUL, VFMul);
|
||||
REGISTER_OP(VFDIV, VFDiv);
|
||||
REGISTER_OP(VFMIN, VFMin);
|
||||
REGISTER_OP(VFMAX, VFMax);
|
||||
REGISTER_OP(VFRECP, VFRecp);
|
||||
REGISTER_OP(VFSQRT, VFSqrt);
|
||||
REGISTER_OP(VFRSQRT, VFRSqrt);
|
||||
REGISTER_OP(VNEG, VNeg);
|
||||
REGISTER_OP(VFNEG, VFNeg);
|
||||
REGISTER_OP(VNOT, VNot);
|
||||
REGISTER_OP(VUMIN, VUMin);
|
||||
REGISTER_OP(VSMIN, VSMin);
|
||||
REGISTER_OP(VUMAX, VUMax);
|
||||
REGISTER_OP(VSMAX, VSMax);
|
||||
REGISTER_OP(VZIP, VZip);
|
||||
REGISTER_OP(VZIP2, VZip2);
|
||||
REGISTER_OP(VUNZIP, VUnZip);
|
||||
REGISTER_OP(VUNZIP2, VUnZip2);
|
||||
REGISTER_OP(VTRN, VTrn);
|
||||
REGISTER_OP(VTRN2, VTrn2);
|
||||
REGISTER_OP(VBSL, VBSL);
|
||||
REGISTER_OP(VCMPEQ, VCMPEQ);
|
||||
REGISTER_OP(VCMPEQZ, VCMPEQZ);
|
||||
REGISTER_OP(VCMPGT, VCMPGT);
|
||||
REGISTER_OP(VCMPGTZ, VCMPGTZ);
|
||||
REGISTER_OP(VCMPLTZ, VCMPLTZ);
|
||||
REGISTER_OP(VFCMPEQ, VFCMPEQ);
|
||||
REGISTER_OP(VFCMPNEQ, VFCMPNEQ);
|
||||
REGISTER_OP(VFCMPLT, VFCMPLT);
|
||||
REGISTER_OP(VFCMPGT, VFCMPGT);
|
||||
REGISTER_OP(VFCMPLE, VFCMPLE);
|
||||
REGISTER_OP(VFCMPORD, VFCMPORD);
|
||||
REGISTER_OP(VFCMPUNO, VFCMPUNO);
|
||||
REGISTER_OP(VUSHL, VUShl);
|
||||
REGISTER_OP(VUSHR, VUShr);
|
||||
REGISTER_OP(VSSHR, VSShr);
|
||||
REGISTER_OP(VUSHLS, VUShlS);
|
||||
REGISTER_OP(VUSHRS, VUShrS);
|
||||
REGISTER_OP(VUSHRSWIDE, VUShrSWide);
|
||||
REGISTER_OP(VSSHRSWIDE, VSShrSWide);
|
||||
REGISTER_OP(VUSHLSWIDE, VUShlSWide);
|
||||
REGISTER_OP(VSSHRS, VSShrS);
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VUSHRI, VUShrI);
|
||||
REGISTER_OP(VSSHRI, VSShrI);
|
||||
REGISTER_OP(VSHLI, VShlI);
|
||||
REGISTER_OP(VUSHRNI, VUShrNI);
|
||||
REGISTER_OP(VUSHRNI2, VUShrNI2);
|
||||
REGISTER_OP(VSXTL, VSXTL);
|
||||
REGISTER_OP(VSXTL2, VSXTL2);
|
||||
REGISTER_OP(VUXTL, VUXTL);
|
||||
REGISTER_OP(VUXTL2, VUXTL2);
|
||||
REGISTER_OP(VSQXTN, VSQXTN);
|
||||
REGISTER_OP(VSQXTN2, VSQXTN2);
|
||||
REGISTER_OP(VSQXTNPAIR, VSQXTNPair);
|
||||
REGISTER_OP(VSQXTUN, VSQXTUN);
|
||||
REGISTER_OP(VSQXTUN2, VSQXTUN2);
|
||||
REGISTER_OP(VSQXTUNPAIR, VSQXTUNPair);
|
||||
REGISTER_OP(VSRSHR, VSRSHR);
|
||||
REGISTER_OP(VSQSHL, VSQSHL);
|
||||
REGISTER_OP(VUMUL, VMul);
|
||||
REGISTER_OP(VSMUL, VMul);
|
||||
REGISTER_OP(VUMULL, VUMull);
|
||||
REGISTER_OP(VSMULL, VSMull);
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUMULH, VUMulH);
|
||||
REGISTER_OP(VSMULH, VSMulH);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VTBL2, VTBL2);
|
||||
REGISTER_OP(VREV32, VRev32);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VFCADD, VFCADD);
|
||||
#define IROP_DISPATCH_DISPATCH
|
||||
#include <FEXCore/IR/IRDefines_Dispatch.inc>
|
||||
#undef REGISTER_OP
|
||||
|
||||
default:
|
||||
|
||||
@@ -26,6 +26,7 @@ $end_info$
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
@@ -43,7 +44,7 @@ public:
|
||||
[[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -58,6 +59,8 @@ private:
|
||||
|
||||
const bool HostSupportsSVE128{};
|
||||
const bool HostSupportsSVE256{};
|
||||
const bool HostSupportsRPRES{};
|
||||
const bool HostSupportsAFP{};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
@@ -113,6 +116,15 @@ private:
|
||||
return PhyReg;
|
||||
}
|
||||
|
||||
// Converts IR-base shift type to ARMEmitter shift type.
|
||||
// Will be a no-op, only a type conversion since the two definitions match.
|
||||
[[nodiscard]] ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
|
||||
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
|
||||
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
|
||||
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
|
||||
ARMEmitter::ShiftType::ROR;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
@@ -208,6 +220,11 @@ private:
|
||||
uint32_t SpillSlots{};
|
||||
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
|
||||
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store register style.
|
||||
OpType RT_LoadRegister;
|
||||
@@ -218,278 +235,21 @@ private:
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
// Dynamic Dispatcher supporting operations
|
||||
DEF_OP(LoadRegisterSRA);
|
||||
DEF_OP(StoreRegisterSRA);
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
|
||||
///< No-op Handler
|
||||
DEF_OP(NoOp);
|
||||
|
||||
///< ALU Ops
|
||||
DEF_OP(TruncElementPair);
|
||||
DEF_OP(Constant);
|
||||
DEF_OP(EntrypointOffset);
|
||||
DEF_OP(InlineConstant);
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(AddNZCV);
|
||||
DEF_OP(AdcNZCV);
|
||||
DEF_OP(SbbNZCV);
|
||||
DEF_OP(TestNZ);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(SubNZCV);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
DEF_OP(UDiv);
|
||||
DEF_OP(Rem);
|
||||
DEF_OP(URem);
|
||||
DEF_OP(MulH);
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(Orlshl);
|
||||
DEF_OP(Orlshr);
|
||||
DEF_OP(Ornror);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
DEF_OP(Lshl);
|
||||
DEF_OP(Lshr);
|
||||
DEF_OP(Ashr);
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
DEF_OP(LURem);
|
||||
DEF_OP(Zext);
|
||||
DEF_OP(Not);
|
||||
DEF_OP(Popcount);
|
||||
DEF_OP(FindLSB);
|
||||
DEF_OP(FindMSB);
|
||||
DEF_OP(FindTrailingZeroes);
|
||||
DEF_OP(CountLeadingZeroes);
|
||||
DEF_OP(Rev);
|
||||
DEF_OP(Bfi);
|
||||
DEF_OP(Bfxil);
|
||||
DEF_OP(Bfe);
|
||||
DEF_OP(Sbfe);
|
||||
DEF_OP(Select);
|
||||
DEF_OP(VExtractToGPR);
|
||||
DEF_OP(Float_ToGPR_ZU);
|
||||
DEF_OP(Float_ToGPR_ZS);
|
||||
DEF_OP(Float_ToGPR_S);
|
||||
DEF_OP(FCmp);
|
||||
|
||||
///< Atomic ops
|
||||
DEF_OP(CASPair);
|
||||
DEF_OP(CAS);
|
||||
DEF_OP(AtomicAdd);
|
||||
DEF_OP(AtomicSub);
|
||||
DEF_OP(AtomicAnd);
|
||||
DEF_OP(AtomicOr);
|
||||
DEF_OP(AtomicXor);
|
||||
DEF_OP(AtomicSwap);
|
||||
DEF_OP(AtomicFetchAdd);
|
||||
DEF_OP(AtomicFetchSub);
|
||||
DEF_OP(AtomicFetchAnd);
|
||||
DEF_OP(AtomicFetchCLR);
|
||||
DEF_OP(AtomicFetchOr);
|
||||
DEF_OP(AtomicFetchXor);
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
DEF_OP(TelemetrySetValue);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
DEF_OP(Jump);
|
||||
DEF_OP(CondJump);
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
DEF_OP(VCastFromGPR);
|
||||
DEF_OP(VDupFromGPR);
|
||||
DEF_OP(Float_FromGPR_S);
|
||||
DEF_OP(Float_FToF);
|
||||
DEF_OP(Vector_SToF);
|
||||
DEF_OP(Vector_FToZS);
|
||||
DEF_OP(Vector_FToS);
|
||||
DEF_OP(Vector_FToF);
|
||||
DEF_OP(Vector_FToI);
|
||||
|
||||
///< Flag ops
|
||||
DEF_OP(GetHostFlag);
|
||||
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadRegisterSRA);
|
||||
DEF_OP(StoreRegisterSRA);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
DEF_OP(FillRegister);
|
||||
DEF_OP(LoadFlag);
|
||||
DEF_OP(StoreFlag);
|
||||
DEF_OP(LoadMem);
|
||||
DEF_OP(StoreMem);
|
||||
DEF_OP(LoadMemTSO);
|
||||
DEF_OP(StoreMemTSO);
|
||||
DEF_OP(VLoadVectorMasked);
|
||||
DEF_OP(VStoreVectorMasked);
|
||||
DEF_OP(VLoadVectorElement);
|
||||
DEF_OP(VStoreVectorElement);
|
||||
DEF_OP(VBroadcastFromMem);
|
||||
DEF_OP(Push);
|
||||
DEF_OP(MemSet);
|
||||
DEF_OP(MemCpy);
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineClean);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
DEF_OP(RDRAND);
|
||||
DEF_OP(Yield);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(LoadNamedVectorConstant);
|
||||
DEF_OP(LoadNamedVectorIndexedConstant);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
DEF_OP(VOr);
|
||||
DEF_OP(VXor);
|
||||
DEF_OP(VAdd);
|
||||
DEF_OP(VSub);
|
||||
DEF_OP(VUQAdd);
|
||||
DEF_OP(VUQSub);
|
||||
DEF_OP(VSQAdd);
|
||||
DEF_OP(VSQSub);
|
||||
DEF_OP(VAddP);
|
||||
DEF_OP(VAddV);
|
||||
DEF_OP(VUMinV);
|
||||
DEF_OP(VURAvg);
|
||||
DEF_OP(VAbs);
|
||||
DEF_OP(VFAbs);
|
||||
DEF_OP(VPopcount);
|
||||
DEF_OP(VFAdd);
|
||||
DEF_OP(VFAddP);
|
||||
DEF_OP(VFSub);
|
||||
DEF_OP(VFMul);
|
||||
DEF_OP(VFDiv);
|
||||
DEF_OP(VFMin);
|
||||
DEF_OP(VFMax);
|
||||
DEF_OP(VFRecp);
|
||||
DEF_OP(VFSqrt);
|
||||
DEF_OP(VFRSqrt);
|
||||
DEF_OP(VNeg);
|
||||
DEF_OP(VFNeg);
|
||||
DEF_OP(VNot);
|
||||
DEF_OP(VUMin);
|
||||
DEF_OP(VSMin);
|
||||
DEF_OP(VUMax);
|
||||
DEF_OP(VSMax);
|
||||
DEF_OP(VZip);
|
||||
DEF_OP(VZip2);
|
||||
DEF_OP(VUnZip);
|
||||
DEF_OP(VUnZip2);
|
||||
DEF_OP(VTrn);
|
||||
DEF_OP(VTrn2);
|
||||
DEF_OP(VBSL);
|
||||
DEF_OP(VCMPEQ);
|
||||
DEF_OP(VCMPEQZ);
|
||||
DEF_OP(VCMPGT);
|
||||
DEF_OP(VCMPGTZ);
|
||||
DEF_OP(VCMPLTZ);
|
||||
DEF_OP(VFCMPEQ);
|
||||
DEF_OP(VFCMPNEQ);
|
||||
DEF_OP(VFCMPLT);
|
||||
DEF_OP(VFCMPGT);
|
||||
DEF_OP(VFCMPLE);
|
||||
DEF_OP(VFCMPORD);
|
||||
DEF_OP(VFCMPUNO);
|
||||
DEF_OP(VUShl);
|
||||
DEF_OP(VUShr);
|
||||
DEF_OP(VSShr);
|
||||
DEF_OP(VUShlS);
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VUShrSWide);
|
||||
DEF_OP(VSShrSWide);
|
||||
DEF_OP(VUShlSWide);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VUShrI);
|
||||
DEF_OP(VSShrI);
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
DEF_OP(VUXTL2);
|
||||
DEF_OP(VSQXTN);
|
||||
DEF_OP(VSQXTN2);
|
||||
DEF_OP(VSQXTNPair);
|
||||
DEF_OP(VSQXTUN);
|
||||
DEF_OP(VSQXTUN2);
|
||||
DEF_OP(VSQXTUNPair);
|
||||
DEF_OP(VSRSHR);
|
||||
DEF_OP(VSQSHL);
|
||||
DEF_OP(VMul);
|
||||
DEF_OP(VUMull);
|
||||
DEF_OP(VSMull);
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUMulH);
|
||||
DEF_OP(VSMulH);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VTBL2);
|
||||
DEF_OP(VRev32);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VFCADD);
|
||||
|
||||
///< Encryption ops
|
||||
DEF_OP(AESImc);
|
||||
DEF_OP(AESEnc);
|
||||
DEF_OP(AESEncLast);
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#define IROP_DISPATCH_DEFS
|
||||
#include <FEXCore/IR/IRDefines_Dispatch.inc>
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
@@ -1520,13 +1520,13 @@ DEF_OP(VBroadcastFromMem) {
|
||||
ElementSize == 4 || ElementSize == 8 ||
|
||||
ElementSize == 16, "Invalid element size");
|
||||
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
if (Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use SVE 256-bit broadcast");
|
||||
}
|
||||
if (Is256Bit && !HostSupportsSVE256) {
|
||||
LOGMAN_MSG_A_FMT("{}: 256-bit vectors must support SVE256", __func__);
|
||||
return;
|
||||
}
|
||||
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B.Zeroing()
|
||||
: PRED_TMP_16B.Zeroing();
|
||||
if (Is256Bit && HostSupportsSVE256) {
|
||||
const auto GoverningPredicate = PRED_TMP_32B.Zeroing();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
|
||||
@@ -143,7 +143,17 @@ DEF_OP(Print) {
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
|
||||
}
|
||||
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
else {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
}
|
||||
}
|
||||
else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -27,6 +27,43 @@ namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class PassManager;
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
DEFAULT,
|
||||
// TSO access behaviour
|
||||
TSO,
|
||||
// Non-TSO access behaviour
|
||||
NONTSO,
|
||||
// Non-temporal streaming
|
||||
STREAM,
|
||||
};
|
||||
|
||||
struct LoadSourceOptions {
|
||||
// Alignment of the load in bytes. -1 signifies unaligned
|
||||
int8_t Align = -1;
|
||||
|
||||
// Whether or not to load the data if a memory access occurs.
|
||||
// If set to false, then the address that would have been loaded from
|
||||
// will be returned instead.
|
||||
//
|
||||
// Note: If returning the address, make sure to apply the segment offset
|
||||
// after with AppendSegmentOffset().
|
||||
//
|
||||
bool LoadData = true;
|
||||
|
||||
// Use to force a load even if the underlying type isn't loadable.
|
||||
bool ForceLoad = false;
|
||||
|
||||
// Specifies the access type of the load.
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT;
|
||||
|
||||
// Whether or not a zero extend should clear the upper bits
|
||||
// in the register (e.g. an 8-bit load would clear the upper 24 bits
|
||||
// or 56 bits depending on the operating mode).
|
||||
// If true, no zero-extension occurs.
|
||||
bool AllowUpperGarbage = false;
|
||||
};
|
||||
|
||||
class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
@@ -98,6 +135,27 @@ public:
|
||||
ClearCachedNamedConstants();
|
||||
}
|
||||
|
||||
IRPair<IROp_Jump> Jump() {
|
||||
CalculateDeferredFlags();
|
||||
return _Jump();
|
||||
}
|
||||
IRPair<IROp_Jump> Jump(OrderedNode *_TargetBlock) {
|
||||
CalculateDeferredFlags();
|
||||
return _Jump(_TargetBlock);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(OrderedNode *_Cmp1, OrderedNode *_Cmp2, OrderedNode *_TrueBlock, OrderedNode *_FalseBlock, CondClassType _Cond = {COND_NEQ}, uint8_t _CompareSize = 0) {
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(_Cmp1, _Cmp2, _TrueBlock, _FalseBlock, _Cond, _CompareSize);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(OrderedNode *ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(ssa0, cond);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(OrderedNode *ssa0, OrderedNode *ssa1, OrderedNode *ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(ssa0, ssa1, ssa2, cond);
|
||||
}
|
||||
|
||||
bool FinishOp(uint64_t NextRIP, bool LastOp) {
|
||||
// If we are switching to a new block and this current block has yet to set a RIP
|
||||
// Then we need to insert an unconditional jump from the current block to the one we are going to
|
||||
@@ -123,7 +181,7 @@ public:
|
||||
_ExitFunction(RelocatedNextRIP);
|
||||
}
|
||||
else if (it != JumpTargets.end()) {
|
||||
_Jump(it->second.BlockEntry);
|
||||
Jump(it->second.BlockEntry);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -185,7 +243,8 @@ public:
|
||||
template<uint32_t SrcIndex>
|
||||
void MOVGPROp(OpcodeArgs);
|
||||
void MOVGPRNTOp(OpcodeArgs);
|
||||
void MOVVectorOp(OpcodeArgs);
|
||||
void MOVVectorAlignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp>
|
||||
void ALUOp(OpcodeArgs);
|
||||
@@ -322,8 +381,6 @@ public:
|
||||
void SGDTOp(OpcodeArgs);
|
||||
|
||||
// SSE
|
||||
void MOVAPS_MOVAPDOp(OpcodeArgs);
|
||||
void MOVUPS_MOVUPDOp(OpcodeArgs);
|
||||
void MOVLPOp(OpcodeArgs);
|
||||
void MOVHPDOp(OpcodeArgs);
|
||||
void MOVSDOp(OpcodeArgs);
|
||||
@@ -333,13 +390,12 @@ public:
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorALUROp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorScalarALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize, bool Scalar>
|
||||
void VectorUnaryOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorUnaryDuplicateOp(OpcodeArgs);
|
||||
|
||||
void MOVQOp(OpcodeArgs);
|
||||
void MOVQMMXOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void MOVMSKOp(OpcodeArgs);
|
||||
void MOVMSKOpOne(OpcodeArgs);
|
||||
@@ -379,7 +435,6 @@ public:
|
||||
void Vector_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Widen>
|
||||
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
@@ -387,7 +442,7 @@ public:
|
||||
void MOVBetweenGPR_FPR(OpcodeArgs);
|
||||
void TZCNT(OpcodeArgs);
|
||||
void LZCNT(OpcodeArgs);
|
||||
template<size_t ElementSize, bool Scalar>
|
||||
template<size_t ElementSize>
|
||||
void VFCMPOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void SHUFOp(OpcodeArgs);
|
||||
@@ -424,11 +479,9 @@ public:
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorALUOp(OpcodeArgs);
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarALUOp(OpcodeArgs);
|
||||
template <IROps IROp, size_t ElementSize, bool Scalar>
|
||||
void AVXVectorUnaryOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
template <size_t ElementSize>
|
||||
void AVXVectorRound(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize, size_t SrcElementSize>
|
||||
@@ -440,10 +493,44 @@ public:
|
||||
template <size_t SrcElementSize, bool Widen>
|
||||
void AVXVector_CVT_Int_To_Float(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorScalarInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarInsertALUOp(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorScalarUnaryInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarUnaryInsertALUOp(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void AVXVector_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
void InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize>
|
||||
void InsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
template <size_t DstElementSize>
|
||||
void AVXInsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void InsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void InsertScalarRound(OpcodeArgs);
|
||||
template <size_t ElementSize>
|
||||
void AVXInsertScalarRound(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void InsertScalarFCMPOp(OpcodeArgs);
|
||||
template <size_t ElementSize>
|
||||
void AVXInsertScalarFCMPOp(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize>
|
||||
void AVXCVTGPR_To_FPR(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
template <size_t ElementSize>
|
||||
void AVXVFCMPOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
@@ -489,6 +576,9 @@ public:
|
||||
void VMOVSDOp(OpcodeArgs);
|
||||
void VMOVSSOp(OpcodeArgs);
|
||||
|
||||
void VMOVAPS_VMOVAPDOp(OpcodeArgs);
|
||||
void VMOVUPS_VMOVUPDOp(OpcodeArgs);
|
||||
|
||||
void VMPSADBWOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
@@ -787,7 +877,7 @@ public:
|
||||
|
||||
template<size_t ElementSize, size_t DstElementSize, bool Signed>
|
||||
void ExtendVectorElements(OpcodeArgs);
|
||||
template<size_t ElementSize, bool Scalar>
|
||||
template<size_t ElementSize>
|
||||
void VectorRound(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
@@ -817,14 +907,18 @@ public:
|
||||
|
||||
static inline constexpr unsigned IndexNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_OF_LOC: return 28;
|
||||
case FEXCore::X86State::RFLAG_CF_LOC: return 29;
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC: return 30;
|
||||
case FEXCore::X86State::RFLAG_SF_LOC: return 31;
|
||||
case FEXCore::X86State::RFLAG_OF_RAW_LOC: return 28;
|
||||
case FEXCore::X86State::RFLAG_CF_RAW_LOC: return 29;
|
||||
case FEXCore::X86State::RFLAG_ZF_RAW_LOC: return 30;
|
||||
case FEXCore::X86State::RFLAG_SF_RAW_LOC: return 31;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
protected:
|
||||
void SaveNZCV() override {
|
||||
}
|
||||
|
||||
private:
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
@@ -848,10 +942,10 @@ private:
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
constexpr static unsigned FullNZCVMask =
|
||||
(1U << FEXCore::X86State::RFLAG_CF_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_ZF_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_SF_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_OF_LOC);
|
||||
(1U << FEXCore::X86State::RFLAG_CF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_SF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
static bool ContainsNZCV(unsigned BitMask) {
|
||||
return (BitMask & FullNZCVMask) != 0;
|
||||
@@ -859,10 +953,10 @@ private:
|
||||
|
||||
static bool IsNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_CF_LOC:
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC:
|
||||
case FEXCore::X86State::RFLAG_SF_LOC:
|
||||
case FEXCore::X86State::RFLAG_OF_LOC:
|
||||
case FEXCore::X86State::RFLAG_CF_RAW_LOC:
|
||||
case FEXCore::X86State::RFLAG_ZF_RAW_LOC:
|
||||
case FEXCore::X86State::RFLAG_SF_RAW_LOC:
|
||||
case FEXCore::X86State::RFLAG_OF_RAW_LOC:
|
||||
return true;
|
||||
|
||||
default:
|
||||
@@ -889,8 +983,7 @@ private:
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
void AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
|
||||
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void AVXVectorVariableBlend(OpcodeArgs);
|
||||
@@ -907,6 +1000,10 @@ private:
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, size_t ElementSize);
|
||||
|
||||
OrderedNode* VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
OrderedNode* ExtendVectorElementsImpl(OpcodeArgs, size_t ElementSize,
|
||||
size_t DstElementSize, bool Signed);
|
||||
|
||||
@@ -930,7 +1027,8 @@ private:
|
||||
|
||||
OrderedNode* PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
const X86Tables::DecodedOperand& Imm,
|
||||
bool IsAVX);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
|
||||
|
||||
@@ -997,25 +1095,69 @@ private:
|
||||
void MOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
void VMOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool Scalar,
|
||||
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src1, OrderedNode *Src2, uint8_t CompType);
|
||||
|
||||
void VTESTOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
|
||||
void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
|
||||
// x86 ALU scalar operations operate in three different ways
|
||||
// - AVX512: Writemask shenanigans that we don't care about.
|
||||
// - AVX/VEX: Two source
|
||||
// - Example 32bit VADDSS Dest, Src1, Src2
|
||||
// - Dest[31:0] = Src1[31:0] + Src2[31:0]
|
||||
// - Dest[127:32] = Src1[127:32]
|
||||
// - SSE: Scalar operation inserts in to the low bits, upper bits completely unaffected.
|
||||
// - Example 32bit ADDSS Dest, Src
|
||||
// - Dest[31:0] = Dest[31:0] + Src[31:0]
|
||||
// - Dest[{256,128}:32] = (Unmodified)
|
||||
OrderedNode* VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertCVTGPR_To_FPRImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t DstElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
OrderedNode* InsertScalarRoundImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
uint64_t Mode, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertScalarFCMPOpImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
uint8_t CompType, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* VectorRoundImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, uint64_t Mode, bool IsScalar);
|
||||
OrderedNode *Src, uint64_t Mode);
|
||||
|
||||
OrderedNode* Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
void Vector_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize);
|
||||
void Vector_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize, bool IsAVX);
|
||||
|
||||
OrderedNode* Vector_CVT_Float_To_IntImpl(OpcodeArgs, size_t SrcElementSize, bool Narrow, bool HostRoundingMode);
|
||||
|
||||
@@ -1045,27 +1187,22 @@ private:
|
||||
|
||||
void UpdatePrefixFromSegment(OrderedNode *Segment, uint32_t SegmentReg);
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
ACCESS_DEFAULT,
|
||||
// TSO access behaviour
|
||||
ACCESS_TSO,
|
||||
// Non-TSO access behaviour
|
||||
ACCESS_NONTSO,
|
||||
// Non-temporal streaming
|
||||
ACCESS_STREAM,
|
||||
};
|
||||
OrderedNode *LoadGPRRegister(uint32_t GPR, int8_t Size = -1, uint8_t Offset = 0, bool AllowUpperGarbage = false);
|
||||
OrderedNode *LoadXMMRegister(uint32_t XMM);
|
||||
void StoreGPRRegister(uint32_t GPR, OrderedNode *const Src, int8_t Size = -1, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, OrderedNode *const Src);
|
||||
|
||||
OrderedNode *GetRelocatedPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT, bool AllowUpperGarbage = false);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT, bool AllowUpperGarbage = false);
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
|
||||
OrderedNode *LoadSource(RegisterClassType Class, X86Tables::DecodedOp const& Op, X86Tables::DecodedOperand const& Operand, uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
OrderedNode *LoadSource_WithOpSize(RegisterClassType Class, X86Tables::DecodedOp const& Op, X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
|
||||
constexpr OpSize GetGuestVectorLength() const {
|
||||
return CTX->HostFeatures.SupportsAVX ? OpSize::i256Bit : OpSize::i128Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]] static uint32_t GPROffset(X86State::X86Reg reg) {
|
||||
LOGMAN_THROW_AA_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
@@ -1089,17 +1226,17 @@ private:
|
||||
|
||||
static inline constexpr unsigned NZCVIndexMask(unsigned BitMask) {
|
||||
unsigned NZCVMask{};
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_OF_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_OF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_CF_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_ZF_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
}
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_SF_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC);
|
||||
if (BitMask & (1U << FEXCore::X86State::RFLAG_SF_RAW_LOC)) {
|
||||
NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
}
|
||||
return NZCVMask;
|
||||
}
|
||||
@@ -1132,8 +1269,8 @@ private:
|
||||
|
||||
// Mask out the NZ bits, clearing CV. Even if the code sets CV after, this can end up faster
|
||||
// moves by allowing orlshl to be used instead of bfi.
|
||||
PossiblySetNZCVBits = (1u << IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_LOC));
|
||||
PossiblySetNZCVBits = (1u << IndexNZCV(FEXCore::X86State::RFLAG_SF_RAW_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
SetNZCV(_And(OpSize::i32Bit, OldNZCV, _Constant(PossiblySetNZCVBits)));
|
||||
}
|
||||
|
||||
@@ -1176,7 +1313,7 @@ private:
|
||||
// bits. This allows us to defer the extract in the usual case. When it is
|
||||
// read, bit 4 is extracted. In order to write a constant value of AF, that
|
||||
// means we need to left-shift here to compensate.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(Constant << 4));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Constant(Constant << 4));
|
||||
}
|
||||
|
||||
void ZeroMultipleFlags(uint32_t BitMask);
|
||||
|
||||
@@ -23,8 +23,8 @@ class OrderedNode;
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Tmp = _Ror(OpSize::i32Bit, _VExtractToGPR(16, 4, Dest, 3), _Constant(32, 2));
|
||||
auto Top = _Add(OpSize::i32Bit, _VExtractToGPR(16, 4, Src, 3), Tmp);
|
||||
@@ -34,8 +34,8 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode *NewVec = _VExtr(16, 8, Dest, Src, 1);
|
||||
|
||||
@@ -46,8 +46,8 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ROR by 31 is equivalent to a ROL by 1
|
||||
auto ThirtyOne = _Constant(32, 31);
|
||||
@@ -102,8 +102,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(32, k_array[Imm8]);
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto W1 = _VExtractToGPR(16, 4, Src, 2);
|
||||
@@ -155,8 +155,8 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))), _Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W4 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto W3 = _VExtractToGPR(16, 4, Dest, 3);
|
||||
@@ -182,8 +182,8 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))), _Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 3);
|
||||
@@ -214,8 +214,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), _Ror(OpSize::i32Bit, E, _Constant(32, 11))), _Ror(OpSize::i32Bit, E, _Constant(32, 25)));
|
||||
};
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
@@ -260,14 +260,14 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEnc(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -280,8 +280,8 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -289,8 +289,8 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -303,8 +303,8 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -312,8 +312,8 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDec(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -326,8 +326,8 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDec(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -335,8 +335,8 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -349,8 +349,8 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode *Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
@@ -358,7 +358,7 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t RCON = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
@@ -375,8 +375,8 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Data.Literal.Value);
|
||||
|
||||
auto Res = _PCLMUL(16, Dest, Src, Selector);
|
||||
@@ -388,8 +388,8 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
|
||||
|
||||
OrderedNode *Res = _PCLMUL(DstSize, Src1, Src2, Selector);
|
||||
|
||||
@@ -20,15 +20,15 @@ $end_info$
|
||||
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
FEXCore::X86State::RFLAG_CF_LOC,
|
||||
FEXCore::X86State::RFLAG_PF_LOC,
|
||||
FEXCore::X86State::RFLAG_AF_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_LOC,
|
||||
FEXCore::X86State::RFLAG_SF_LOC,
|
||||
FEXCore::X86State::RFLAG_CF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_PF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_AF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_SF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_TF_LOC,
|
||||
FEXCore::X86State::RFLAG_IF_LOC,
|
||||
FEXCore::X86State::RFLAG_DF_LOC,
|
||||
FEXCore::X86State::RFLAG_OF_LOC,
|
||||
FEXCore::X86State::RFLAG_OF_RAW_LOC,
|
||||
FEXCore::X86State::RFLAG_IOPL_LOC,
|
||||
FEXCore::X86State::RFLAG_NT_LOC,
|
||||
FEXCore::X86State::RFLAG_RF_LOC,
|
||||
@@ -80,9 +80,9 @@ void OpDispatchBuilder::ZeroMultipleFlags(uint32_t FlagsMask) {
|
||||
}
|
||||
|
||||
// PF is stored inverted, so invert it when we zero.
|
||||
if (FlagsMask & (1u << X86State::RFLAG_PF_LOC)) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(_Constant(1));
|
||||
FlagsMask &= ~(1u << X86State::RFLAG_PF_LOC);
|
||||
if (FlagsMask & (1u << X86State::RFLAG_PF_RAW_LOC)) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(_Constant(1));
|
||||
FlagsMask &= ~(1u << X86State::RFLAG_PF_RAW_LOC);
|
||||
}
|
||||
|
||||
// Handle remaining masks.
|
||||
@@ -114,7 +114,7 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
for (size_t i = 0; i < NumFlags; ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
|
||||
if (FlagOffset == FEXCore::X86State::RFLAG_AF_LOC) {
|
||||
if (FlagOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
// AF is in bit 4 architecturally, and we need to store it to bit 4 of our
|
||||
// AF register, with garbage in the other bits. The extract is deferred.
|
||||
// We also defer a XOR with the result bit, which is implemented as XOR
|
||||
@@ -122,9 +122,9 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
// that will be a no-op and we get the right result.
|
||||
//
|
||||
// So we write out the whole flags byte to AF without an extract.
|
||||
static_assert(FEXCore::X86State::RFLAG_AF_LOC == 4);
|
||||
SetRFLAG(Src, FEXCore::X86State::RFLAG_AF_LOC);
|
||||
} else if (FlagOffset == FEXCore::X86State::RFLAG_PF_LOC) {
|
||||
static_assert(FEXCore::X86State::RFLAG_AF_RAW_LOC == 4);
|
||||
SetRFLAG(Src, FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
} else if (FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
// PF is stored parity flipped
|
||||
OrderedNode *Tmp = _Bfe(OpSize::i32Bit, 1, FlagOffset, Src);
|
||||
Tmp = _Xor(OpSize::i32Bit, Tmp, _Constant(1));
|
||||
@@ -143,13 +143,13 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
OrderedNode *Original = _Constant(0);
|
||||
|
||||
// SF/ZF and N/Z are together on both arm64 and x86_64, so we special case that.
|
||||
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_LOC)) &&
|
||||
(FlagsMask & (1 << FEXCore::X86State::RFLAG_ZF_LOC));
|
||||
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_RAW_LOC)) &&
|
||||
(FlagsMask & (1 << FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
|
||||
// Handle CF first, since it's at bit 0 and hence doesn't need shift or OR.
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_LOC)) {
|
||||
static_assert(FEXCore::X86State::RFLAG_CF_LOC == 0);
|
||||
Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_RAW_LOC)) {
|
||||
static_assert(FEXCore::X86State::RFLAG_CF_RAW_LOC == 0);
|
||||
Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FlagOffsets.size(); ++i) {
|
||||
@@ -158,10 +158,10 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if ((GetNZ && (FlagOffset == FEXCore::X86State::RFLAG_SF_LOC ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_ZF_LOC)) ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_CF_LOC ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_PF_LOC) {
|
||||
if ((GetNZ && (FlagOffset == FEXCore::X86State::RFLAG_SF_RAW_LOC ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_ZF_RAW_LOC)) ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_CF_RAW_LOC ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
// Already handled
|
||||
continue;
|
||||
}
|
||||
@@ -169,7 +169,7 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// Note that the Bfi only considers the bottom bit of the flag, the rest of
|
||||
// the byte is allowed to be garbage.
|
||||
OrderedNode *Flag;
|
||||
if (FlagOffset == FEXCore::X86State::RFLAG_AF_LOC)
|
||||
if (FlagOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC)
|
||||
Flag = LoadAF();
|
||||
else
|
||||
Flag = GetRFLAG(FlagOffset);
|
||||
@@ -180,23 +180,23 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// Raw PF value needs to have its bottom bit masked out and inverted. The
|
||||
// naive sequence is and/eor/orlshl. But we can do the inversion implicitly
|
||||
// instead.
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_LOC)) {
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_RAW_LOC)) {
|
||||
// Set every bit except the bottommost.
|
||||
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(), _Constant(~1ull));
|
||||
|
||||
// Rotate the bottom bit to the appropriate location for PF, so we get
|
||||
// something like 111P1111. Then invert that to get 000p0000. Then OR that
|
||||
// into the flags. This is 1 A64 instruction :-)
|
||||
auto RightRotation = 64 - FEXCore::X86State::RFLAG_PF_LOC;
|
||||
auto RightRotation = 64 - FEXCore::X86State::RFLAG_PF_RAW_LOC;
|
||||
Original = _Ornror(OpSize::i64Bit, Original, OnesInvPF, RightRotation);
|
||||
}
|
||||
|
||||
// OR in the SF/ZF flags at the end, allowing the lshr to fold with the OR
|
||||
if (GetNZ) {
|
||||
static_assert(FEXCore::X86State::RFLAG_SF_LOC == (FEXCore::X86State::RFLAG_ZF_LOC + 1));
|
||||
static_assert(FEXCore::X86State::RFLAG_SF_RAW_LOC == (FEXCore::X86State::RFLAG_ZF_RAW_LOC + 1));
|
||||
auto NZCV = GetNZCV();
|
||||
auto NZ = _And(OpSize::i64Bit, NZCV, _Constant(0b11u << 30));
|
||||
Original = _Orlshr(OpSize::i64Bit, Original, NZ, 31 - FEXCore::X86State::RFLAG_SF_LOC);
|
||||
Original = _Orlshr(OpSize::i64Bit, Original, NZ, 31 - FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
}
|
||||
|
||||
// The constant is OR'ed in at the end, to avoid a pointless or xzr, #2.
|
||||
@@ -238,12 +238,12 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode *Res, OrderedNo
|
||||
}
|
||||
|
||||
auto OF = _Bfe(OpSize, 1, SrcSize * 8 - 1, Anded);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(OF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OF);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadPFRaw() {
|
||||
// Read the stored byte. This is the original 8-bit result, it needs parity calculated.
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
// Cast the input to a 32-bit FPR. Logically we only need 8-bit, but that would
|
||||
// generate unwanted an ubfx instruction. VPopcount will ignore the upper bits anyway.
|
||||
@@ -256,10 +256,10 @@ OrderedNode *OpDispatchBuilder::LoadPFRaw() {
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadAF() {
|
||||
// Read the stored byte. This is the XOR of the arguments.
|
||||
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// Read the result, stored as the PF byte for deferred PF calculation.
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
// What's left is to XOR and extract. This is the deferred part.
|
||||
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i32Bit, AFByte, PFByte));
|
||||
@@ -272,11 +272,11 @@ void OpDispatchBuilder::FixupAF() {
|
||||
//
|
||||
// (AF[4] ^ PF[4]) ^ PF[4] = AF[4]
|
||||
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
|
||||
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
OrderedNode *XorRes = _Xor(OpSize::i32Bit, AFByte, PFByte);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(XorRes);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculatePF(OrderedNode *Res, OrderedNode *condition) {
|
||||
@@ -285,12 +285,12 @@ void OpDispatchBuilder::CalculatePF(OrderedNode *Res, OrderedNode *condition) {
|
||||
// we need the existing /encoded/ value rather than the decoded PF value. In particular,
|
||||
// this does not calculate a popcount.
|
||||
if (condition) {
|
||||
auto OldFlag = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
|
||||
auto OldFlag = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
Res = _Select(FEXCore::IR::COND_EQ, condition, _Constant(0), OldFlag, Res);
|
||||
}
|
||||
|
||||
// Calculation is entirely deferred until load, just store the 8-bit result.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
@@ -298,7 +298,7 @@ void OpDispatchBuilder::CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode
|
||||
// there's no sense XOR'ing at all. This affects INC.
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src2), &Const) && (Const & (1u << 4)) == 0) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(Src1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Src1);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -306,7 +306,7 @@ void OpDispatchBuilder::CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit.
|
||||
OrderedNode *XorRes = _Xor(OpSize, Src1, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(XorRes);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
@@ -527,7 +527,7 @@ void OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
auto SelectOpLT = _Select(FEXCore::IR::COND_ULT, Res, Src2, One, Zero);
|
||||
auto SelectOpLE = _Select(FEXCore::IR::COND_ULE, Res, Src2, One, Zero);
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, One, SelectOpLE, SelectOpLT);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(SelectCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
|
||||
}
|
||||
|
||||
// Signed
|
||||
@@ -555,7 +555,7 @@ void OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
auto SelectOpLT = _Select(FEXCore::IR::COND_UGT, Res, Src1, One, Zero);
|
||||
auto SelectOpLE = _Select(FEXCore::IR::COND_UGE, Res, Src1, One, Zero);
|
||||
auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, One, SelectOpLE, SelectOpLT);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(SelectCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(SelectCF);
|
||||
}
|
||||
|
||||
// Signed
|
||||
@@ -570,7 +570,7 @@ void OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
CalculatePF(Res);
|
||||
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// TODO: Could do this path for small sources if we have FEAT_FlagM
|
||||
if (SrcSize >= 4) {
|
||||
@@ -584,7 +584,7 @@ void OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
if (UpdateCF) {
|
||||
// Grab carry bit from unmasked output.
|
||||
auto Bfe = _Bfe(OpSize::i32Bit, 1, SrcSize * 8, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Bfe);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Bfe);
|
||||
}
|
||||
|
||||
CalculateOF(SrcSize, Res, Src1, Src2, true);
|
||||
@@ -592,7 +592,7 @@ void OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
|
||||
// We stomped over CF while calculation flags, restore it.
|
||||
if (!UpdateCF)
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(OldCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
@@ -602,7 +602,7 @@ void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
CalculatePF(Res);
|
||||
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// TODO: Could do this path for small sources if we have FEAT_FlagM
|
||||
if (SrcSize >= 4) {
|
||||
@@ -615,7 +615,7 @@ void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
if (UpdateCF) {
|
||||
// Grab carry bit from unmasked output
|
||||
auto Bfe = _Bfe(OpSize::i32Bit, 1, SrcSize * 8, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Bfe);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Bfe);
|
||||
}
|
||||
|
||||
CalculateOF(SrcSize, Res, Src1, Src2, false);
|
||||
@@ -623,7 +623,7 @@ void OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
|
||||
// We stomped over CF while calculation flags, restore it.
|
||||
if (!UpdateCF)
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(OldCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(OldCF);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High) {
|
||||
@@ -632,8 +632,8 @@ void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
// PF/AF/ZF/SF
|
||||
// Undefined
|
||||
{
|
||||
_InvalidateFlags(1 << X86State::RFLAG_PF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_PF_RAW_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
}
|
||||
|
||||
// CF/OF
|
||||
@@ -643,8 +643,8 @@ void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, Or
|
||||
|
||||
auto SignBit = _Sbfe(OpSize::i64Bit, 1, SrcSize * 8 - 1, Res);
|
||||
|
||||
auto CV = _Constant((1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_OF_LOC)));
|
||||
auto CV = _Constant((1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC)));
|
||||
|
||||
// Set CV accordingly and zero NZ regardless
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, High, SignBit, Zero, CV));
|
||||
@@ -657,8 +657,8 @@ void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
|
||||
// AF/SF/PF/ZF
|
||||
// Undefined
|
||||
{
|
||||
_InvalidateFlags(1 << X86State::RFLAG_PF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_PF_RAW_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
}
|
||||
|
||||
// CF/OF
|
||||
@@ -666,8 +666,8 @@ void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// The result register will be all zero if it can't fit due to how multiplication behaves
|
||||
|
||||
auto CV = _Constant((1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_OF_LOC)));
|
||||
auto CV = _Constant((1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC)));
|
||||
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, High, Zero, Zero, CV));
|
||||
}
|
||||
@@ -676,7 +676,7 @@ void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode *High) {
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
@@ -700,21 +700,21 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *R
|
||||
auto Size = _Constant(SrcSize * 8);
|
||||
auto ShiftAmt = _Sub(OpSize, Size, Src2);
|
||||
auto LastBit = _Bfe(OpSize, 1, 0, _Lshr(OpSize, Src1, ShiftAmt));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(LastBit);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit);
|
||||
}
|
||||
|
||||
CalculatePF(Res, Src2);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// OF
|
||||
{
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
// When Shift > 1 then OF is undefined
|
||||
auto val = _Bfe(OpSize, 1, SrcSize * 8 - 1, _Xor(OpSize, Src1, Res));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(val);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val);
|
||||
}
|
||||
|
||||
// Now select between the two
|
||||
@@ -738,21 +738,21 @@ void OpDispatchBuilder::CalculateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *
|
||||
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, One);
|
||||
const auto CFSize = IR::SizeToOpSize(std::max<uint8_t>(4u, SrcSize));
|
||||
auto LastBit = _Bfe(CFSize, 1, 0, _Lshr(CFSize, Src1, ShiftAmt));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(LastBit);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit);
|
||||
}
|
||||
|
||||
CalculatePF(Res, Src2);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// OF
|
||||
{
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// OF flag is set if a sign change occurred
|
||||
auto val = _Bfe(OpSize, 1, SrcSize * 8 - 1, _Xor(OpSize, Src1, Res));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(val);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val);
|
||||
}
|
||||
|
||||
// Now select between the two
|
||||
@@ -776,14 +776,14 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRight(uint8_t SrcSize, OrderedNo
|
||||
const auto CFSize = IR::SizeToOpSize(std::max<uint32_t>(4u, GetOpSize(Src1)));
|
||||
auto ShiftAmt = _Sub(OpSize::i64Bit, Src2, One);
|
||||
auto LastBit = _Bfe(CFSize, 1, 0, _Lshr(CFSize, Src1, ShiftAmt));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(LastBit);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(LastBit);
|
||||
}
|
||||
|
||||
CalculatePF(Res, Src2);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// Now select between the two
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
|
||||
@@ -805,21 +805,21 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Order
|
||||
if (SrcSizeBits < Shift) {
|
||||
Shift &= (SrcSizeBits - 1);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(OpSize, 1, SrcSizeBits - Shift, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Bfe(OpSize, 1, SrcSizeBits - Shift, Src1));
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// OF
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
if (Shift == 1) {
|
||||
auto Xor = _Xor(OpSize, Res, Src1);
|
||||
auto OF = _Bfe(OpSize, 1, SrcSize * 8 - 1, Xor);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(OF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OF);
|
||||
} else {
|
||||
// Undefined, we choose to zero as part of SetNZ_ZeroCV
|
||||
}
|
||||
@@ -834,14 +834,14 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(IR::SizeToOpSize(std::max<uint32_t>(4u, GetOpSize(Src1))), 1, Shift-1, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Bfe(IR::SizeToOpSize(std::max<uint32_t>(4u, GetOpSize(Src1))), 1, Shift-1, Src1));
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// OF
|
||||
// Only defined when Shift is 1 else undefined. Only is set if the top bit was set to 1 when
|
||||
@@ -853,24 +853,24 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
const auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
// Stash OF before overwriting it
|
||||
auto OldOF = Shift != 1 ? GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC) : NULL;
|
||||
auto OldOF = Shift != 1 ? GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC) : NULL;
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(OpSize, 1, Shift-1, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(_Bfe(OpSize, 1, Shift-1, Src1));
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_LOC);
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// Preserve OF if it won't be written
|
||||
if (Shift != 1) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(OldOF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(OldOF);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -886,7 +886,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Orde
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// Is set to the MSB of the original value
|
||||
if (Shift == 1) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Bfe(OpSize, 1, SrcSize * 8 - 1, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(_Bfe(OpSize, 1, SrcSize * 8 - 1, Src1));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -905,7 +905,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
|
||||
// XOR of Result and Src1
|
||||
if (Shift == 1) {
|
||||
auto val = _Bfe(OpSize, 1, SrcSize * 8 - 1, _Xor(OpSize, Src1, Res));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(val);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(val);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -921,12 +921,12 @@ void OpDispatchBuilder::CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
auto NewCF = _Bfe(OpSize, 1, SizeBits - 1, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
|
||||
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 2, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
|
||||
|
||||
// Now select: if shift == 0, don't update flags
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
|
||||
@@ -949,13 +949,13 @@ void OpDispatchBuilder::CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *
|
||||
//auto Size = _Constant(GetSrcSize(Res) * 8);
|
||||
//auto ShiftAmt = _Sub(OpSize::i64Bit, Size, Src2);
|
||||
auto NewCF = _Bfe(OpSize, 1, 0, Res);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
|
||||
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 1, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
|
||||
|
||||
// Now select: if shift == 0, don't update flags
|
||||
SetNZCV(_Select(FEXCore::IR::COND_EQ, Src2, Zero, OldNZCV, GetNZCV()));
|
||||
@@ -976,7 +976,7 @@ void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, Ord
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
|
||||
}
|
||||
|
||||
// OF
|
||||
@@ -985,7 +985,7 @@ void OpDispatchBuilder::CalculateFlags_RotateRightImmediate(uint8_t SrcSize, Ord
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 2, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1005,7 +1005,7 @@ void OpDispatchBuilder::CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, Orde
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(NewCF);
|
||||
}
|
||||
|
||||
// OF
|
||||
@@ -1016,7 +1016,7 @@ void OpDispatchBuilder::CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, Orde
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, SizeBits - 1, Res), NewCF);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1026,21 +1026,21 @@ void OpDispatchBuilder::CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, O
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(PF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
|
||||
// Zero AF. Note that we set the PF byte to 0/1 above, so PF[4] is 0 so the
|
||||
// XOR with PF will have no effect, so setting the AF byte to zero will indeed
|
||||
// zero AF as intended.
|
||||
uint32_t FlagsMaskToZero =
|
||||
(1U << X86State::RFLAG_AF_LOC) |
|
||||
(1U << X86State::RFLAG_SF_LOC) |
|
||||
(1U << X86State::RFLAG_OF_LOC);
|
||||
(1U << X86State::RFLAG_AF_RAW_LOC) |
|
||||
(1U << X86State::RFLAG_SF_RAW_LOC) |
|
||||
(1U << X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
ZeroMultipleFlags(FlagsMaskToZero);
|
||||
}
|
||||
@@ -1062,14 +1062,14 @@ void OpDispatchBuilder::CalculateFlags_BEXTR(OrderedNode *Src) {
|
||||
ZeroMultipleFlags(FullNZCVMask);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
// ZF
|
||||
auto ZeroOp = _Select(IR::COND_EQ,
|
||||
Src, Zero,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZeroOp);
|
||||
SetRFLAG<X86State::RFLAG_ZF_RAW_LOC>(ZeroOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src) {
|
||||
@@ -1087,14 +1087,14 @@ void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src) {
|
||||
SetNZ_ZeroCV(SrcSize, Src);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
|
||||
// CF
|
||||
{
|
||||
auto CFOp = _Select(IR::COND_NEQ, Src, Zero, One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1105,17 +1105,17 @@ void OpDispatchBuilder::CalculateFlags_BLSMSK(OrderedNode *Src) {
|
||||
auto One = _Constant(1);
|
||||
|
||||
uint32_t FlagsMaskToZero =
|
||||
(1U << X86State::RFLAG_ZF_LOC) |
|
||||
(1U << X86State::RFLAG_OF_LOC);
|
||||
(1U << X86State::RFLAG_ZF_RAW_LOC) |
|
||||
(1U << X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
ZeroMultipleFlags(FlagsMaskToZero);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
auto CFOp = _Select(IR::COND_NEQ, Src, Zero, One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
@@ -1126,13 +1126,13 @@ void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode *Result
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
// CF
|
||||
{
|
||||
auto CFOp = _Select(IR::COND_NEQ, Src, Zero, One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1146,12 +1146,12 @@ void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode *Src) {
|
||||
// Set flags
|
||||
uint32_t FlagsMaskToZero =
|
||||
FullNZCVMask |
|
||||
(1U << X86State::RFLAG_AF_LOC) |
|
||||
(1U << X86State::RFLAG_PF_LOC);
|
||||
(1U << X86State::RFLAG_AF_RAW_LOC) |
|
||||
(1U << X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
ZeroMultipleFlags(FlagsMaskToZero);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(ZFResult);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
@@ -1162,18 +1162,18 @@ void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result
|
||||
auto One = _Constant(1);
|
||||
|
||||
// OF cleared
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_OF_RAW_LOC>(Zero);
|
||||
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_LOC));
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) |
|
||||
(1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
// ZF
|
||||
{
|
||||
auto ZFOp = _Select(IR::COND_EQ,
|
||||
Result, Zero,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZFOp);
|
||||
SetRFLAG<X86State::RFLAG_ZF_RAW_LOC>(ZFOp);
|
||||
}
|
||||
|
||||
// CF
|
||||
@@ -1181,7 +1181,7 @@ void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result
|
||||
auto CFOp = _Select(IR::COND_UGT,
|
||||
Src, Bounds,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1195,8 +1195,8 @@ void OpDispatchBuilder::CalculateFlags_TZCNT(OrderedNode *Src) {
|
||||
_Constant(1), Zero);
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Bfe(OpSize::i32Bit, 1, 0, Src));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Bfe(OpSize::i32Bit, 1, 0, Src));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src) {
|
||||
@@ -1211,8 +1211,8 @@ void OpDispatchBuilder::CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src)
|
||||
_Constant(1), Zero);
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Bfe(OpSize, 1, SrcSize * 8 - 1, Src));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Bfe(OpSize, 1, SrcSize * 8 - 1, Src));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BITSELECT(OrderedNode *Src) {
|
||||
@@ -1227,7 +1227,7 @@ void OpDispatchBuilder::CalculateFlags_BITSELECT(OrderedNode *Src) {
|
||||
Src, ZeroConst,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFSelectOp);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(ZFSelectOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RDRAND(OrderedNode *Src) {
|
||||
@@ -1236,12 +1236,12 @@ void OpDispatchBuilder::CalculateFlags_RDRAND(OrderedNode *Src) {
|
||||
|
||||
uint32_t FlagsMaskToZero =
|
||||
FullNZCVMask |
|
||||
(1U << X86State::RFLAG_AF_LOC) |
|
||||
(1U << X86State::RFLAG_PF_LOC);
|
||||
(1U << X86State::RFLAG_AF_RAW_LOC) |
|
||||
(1U << X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
ZeroMultipleFlags(FlagsMaskToZero);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Src);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
}
|
||||
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
@@ -139,7 +139,7 @@ void OpDispatchBuilder::FLD(OpcodeArgs) {
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Read from memory
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
}
|
||||
else {
|
||||
// Implicit arg
|
||||
@@ -178,7 +178,7 @@ void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
// Read from memory
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags, -1);
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode *converted = _F80BCDLoad(data);
|
||||
_StoreContextIndexed(converted, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -238,7 +238,7 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
size_t read_width = GetSrcSize(Op);
|
||||
|
||||
// Read from memory
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
|
||||
auto zero = _Constant(0);
|
||||
|
||||
@@ -334,11 +334,11 @@ void OpDispatchBuilder::FADD(OpcodeArgs) {
|
||||
// Memory arg
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -395,11 +395,11 @@ void OpDispatchBuilder::FMUL(OpcodeArgs) {
|
||||
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -458,11 +458,11 @@ void OpDispatchBuilder::FDIV(OpcodeArgs) {
|
||||
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -543,11 +543,11 @@ void OpDispatchBuilder::FSUB(OpcodeArgs) {
|
||||
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -724,11 +724,11 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
// Memory arg
|
||||
if constexpr (width == 16 || width == 32 || width == 64) {
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, width / 8);
|
||||
}
|
||||
else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, width / 8);
|
||||
}
|
||||
}
|
||||
@@ -763,13 +763,13 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(PF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
}
|
||||
|
||||
if constexpr (poptwice) {
|
||||
@@ -1014,7 +1014,7 @@ void OpDispatchBuilder::X87ATAN(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
@@ -1052,7 +1052,7 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
{
|
||||
@@ -1099,7 +1099,7 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
@@ -1110,7 +1110,7 @@ void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87LDSW(OpcodeArgs) {
|
||||
OrderedNode *NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
ReconstructX87StateFromFSW(NewFSW);
|
||||
}
|
||||
|
||||
@@ -1139,7 +1139,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
OrderedNode *Top = GetX87Top();
|
||||
@@ -1213,7 +1213,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
|
||||
@@ -66,7 +66,7 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
@@ -89,7 +89,7 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
//ignore the rounding precision, we're always 64-bit in F64.
|
||||
//extract rounding mode
|
||||
OrderedNode *roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
@@ -112,7 +112,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Read from memory
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
if constexpr (width == 32) {
|
||||
converted = _Float_FToF(8, 4, data);
|
||||
@@ -153,7 +153,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
// Read from memory
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags, -1);
|
||||
OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode *converted = _F80BCDLoad(data);
|
||||
converted = _F80CVT(8, converted);
|
||||
_StoreContextIndexed(converted, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -210,7 +210,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
|
||||
size_t read_width = GetSrcSize(Op);
|
||||
// Read from memory
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags, -1);
|
||||
auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags);
|
||||
if(read_width == 2) {
|
||||
data = _Sbfe(OpSize::i64Bit, read_width * 8, 0, data);
|
||||
}
|
||||
@@ -292,16 +292,16 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -353,16 +353,16 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -417,16 +417,16 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -503,16 +503,16 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -661,16 +661,16 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
if constexpr (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
if(width == 16) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(8, width == 64 ? 8 : 4, arg);
|
||||
} else if constexpr (width == 32) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(8, 4, arg);
|
||||
} else if constexpr (width == 64) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
// Implicit arg
|
||||
@@ -704,13 +704,13 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(HostFlag_ZF);
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(PF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
}
|
||||
|
||||
if constexpr (poptwice) {
|
||||
@@ -921,7 +921,7 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
OrderedNode *Top = GetX87Top();
|
||||
@@ -999,7 +999,7 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
|
||||
@@ -79,7 +79,7 @@ namespace FEXCore {
|
||||
const char *Name;
|
||||
};
|
||||
|
||||
static thread_local FEXCore::Core::InternalThreadState *Thread;
|
||||
static thread_local FEXCore::Core::InternalThreadState *Thread = nullptr;
|
||||
|
||||
|
||||
struct ExportEntry { uint8_t *sha256; ThunkedFunction* Fn; };
|
||||
@@ -171,8 +171,18 @@ namespace FEXCore {
|
||||
Set arg0/1 to arg regs, use CTX::HandleCallback to handle the callback
|
||||
*/
|
||||
static void CallCallback(void *callback, void *arg0, void* arg1) {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = (uintptr_t)arg1;
|
||||
if (!Thread) {
|
||||
ERROR_AND_DIE_FMT("Thunked library attempted to invoke guest callback asynchronously");
|
||||
}
|
||||
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = (uintptr_t)arg1;
|
||||
} else {
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RCX] = (uintptr_t)arg0;
|
||||
Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDX] = (uintptr_t)arg1;
|
||||
}
|
||||
|
||||
Thread->CTX->HandleCallback(Thread, (uintptr_t)callback);
|
||||
}
|
||||
@@ -220,7 +230,7 @@ namespace FEXCore {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, gregs[X86State::REG_R11]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
}
|
||||
else {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, mm[0][0]), IR::GPRClass, IR::GPRFixedClass, GPRSize);
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(8, 8, emit->_Constant(Entrypoint)), offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
}, CTX->ThunkHandler.get(), (void*)args->target_addr);
|
||||
|
||||
@@ -156,35 +156,43 @@
|
||||
"MemOffsetType": "MemOffsetType",
|
||||
"BreakDefinition": "BreakDefinition",
|
||||
"RoundType": "RoundType",
|
||||
"FloatCompareOp": "FloatCompareOp",
|
||||
"NamedVectorConstant": "FEXCore::IR::NamedVectorConstant",
|
||||
"IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant"
|
||||
"IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant",
|
||||
"ShiftType": "FEXCore::IR::ShiftType"
|
||||
},
|
||||
"Ops": {
|
||||
"Misc": {
|
||||
"Dummy": {
|
||||
"HasSideEffects": true,
|
||||
"SwitchGen": false
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions": {
|
||||
"SwitchGen": false
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"CodeBlock SSA:$Begin, SSA:$Last": {
|
||||
"SwitchGen": false,
|
||||
"RAOverride": "0"
|
||||
"RAOverride": "0",
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"BeginBlock SSA:$BlockHeader": {
|
||||
"HasSideEffects": true,
|
||||
"SwitchGen": false,
|
||||
"RAOverride": "0"
|
||||
"RAOverride": "0",
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"InvalidateFlags u64:$Flags": {
|
||||
"HasSideEffects": true
|
||||
"HasSideEffects": true,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
|
||||
"EndBlock SSA:$BlockHeader": {
|
||||
"HasSideEffects": true,
|
||||
"SwitchGen": false,
|
||||
"RAOverride": "0"
|
||||
"RAOverride": "0",
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
|
||||
"GuestOpcode u32:$GuestEntryOffset": {
|
||||
@@ -195,6 +203,7 @@
|
||||
"GPR = ValidateCode u64:$CodeOriginalLow, u64:$CodeOriginalhigh, i64:$Offset, u8:$CodeLength": {
|
||||
"HasSideEffects": true,
|
||||
"HasDest": true,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
@@ -240,6 +249,7 @@
|
||||
"The second GPR pair element is a bool if the number is valid",
|
||||
"RNG hardware is allowed to fail early and return. Software must always check this"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
},
|
||||
@@ -337,7 +347,8 @@
|
||||
"Desc": ["Loads a value from the static-ra context with offset",
|
||||
"Dest = Ctx[Offset]"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true
|
||||
},
|
||||
|
||||
"StoreRegister SSA:$Value, i1:$IsPrewrite, u32:$Offset, RegisterClass:$Class, RegisterClass:$StaticClass, u8:#Size": {
|
||||
@@ -348,6 +359,7 @@
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true,
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class"
|
||||
]
|
||||
@@ -472,7 +484,8 @@
|
||||
"SSA = LoadMemTSO RegisterClass:$Class, u8:#Size, GPR:$Addr, GPR:$Offset, u8:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a x86 TSO compatible load from memory. Offset must be Invalid()."
|
||||
],
|
||||
"DestSize": "Size"
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true
|
||||
},
|
||||
|
||||
"StoreMemTSO RegisterClass:$Class, u8:#Size, SSA:$Value, GPR:$Addr, GPR:$Offset, u8:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
@@ -480,6 +493,7 @@
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true,
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class"
|
||||
]
|
||||
@@ -488,6 +502,7 @@
|
||||
"FPR = VLoadVectorMasked u8:#RegisterSize, u8:#ElementSize, FPR:$Mask, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a masked load similar to VPMASKMOV/VMASKMOV where the upper bit of each element",
|
||||
"determines whether or not that element will be loaded from memory"],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -495,6 +510,7 @@
|
||||
"Desc": ["Does a masked store similar to VPMASKMOV/VMASKMOV where the upper bit of each element",
|
||||
"determines whether or not that element will be stored to memory"],
|
||||
"HasSideEffects": true,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -586,6 +602,7 @@
|
||||
],
|
||||
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -600,6 +617,7 @@
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"NumElements": "2",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
|
||||
@@ -607,7 +625,8 @@
|
||||
},
|
||||
"AtomicAdd OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer add"
|
||||
"Desc": ["Atomic integer add",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -616,7 +635,8 @@
|
||||
},
|
||||
"AtomicSub OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer sub"
|
||||
"Desc": ["Atomic integer sub",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -625,7 +645,18 @@
|
||||
},
|
||||
"AtomicAnd OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer and"
|
||||
"Desc": ["Atomic integer and",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"AtomicCLR OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer binary clear",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -634,7 +665,8 @@
|
||||
},
|
||||
"AtomicOr OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer or"
|
||||
"Desc": ["Atomic integer or",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -643,7 +675,18 @@
|
||||
},
|
||||
"AtomicXor OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer xor"
|
||||
"Desc": ["Atomic integer xor",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"AtomicNeg OpSize:#Size, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer two's complement negate",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -663,7 +706,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and add",
|
||||
"Atomically fetches %Addr and adds %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -674,7 +718,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and sub",
|
||||
"Atomically fetches %Addr and subtracts %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -685,7 +730,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and binary and",
|
||||
"Atomically fetches %Addr and binary ands %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -698,7 +744,8 @@
|
||||
"Atomically fetches %Addr and binary clears %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"Matches ARM ldclral semantics",
|
||||
"eg: Dest[Addr] &= ~Value"
|
||||
"eg: Dest[Addr] &= ~Value",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -709,7 +756,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and binary or",
|
||||
"Atomically fetches %Addr and binary ors %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -720,7 +768,8 @@
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and binary exclusive or",
|
||||
"Atomically fetches %Addr and binary exclusive ors %value to the memory location",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -730,7 +779,8 @@
|
||||
"GPR = AtomicFetchNeg OpSize:#Size, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer fetch and two's complement negate",
|
||||
"Dest is the value prior to operating on the value in memory"
|
||||
"Dest is the value prior to operating on the value in memory",
|
||||
"IR layout must match NonFetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -742,6 +792,7 @@
|
||||
"Desc": ["Set Telemetry value if the passed in 32-bit value isn't zero.",
|
||||
"Only useful for 32-bit applications."
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "8"
|
||||
}
|
||||
},
|
||||
@@ -819,6 +870,7 @@
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -848,6 +900,7 @@
|
||||
"In the case of zero returns ~0U"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -903,6 +956,7 @@
|
||||
"GPR = AddNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs"],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -910,6 +964,7 @@
|
||||
"GPR = AdcNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -917,6 +972,7 @@
|
||||
"GPR = SbbNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -930,11 +986,22 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = SubShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
|
||||
"Desc": [ "Integer Sub with shifted register",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"_Shift != ShiftType::ROR"
|
||||
]
|
||||
},
|
||||
"GPR = SubNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, u8:$InvertCarry": {
|
||||
"Desc": ["Return NZCV for the difference of two GPRs. ",
|
||||
"If InvertCarry is nonzero, carry flag uses x86 definition, inverted from arm64.",
|
||||
""],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -996,6 +1063,7 @@
|
||||
},
|
||||
"GPR = TestNZ u8:$Size, GPR:$Src1": {
|
||||
"Desc": ["Return NZCV for a GPR, setting N and Z accordingly and zeroing C and V"],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "4"
|
||||
},
|
||||
"GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
@@ -1147,6 +1215,7 @@
|
||||
"Dest = Cmp1 <Cond> Cmp2 ? TrueVal : FalseVal"
|
||||
],
|
||||
"DestSize": "ResultSize",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_CompareSize == FEXCore::IR::OpSize::i32Bit || _CompareSize == FEXCore::IR::OpSize::i64Bit || _CompareSize == FEXCore::IR::OpSize::i128Bit",
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit",
|
||||
@@ -1255,9 +1324,164 @@
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "4"
|
||||
}
|
||||
},
|
||||
"VectorScalar": {
|
||||
"FPR = VFAddScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'add' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFSubScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'sub' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMulScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'mul' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFDivScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'div' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMinScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'min' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
|
||||
"Additionally matches x86 zero and NaN semantics",
|
||||
"If both source operands are zero, return the second operand (in the case of negative and positive zero)",
|
||||
"If either source operand is NaN then return the second operand."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"ImplicitFlagClobber": true
|
||||
},
|
||||
"FPR = VFMaxScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'max' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
|
||||
"Additionally matches x86 zero and NaN semantics",
|
||||
"If both source operands are zero, return the second operand (in the case of negative and positive zero)",
|
||||
"If either source operand is NaN then return the second operand."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"ImplicitFlagClobber": true
|
||||
},
|
||||
"FPR = VFSqrtScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'sqrt' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFRSqrtScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'rsqrt' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFRecpScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'recip' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFToFScalarInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cvt' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
},
|
||||
"FPR = VSToFVectorInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i8:$HasTwoElements, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a Vector 'scvt' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
|
||||
"HasTwoElements is slightly different than most of these scalar operations.",
|
||||
"Handles the edge case of cvtpi2ps xmm0, mm0 which is two elements in the lower 64-bits"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
},
|
||||
"FPR = VSToFGPRInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector, GPR:$Src, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cvt' between Vector1 and GPR.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
},
|
||||
"FPR = VFToIScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, RoundType:$Round, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar round float to integral on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Rounding mode determined by argument",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFCMPScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FloatCompareOp:$Op, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cmp' between Vector1 and Vecto2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Compare op determined by argument",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
"FPR = VMov u8:#RegisterSize, FPR:$Source": {
|
||||
"Desc" : ["Copy vector register",
|
||||
@@ -1274,7 +1498,7 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"FPR = VectorImm u8:#RegisterSize, u8:#ElementSize, u8:$Immediate": {
|
||||
"FPR = VectorImm u8:#RegisterSize, u8:#ElementSize, u8:$Immediate, u8:$ShiftAmount{0}": {
|
||||
"Desc": ["Generates a vector with each element containg the immediate zexted"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -1357,6 +1581,7 @@
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VCMPEQZ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1365,6 +1590,7 @@
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
|
||||
"Compares the vector against zero"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1373,6 +1599,7 @@
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
|
||||
"Compares the vector against zero"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1600,6 +1827,13 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFAddV u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Does a horizontal float vector add of elements across the source vector",
|
||||
"Result is a zero extended scalar"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFSub u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1614,18 +1848,16 @@
|
||||
},
|
||||
|
||||
"FPR = VFMin u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMax u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUMul u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSMul u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"FPR = VMul u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1732,7 +1964,8 @@
|
||||
|
||||
"FPR = VCMPEQ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
"NumElements": "RegisterSize / ElementSize",
|
||||
"ImplicitFlagClobber": true
|
||||
},
|
||||
|
||||
"FPR = VCMPGT u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
@@ -1740,6 +1973,7 @@
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero"
|
||||
],
|
||||
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1790,6 +2024,15 @@
|
||||
],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VTBX1 u8:#RegisterSize, FPR:$VectorSrcDst, FPR:$VectorTable, FPR:$VectorIndices": {
|
||||
"Desc": ["Does a vector table lookup from one register in to the destination",
|
||||
"Lookup is byte sized per byte element.",
|
||||
"Any index larger than what the registers provide will result in not modifying that element",
|
||||
"Table is always treated as a 128bit register",
|
||||
"Indices matches destination size. Either 64bit or 128bit"
|
||||
],
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VBSL u8:#RegisterSize, FPR:$VectorMask, FPR:$VectorTrue, FPR:$VectorFalse": {
|
||||
"Desc": ["Does a vector bitwise select.",
|
||||
"If the bit in the field is 1 then the corresponding bit is pulled from VectorTrue",
|
||||
@@ -1807,7 +2050,8 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"DestSize": "4"
|
||||
"DestSize": "4",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = VPCMPISTRX FPR:$LHS, FPR:$RHS, u8:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPISTRI/PCMPISTRM instruction",
|
||||
@@ -1818,7 +2062,8 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"DestSize": "4"
|
||||
"DestSize": "4",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = VFCADD u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, u16:$Rotate": {
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -1925,111 +2170,143 @@
|
||||
}
|
||||
},
|
||||
"F64": {
|
||||
"FPR = F64ATAN FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64FPREM FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64FPREM1 FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64SCALE FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64F2XM1 FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64FYL2X FPR:$Src, FPR:$Src2": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64TAN FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64SIN FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
},
|
||||
"FPR = F64COS FPR:$Src": {
|
||||
"DestSize": "8"
|
||||
}
|
||||
"FPR = F64ATAN FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64FPREM FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64FPREM1 FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64SCALE FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64F2XM1 FPR:$Src": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64FYL2X FPR:$Src, FPR:$Src2": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64TAN FPR:$Src": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64SIN FPR:$Src": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F64COS FPR:$Src": {
|
||||
"DestSize": "8",
|
||||
"JITDispatch": false
|
||||
}
|
||||
},
|
||||
"F80": {
|
||||
"FPR = F80Add FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80Sub FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80Mul FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
|
||||
"FPR = F80Div FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80ATAN FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80FPREM FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80FPREM1 FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80SCALE FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80CVT u8:#Size, FPR:$X80Src": {
|
||||
"DestSize": "Size"
|
||||
"DestSize": "Size",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = F80CVTInt u8:#Size, FPR:$X80Src, i1:$Truncate": {
|
||||
"DestSize": "Size"
|
||||
"DestSize": "Size",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80CVTTo FPR:$X80Src, u8:$SrcSize": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80CVTToInt GPR:$Src, u8:$SrcSize": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80Round FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80F2XM1 FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80TAN FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80SIN FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80COS FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80SQRT FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80XTRACT_EXP FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80XTRACT_SIG FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = F80Cmp FPR:$X80Src1, FPR:$X80Src2, u32:$Flags": {
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"DestSize": "4"
|
||||
"DestSize": "4",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80BCDLoad FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80BCDStore FPR:$X80Src": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
},
|
||||
|
||||
"FPR = F80FYL2X FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
"DestSize": "16",
|
||||
"JITDispatch": false
|
||||
}
|
||||
},
|
||||
"Backend": {
|
||||
|
||||
@@ -238,6 +238,18 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::FloatCompareOp Arg) {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: *out << "FEQ"; break;
|
||||
case FloatCompareOp::LT: *out << "FLT"; break;
|
||||
case FloatCompareOp::LE: *out << "FLE"; break;
|
||||
case FloatCompareOp::UNO: *out << "UNO"; break;
|
||||
case FloatCompareOp::NEQ: *out << "NEQ"; break;
|
||||
case FloatCompareOp::ORD: *out << "ORD"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::BreakDefinition Arg) {
|
||||
*out << "{" << Arg.ErrorRegister << ".";
|
||||
*out << static_cast<uint32_t>(Arg.Signal) << ".";
|
||||
@@ -245,6 +257,16 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
*out << static_cast<uint32_t>(Arg.si_code) << "}";
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::ShiftType Arg) {
|
||||
switch (Arg) {
|
||||
case ShiftType::LSL: *out << "LSL"; break;
|
||||
case ShiftType::LSR: *out << "LSR"; break;
|
||||
case ShiftType::ASR: *out << "ASR"; break;
|
||||
case ShiftType::ROR: *out << "ROR"; break;
|
||||
default: *out << "<Unknown Shift Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
|
||||
@@ -6,12 +6,11 @@ tags: ir|parser
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/StringUtils.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
@@ -277,7 +277,8 @@ void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& Cur
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto BlockOp = BlockIROp->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
for (auto [UnaryOpNode, UnaryOpHdr] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)) {
|
||||
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)
|
||||
&& !ImplicitFlagClobber(UnaryOpHdr->Op)) {
|
||||
// could be moved
|
||||
auto SelectOpNode = IREmit->UnwrapNode(UnaryOpHdr->Args[0]);
|
||||
auto SelectOpHdr = IREmit->GetOpHeader(UnaryOpHdr->Args[0]);
|
||||
@@ -419,7 +420,6 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_AND: {
|
||||
// if AND's arguments are imms, they are masking
|
||||
for (int i = 0; i < IR::GetArgs(IROp->Op); i++) {
|
||||
@@ -494,27 +494,6 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VFADD:
|
||||
case OP_VFSUB:
|
||||
case OP_VFMUL:
|
||||
case OP_VFDIV:
|
||||
case OP_FCMP: {
|
||||
auto flopSize = IROp->Size;
|
||||
for (int i = 0; i < IR::GetArgs(IROp->Op); i++) {
|
||||
auto argHeader = IREmit->GetOpHeader(IROp->Args[i]);
|
||||
|
||||
if (argHeader->Op == OP_VMOV) {
|
||||
auto source = argHeader->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
if (sourceHeader->Size >= flopSize) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(source));
|
||||
//printf("VMOV bypassed\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_VMOV: {
|
||||
// elim from load mem
|
||||
auto source = IROp->Args[0];
|
||||
@@ -668,13 +647,31 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
bool IsConstant1 = IREmit->IsValueConstant(Op->Header.Args[0], &Constant1);
|
||||
bool IsConstant2 = IREmit->IsValueConstant(Op->Header.Args[1], &Constant2);
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
if (IsConstant1 && IsConstant2) {
|
||||
uint64_t NewConstant = (Constant1 + Constant2) & getMask(Op) ;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
else if (IsConstant2 && !IsImmAddSub(Constant2) && IsImmAddSub(-Constant2)) {
|
||||
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
|
||||
// This means we can convert the operation in to a subtract.
|
||||
// Change the IR operation itself.
|
||||
IROp->Op = OP_SUB;
|
||||
// Set the write cursor to just before this operation.
|
||||
auto CodeIter = CurrentIR.at(CodeNode);
|
||||
--CodeIter;
|
||||
IREmit->SetWriteCursor(std::get<0>(*CodeIter));
|
||||
|
||||
// Negate the constant.
|
||||
auto NegConstant = IREmit->_Constant(-Constant2);
|
||||
|
||||
// Replace the second source with the negated constant.
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Src2_Index, NegConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUB: {
|
||||
@@ -690,6 +687,21 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUBSHIFT: {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
|
||||
uint64_t Constant1, Constant2;
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(IROp->Args[1], &Constant2) &&
|
||||
Op->Shift == IR::ShiftType::LSL) {
|
||||
// Optimize the LSL case when we know both sources are constant.
|
||||
// This is a pattern that shows up with direction flag calculations if DF was set just before the operation.
|
||||
uint64_t NewConstant = (Constant1 - (Constant2 << Op->ShiftAmount)) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_AND: {
|
||||
auto Op = IROp->CW<IR::IROp_And>();
|
||||
uint64_t Constant1{};
|
||||
|
||||
@@ -26,7 +26,7 @@ private:
|
||||
bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DCE");
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
int NumRemoved = 0;
|
||||
bool Changed = false;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
|
||||
@@ -41,28 +41,71 @@ bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
auto [CodeNode, IROp] = CodeLast();
|
||||
|
||||
bool HasSideEffects = IR::HasSideEffects(IROp->Op);
|
||||
if (IROp->Op == OP_SYSCALL ||
|
||||
IROp->Op == OP_INLINESYSCALL) {
|
||||
FEXCore::IR::SyscallFlags Flags{};
|
||||
if (IROp->Op == OP_SYSCALL) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
else {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) == FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) {
|
||||
HasSideEffects = false;
|
||||
switch (IROp->Op) {
|
||||
case OP_SYSCALL:
|
||||
case OP_INLINESYSCALL: {
|
||||
FEXCore::IR::SyscallFlags Flags{};
|
||||
if (IROp->Op == OP_SYSCALL) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
else {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) == FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) {
|
||||
HasSideEffects = false;
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_ATOMICFETCHADD:
|
||||
case OP_ATOMICFETCHSUB:
|
||||
case OP_ATOMICFETCHAND:
|
||||
case OP_ATOMICFETCHCLR:
|
||||
case OP_ATOMICFETCHOR:
|
||||
case OP_ATOMICFETCHXOR:
|
||||
case OP_ATOMICFETCHNEG: {
|
||||
// If the result of the atomic fetch is completely unused, convert it to a non-fetching atomic operation.
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
switch (IROp->Op) {
|
||||
case OP_ATOMICFETCHADD:
|
||||
IROp->Op = OP_ATOMICADD;
|
||||
break;
|
||||
case OP_ATOMICFETCHSUB:
|
||||
IROp->Op = OP_ATOMICSUB;
|
||||
break;
|
||||
case OP_ATOMICFETCHAND:
|
||||
IROp->Op = OP_ATOMICAND;
|
||||
break;
|
||||
case OP_ATOMICFETCHCLR:
|
||||
IROp->Op = OP_ATOMICCLR;
|
||||
break;
|
||||
case OP_ATOMICFETCHOR:
|
||||
IROp->Op = OP_ATOMICOR;
|
||||
break;
|
||||
case OP_ATOMICFETCHXOR:
|
||||
IROp->Op = OP_ATOMICXOR;
|
||||
break;
|
||||
case OP_ATOMICFETCHNEG:
|
||||
IROp->Op = OP_ATOMICNEG;
|
||||
break;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
|
||||
// Skip over anything that has side effects
|
||||
// Use count tracking can't safely remove anything with side effects
|
||||
if (!HasSideEffects) {
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
NumRemoved++;
|
||||
Changed = true;
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
@@ -74,7 +117,7 @@ bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
return NumRemoved != 0;
|
||||
return Changed;
|
||||
}
|
||||
|
||||
void DeadCodeElimination::markUsed(OrderedNodeWrapper *CodeOp, IROp_Header *IROp) {
|
||||
|
||||
@@ -24,24 +24,48 @@ static bool LoadFileImpl(T &Data, const fextl::string &Filepath, size_t FixedSiz
|
||||
size_t FileSize{};
|
||||
if (FixedSize == 0) {
|
||||
struct stat buf;
|
||||
if (fstat(FD, &buf) != 0) {
|
||||
close(FD);
|
||||
return false;
|
||||
if (fstat(FD, &buf) == 0) {
|
||||
FileSize = buf.st_size;
|
||||
}
|
||||
|
||||
FileSize = buf.st_size;
|
||||
}
|
||||
else {
|
||||
FileSize = FixedSize;
|
||||
}
|
||||
|
||||
ssize_t Read = -1;
|
||||
if (FileSize > 0) {
|
||||
bool LoadedFile{};
|
||||
if (FileSize) {
|
||||
// File size is known upfront
|
||||
Data.resize(FileSize);
|
||||
Read = pread(FD, &Data.at(0), FileSize, 0);
|
||||
|
||||
LoadedFile = Read == FileSize;
|
||||
}
|
||||
else {
|
||||
// The file is either empty or its size is unknown (e.g. procfs data).
|
||||
// Try reading in chunks instead
|
||||
ssize_t CurrentOffset = 0;
|
||||
constexpr size_t READ_SIZE = 4096;
|
||||
Data.resize(READ_SIZE);
|
||||
|
||||
while ((Read = pread(FD, &Data.at(CurrentOffset), READ_SIZE, CurrentOffset)) == READ_SIZE) {
|
||||
CurrentOffset += Read;
|
||||
Data.resize(CurrentOffset + Read);
|
||||
}
|
||||
|
||||
if (Read == -1) {
|
||||
Data.clear();
|
||||
close(FD);
|
||||
return false;
|
||||
}
|
||||
|
||||
// Final resize to ensure there is no garbage data past the end.
|
||||
Data.resize(CurrentOffset + Read);
|
||||
|
||||
LoadedFile = true;
|
||||
}
|
||||
close(FD);
|
||||
return Read == FileSize;
|
||||
return LoadedFile;
|
||||
}
|
||||
|
||||
ssize_t LoadFileToBuffer(const fextl::string &Filepath, std::span<char> Buffer) {
|
||||
|
||||
@@ -118,7 +118,7 @@ namespace Handler {
|
||||
}
|
||||
Begin = End + 1;
|
||||
End = View.find_first_of(',', Begin);
|
||||
Option = View.substr(Begin, End);
|
||||
Option = View.substr(Begin, End - Begin);
|
||||
}
|
||||
|
||||
return fextl::fmt::format("{}", EnumMask);
|
||||
|
||||
@@ -136,7 +136,7 @@ namespace CPU {
|
||||
[[nodiscard]] virtual CompiledCode CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) = 0;
|
||||
FEXCore::IR::RegisterAllocationData *RAData) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
|
||||
@@ -87,6 +87,11 @@ namespace FEXCore::Context {
|
||||
void *VDSO_kernel_rt_sigreturn;
|
||||
};
|
||||
|
||||
struct ThreadsState {
|
||||
FEXCore::Core::InternalThreadState* ParentThread;
|
||||
fextl::vector<FEXCore::Core::InternalThreadState*>* Threads;
|
||||
};
|
||||
|
||||
using CodeRangeInvalidationFn = std::function<void(uint64_t start, uint64_t Length)>;
|
||||
|
||||
using CustomCPUFactoryType = std::function<fextl::unique_ptr<CPU::CPUBackend>(Context*, Core::InternalThreadState *Thread)>;
|
||||
@@ -141,6 +146,18 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void Pause() = 0;
|
||||
|
||||
/**
|
||||
* @brief Waits for all threads to be idle.
|
||||
*
|
||||
* Idling can happen when the process is shutting down or the debugger has asked for all threads to pause.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void WaitForIdle() = 0;
|
||||
|
||||
/**
|
||||
* @brief When resuming from a paused state, waits for all threads to start executing before returning.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void WaitForThreadsToRun() = 0;
|
||||
|
||||
/**
|
||||
* @brief Starts (or continues) the CPU core
|
||||
*
|
||||
@@ -182,6 +199,7 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void ExecuteThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void CompileRIPCount(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, uint64_t MaxInst) = 0;
|
||||
|
||||
/**
|
||||
* @brief Gets the program exit status
|
||||
@@ -317,6 +335,14 @@ namespace FEXCore::Context {
|
||||
*
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void EnableExitOnHLT() = 0;
|
||||
|
||||
/**
|
||||
* @brief Gets the thread data for FEX's internal tracked threads.
|
||||
*
|
||||
* @return struct containing all the thread information.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual ThreadsState GetThreads() = 0;
|
||||
|
||||
private:
|
||||
};
|
||||
|
||||
|
||||
@@ -36,9 +36,10 @@ class HostFeatures final {
|
||||
bool SupportsFCMA{};
|
||||
bool SupportsFlagM{};
|
||||
bool SupportsFlagM2{};
|
||||
bool SupportsRPRES{};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsFlushInputsToZero{};
|
||||
bool SupportsAFP{};
|
||||
bool SupportsFloatExceptions{};
|
||||
};
|
||||
}
|
||||
@@ -56,24 +56,24 @@ enum X86Reg : uint32_t {
|
||||
* @name RFLAG register bit locations
|
||||
* @{ */
|
||||
enum X86RegLocation : uint32_t {
|
||||
RFLAG_CF_LOC = 0,
|
||||
RFLAG_RESERVED_LOC = 1, // Reserved Bit, Read-as-1
|
||||
RFLAG_PF_LOC = 2,
|
||||
RFLAG_AF_LOC = 4,
|
||||
RFLAG_ZF_LOC = 6,
|
||||
RFLAG_SF_LOC = 7,
|
||||
RFLAG_TF_LOC = 8,
|
||||
RFLAG_IF_LOC = 9,
|
||||
RFLAG_DF_LOC = 10,
|
||||
RFLAG_OF_LOC = 11,
|
||||
RFLAG_IOPL_LOC = 12,
|
||||
RFLAG_NT_LOC = 14,
|
||||
RFLAG_RF_LOC = 16,
|
||||
RFLAG_VM_LOC = 17,
|
||||
RFLAG_AC_LOC = 18,
|
||||
RFLAG_VIF_LOC = 19,
|
||||
RFLAG_VIP_LOC = 20,
|
||||
RFLAG_ID_LOC = 21,
|
||||
RFLAG_CF_RAW_LOC = 0, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_RESERVED_LOC = 1, // Reserved Bit, Read-as-1
|
||||
RFLAG_PF_RAW_LOC = 2, // Contains multiple bits, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_AF_RAW_LOC = 4, // Contains multiple bits, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_ZF_RAW_LOC = 6, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_SF_RAW_LOC = 7, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_TF_LOC = 8,
|
||||
RFLAG_IF_LOC = 9,
|
||||
RFLAG_DF_LOC = 10,
|
||||
RFLAG_OF_RAW_LOC = 11, // Not used directly, needs to be reconstructed using `ReconstructCompactedEFLAGS`
|
||||
RFLAG_IOPL_LOC = 12,
|
||||
RFLAG_NT_LOC = 14,
|
||||
RFLAG_RF_LOC = 16,
|
||||
RFLAG_VM_LOC = 17,
|
||||
RFLAG_AC_LOC = 18,
|
||||
RFLAG_VIF_LOC = 19,
|
||||
RFLAG_VIP_LOC = 20,
|
||||
RFLAG_ID_LOC = 21,
|
||||
|
||||
// So we can implement arm64-like flag manipulaton on the x86 jit..
|
||||
// SF/ZF/CF/OF packed into a 32-bit word, matching arm64's NZCV structure (not semantics).
|
||||
|
||||
@@ -527,6 +527,12 @@ enum NamedVectorConstant : uint8_t {
|
||||
NAMED_VECTOR_PADDSUBPD_INVERT_UPPER,
|
||||
NAMED_VECTOR_MOVMSKPS_SHIFT,
|
||||
NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE,
|
||||
NAMED_VECTOR_BLENDPS_0110B,
|
||||
NAMED_VECTOR_BLENDPS_0111B,
|
||||
NAMED_VECTOR_BLENDPS_1001B,
|
||||
NAMED_VECTOR_BLENDPS_1011B,
|
||||
NAMED_VECTOR_BLENDPS_1101B,
|
||||
NAMED_VECTOR_BLENDPS_1110B,
|
||||
NAMED_VECTOR_CONST_POOL_MAX,
|
||||
// Beginning of named constants that don't have a constant pool backing.
|
||||
NAMED_VECTOR_ZERO = NAMED_VECTOR_CONST_POOL_MAX,
|
||||
@@ -541,6 +547,9 @@ enum IndexNamedVectorConstant : uint8_t {
|
||||
INDEXED_NAMED_VECTOR_PSHUFHW,
|
||||
INDEXED_NAMED_VECTOR_PSHUFD,
|
||||
INDEXED_NAMED_VECTOR_SHUFPS,
|
||||
INDEXED_NAMED_VECTOR_DPPS_MASK,
|
||||
INDEXED_NAMED_VECTOR_DPPD_MASK,
|
||||
INDEXED_NAMED_VECTOR_PBLENDW,
|
||||
INDEXED_NAMED_VECTOR_MAX,
|
||||
};
|
||||
|
||||
@@ -555,6 +564,22 @@ enum OpSize : uint8_t {
|
||||
i256Bit = 32,
|
||||
};
|
||||
|
||||
enum class FloatCompareOp : uint8_t {
|
||||
EQ = 0,
|
||||
LT,
|
||||
LE,
|
||||
UNO,
|
||||
NEQ,
|
||||
ORD,
|
||||
};
|
||||
|
||||
enum class ShiftType : uint8_t {
|
||||
LSL = 0,
|
||||
LSR,
|
||||
ASR,
|
||||
ROR,
|
||||
};
|
||||
|
||||
// Converts a size stored as an integer in to an OpSize enum.
|
||||
// This is a nop operation and will be eliminated by the compiler.
|
||||
static inline OpSize SizeToOpSize(uint8_t Size) {
|
||||
@@ -568,6 +593,7 @@ static inline OpSize SizeToOpSize(uint8_t Size) {
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
#define IROP_ENUM
|
||||
#define IROP_STRUCTS
|
||||
#define IROP_SIZES
|
||||
|
||||
@@ -27,6 +27,8 @@ friend class FEXCore::IR::PassManager;
|
||||
ResetWorkingList();
|
||||
}
|
||||
|
||||
virtual ~IREmitter() = default;
|
||||
|
||||
void ReownOrClaimBuffer() {
|
||||
DualListData.ReownOrClaimBuffer();
|
||||
}
|
||||
@@ -332,6 +334,10 @@ friend class FEXCore::IR::PassManager;
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
virtual void SaveNZCV() {
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
}
|
||||
|
||||
OrderedNode *CurrentWriteCursor = nullptr;
|
||||
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
|
||||
File renamed without changes.
@@ -0,0 +1,22 @@
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <catch2/catch.hpp>
|
||||
|
||||
TEST_CASE("LoadFile-Doesn'tExist") {
|
||||
fextl::string MapsFile;
|
||||
auto Read = FEXCore::FileLoading::LoadFile(MapsFile, "/tmp/a/b/c/d/e/z");
|
||||
REQUIRE(MapsFile.size() == 0);
|
||||
REQUIRE(Read == false);
|
||||
}
|
||||
|
||||
TEST_CASE("LoadFile-procfs") {
|
||||
fextl::string MapsFile;
|
||||
FEXCore::FileLoading::LoadFile(MapsFile, "/proc/self/maps");
|
||||
REQUIRE(MapsFile.size() != 0);
|
||||
}
|
||||
|
||||
TEST_CASE("LoadFile-Buffer") {
|
||||
fextl::string MapsFile;
|
||||
MapsFile.resize(16);
|
||||
auto Read = FEXCore::FileLoading::LoadFileToBuffer("/proc/self/maps", MapsFile);
|
||||
REQUIRE(MapsFile.size() == Read);
|
||||
}
|
||||
@@ -1709,6 +1709,9 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Evaluate into flags") {
|
||||
TEST_SINGLE(setf8(WReg::w30), "setf8 w30");
|
||||
TEST_SINGLE(setf16(WReg::w30), "setf16 w30");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Carry flag invert") {
|
||||
TEST_SINGLE(cfinv(), "cfinv");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Conditional compare - register") {
|
||||
TEST_SINGLE(ccmn(Size::i32Bit, Reg::r29, Reg::r28, StatusFlags::None, Condition::CC_AL), "ccmn w29, w28, #nzcv, al");
|
||||
TEST_SINGLE(ccmn(Size::i32Bit, Reg::r29, Reg::r28, StatusFlags::Flag_N, Condition::CC_AL), "ccmn w29, w28, #Nzcv, al");
|
||||
|
||||
@@ -1805,175 +1805,175 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE predicate read from FFR (u
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE predicate initialize") {
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.b, pow2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.h, pow2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.s, pow2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.d, pow2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.b, pow2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.h, pow2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.s, pow2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrue p6.d, pow2");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.b, pow2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.h, pow2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.s, pow2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.d, pow2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.b, pow2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.h, pow2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.s, pow2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_POW2), "ptrues p6.d, pow2");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.b, vl1");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.h, vl1");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.s, vl1");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.d, vl1");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.b, vl1");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.h, vl1");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.s, vl1");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrue p6.d, vl1");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.b, vl1");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.h, vl1");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.s, vl1");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.d, vl1");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.b, vl1");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.h, vl1");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.s, vl1");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL1), "ptrues p6.d, vl1");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.b, vl2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.h, vl2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.s, vl2");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.d, vl2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.b, vl2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.h, vl2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.s, vl2");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrue p6.d, vl2");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.b, vl2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.h, vl2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.s, vl2");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.d, vl2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.b, vl2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.h, vl2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.s, vl2");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL2), "ptrues p6.d, vl2");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.b, vl3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.h, vl3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.s, vl3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.d, vl3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.b, vl3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.h, vl3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.s, vl3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrue p6.d, vl3");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.b, vl3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.h, vl3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.s, vl3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.d, vl3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.b, vl3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.h, vl3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.s, vl3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL3), "ptrues p6.d, vl3");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.b, vl4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.h, vl4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.s, vl4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.d, vl4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.b, vl4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.h, vl4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.s, vl4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrue p6.d, vl4");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.b, vl4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.h, vl4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.s, vl4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.d, vl4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.b, vl4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.h, vl4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.s, vl4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL4), "ptrues p6.d, vl4");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.b, vl5");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.h, vl5");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.s, vl5");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.d, vl5");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.b, vl5");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.h, vl5");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.s, vl5");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrue p6.d, vl5");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.b, vl5");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.h, vl5");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.s, vl5");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.d, vl5");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.b, vl5");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.h, vl5");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.s, vl5");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL5), "ptrues p6.d, vl5");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.b, vl6");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.h, vl6");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.s, vl6");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.d, vl6");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.b, vl6");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.h, vl6");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.s, vl6");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrue p6.d, vl6");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.b, vl6");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.h, vl6");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.s, vl6");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.d, vl6");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.b, vl6");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.h, vl6");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.s, vl6");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL6), "ptrues p6.d, vl6");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.b, vl7");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.h, vl7");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.s, vl7");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.d, vl7");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.b, vl7");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.h, vl7");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.s, vl7");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrue p6.d, vl7");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.b, vl7");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.h, vl7");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.s, vl7");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.d, vl7");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.b, vl7");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.h, vl7");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.s, vl7");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL7), "ptrues p6.d, vl7");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.b, vl8");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.h, vl8");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.s, vl8");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.d, vl8");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.b, vl8");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.h, vl8");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.s, vl8");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrue p6.d, vl8");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.b, vl8");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.h, vl8");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.s, vl8");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.d, vl8");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.b, vl8");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.h, vl8");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.s, vl8");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL8), "ptrues p6.d, vl8");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.b, vl16");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.h, vl16");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.s, vl16");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.d, vl16");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.b, vl16");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.h, vl16");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.s, vl16");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrue p6.d, vl16");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.b, vl16");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.h, vl16");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.s, vl16");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.d, vl16");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.b, vl16");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.h, vl16");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.s, vl16");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL16), "ptrues p6.d, vl16");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.b, vl32");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.h, vl32");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.s, vl32");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.d, vl32");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.b, vl32");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.h, vl32");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.s, vl32");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrue p6.d, vl32");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.b, vl32");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.h, vl32");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.s, vl32");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.d, vl32");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.b, vl32");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.h, vl32");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.s, vl32");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL32), "ptrues p6.d, vl32");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.b, vl64");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.h, vl64");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.s, vl64");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.d, vl64");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.b, vl64");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.h, vl64");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.s, vl64");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrue p6.d, vl64");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.b, vl64");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.h, vl64");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.s, vl64");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.d, vl64");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.b, vl64");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.h, vl64");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.s, vl64");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL64), "ptrues p6.d, vl64");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.b, vl128");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.h, vl128");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.s, vl128");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.d, vl128");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.b, vl128");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.h, vl128");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.s, vl128");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrue p6.d, vl128");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.b, vl128");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.h, vl128");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.s, vl128");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.d, vl128");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.b, vl128");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.h, vl128");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.s, vl128");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL128), "ptrues p6.d, vl128");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.b, vl256");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.h, vl256");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.s, vl256");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.d, vl256");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.b, vl256");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.h, vl256");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.s, vl256");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrue p6.d, vl256");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.b, vl256");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.h, vl256");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.s, vl256");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.d, vl256");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.b, vl256");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.h, vl256");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.s, vl256");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_VL256), "ptrues p6.d, vl256");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.b, mul4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.h, mul4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.s, mul4");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.d, mul4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.b, mul4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.h, mul4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.s, mul4");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrue p6.d, mul4");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.b, mul4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.h, mul4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.s, mul4");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.d, mul4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.b, mul4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.h, mul4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.s, mul4");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_MUL4), "ptrues p6.d, mul4");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.b, mul3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.h, mul3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.s, mul3");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.d, mul3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.b, mul3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.h, mul3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.s, mul3");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrue p6.d, mul3");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.b, mul3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.h, mul3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.s, mul3");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.d, mul3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.b, mul3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.h, mul3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.s, mul3");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_MUL3), "ptrues p6.d, mul3");
|
||||
|
||||
TEST_SINGLE(ptrue<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.b");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.h");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.s");
|
||||
TEST_SINGLE(ptrue<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.d");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.b");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.h");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.s");
|
||||
TEST_SINGLE(ptrue(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrue p6.d");
|
||||
|
||||
TEST_SINGLE(ptrues<SubRegSize::i8Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.b");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i16Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.h");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i32Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.s");
|
||||
TEST_SINGLE(ptrues<SubRegSize::i64Bit>(PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.d");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i8Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.b");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i16Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.h");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i32Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.s");
|
||||
TEST_SINGLE(ptrues(SubRegSize::i64Bit, PReg::p6, PredicatePattern::SVE_ALL), "ptrues p6.d");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE integer compare scalar count and limit") {
|
||||
|
||||
@@ -687,6 +687,53 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point data-process
|
||||
TEST_SINGLE(frintx(HReg::h30, HReg::h29), "frintx h30, h29");
|
||||
TEST_SINGLE(frinti(HReg::h30, HReg::h29), "frinti h30, h29");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point data-processing (1 source sized)") {
|
||||
TEST_SINGLE(fmov(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "fmov s30, s29");
|
||||
TEST_SINGLE(fabs(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "fabs s30, s29");
|
||||
TEST_SINGLE(fneg(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "fneg s30, s29");
|
||||
TEST_SINGLE(fsqrt(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "fsqrt s30, s29");
|
||||
TEST_SINGLE(frintn(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintn s30, s29");
|
||||
TEST_SINGLE(frintp(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintp s30, s29");
|
||||
TEST_SINGLE(frintm(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintm s30, s29");
|
||||
TEST_SINGLE(frintz(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintz s30, s29");
|
||||
TEST_SINGLE(frinta(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frinta s30, s29");
|
||||
TEST_SINGLE(frintx(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frintx s30, s29");
|
||||
TEST_SINGLE(frinti(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frinti s30, s29");
|
||||
TEST_SINGLE(frint32z(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frint32z s30, s29");
|
||||
TEST_SINGLE(frint32x(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frint32x s30, s29");
|
||||
TEST_SINGLE(frint64z(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frint64z s30, s29");
|
||||
TEST_SINGLE(frint64x(ScalarRegSize::i32Bit, VReg::v30, VReg::v29), "frint64x s30, s29");
|
||||
|
||||
TEST_SINGLE(fmov(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "fmov d30, d29");
|
||||
TEST_SINGLE(fabs(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "fabs d30, d29");
|
||||
TEST_SINGLE(fneg(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "fneg d30, d29");
|
||||
TEST_SINGLE(fsqrt(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "fsqrt d30, d29");
|
||||
TEST_SINGLE(frintn(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintn d30, d29");
|
||||
TEST_SINGLE(frintp(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintp d30, d29");
|
||||
TEST_SINGLE(frintm(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintm d30, d29");
|
||||
TEST_SINGLE(frintz(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintz d30, d29");
|
||||
TEST_SINGLE(frinta(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frinta d30, d29");
|
||||
TEST_SINGLE(frintx(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frintx d30, d29");
|
||||
TEST_SINGLE(frinti(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frinti d30, d29");
|
||||
TEST_SINGLE(frint32z(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frint32z d30, d29");
|
||||
TEST_SINGLE(frint32x(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frint32x d30, d29");
|
||||
TEST_SINGLE(frint64z(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frint64z d30, d29");
|
||||
TEST_SINGLE(frint64x(ScalarRegSize::i64Bit, VReg::v30, VReg::v29), "frint64x d30, d29");
|
||||
|
||||
TEST_SINGLE(fmov(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "fmov h30, h29");
|
||||
TEST_SINGLE(fabs(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "fabs h30, h29");
|
||||
TEST_SINGLE(fneg(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "fneg h30, h29");
|
||||
TEST_SINGLE(fsqrt(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "fsqrt h30, h29");
|
||||
TEST_SINGLE(frintn(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintn h30, h29");
|
||||
TEST_SINGLE(frintp(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintp h30, h29");
|
||||
TEST_SINGLE(frintm(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintm h30, h29");
|
||||
TEST_SINGLE(frintz(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintz h30, h29");
|
||||
TEST_SINGLE(frinta(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frinta h30, h29");
|
||||
TEST_SINGLE(frintx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frintx h30, h29");
|
||||
TEST_SINGLE(frinti(ScalarRegSize::i16Bit, VReg::v30, VReg::v29), "frinti h30, h29");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point compare") {
|
||||
// Commented out lines showcase unallocated encodings.
|
||||
//TEST_SINGLE(fcmp(ScalarRegSize::i8Bit, VReg::v30, VReg::v29), "fcmp b30, b29");
|
||||
@@ -824,6 +871,39 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point data-process
|
||||
TEST_SINGLE(fminnm(HReg::h30, HReg::h29, HReg::h28), "fminnm h30, h29, h28");
|
||||
TEST_SINGLE(fnmul(HReg::h30, HReg::h29, HReg::h28), "fnmul h30, h29, h28");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point data-processing (2 source sized)") {
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fmul s30, s29, s28");
|
||||
TEST_SINGLE(fdiv(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fdiv s30, s29, s28");
|
||||
TEST_SINGLE(fadd(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fadd s30, s29, s28");
|
||||
TEST_SINGLE(fsub(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fsub s30, s29, s28");
|
||||
TEST_SINGLE(fmax(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fmax s30, s29, s28");
|
||||
TEST_SINGLE(fmin(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fmin s30, s29, s28");
|
||||
TEST_SINGLE(fmaxnm(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fmaxnm s30, s29, s28");
|
||||
TEST_SINGLE(fminnm(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fminnm s30, s29, s28");
|
||||
TEST_SINGLE(fnmul(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28), "fnmul s30, s29, s28");
|
||||
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fmul d30, d29, d28");
|
||||
TEST_SINGLE(fdiv(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fdiv d30, d29, d28");
|
||||
TEST_SINGLE(fadd(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fadd d30, d29, d28");
|
||||
TEST_SINGLE(fsub(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fsub d30, d29, d28");
|
||||
TEST_SINGLE(fmax(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fmax d30, d29, d28");
|
||||
TEST_SINGLE(fmin(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fmin d30, d29, d28");
|
||||
TEST_SINGLE(fmaxnm(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fmaxnm d30, d29, d28");
|
||||
TEST_SINGLE(fminnm(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fminnm d30, d29, d28");
|
||||
TEST_SINGLE(fnmul(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28), "fnmul d30, d29, d28");
|
||||
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fmul h30, h29, h28");
|
||||
TEST_SINGLE(fdiv(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fdiv h30, h29, h28");
|
||||
TEST_SINGLE(fadd(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fadd h30, h29, h28");
|
||||
TEST_SINGLE(fsub(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fsub h30, h29, h28");
|
||||
TEST_SINGLE(fmax(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fmax h30, h29, h28");
|
||||
TEST_SINGLE(fmin(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fmin h30, h29, h28");
|
||||
TEST_SINGLE(fmaxnm(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fmaxnm h30, h29, h28");
|
||||
TEST_SINGLE(fminnm(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fminnm h30, h29, h28");
|
||||
TEST_SINGLE(fnmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v28), "fnmul h30, h29, h28");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Floating-point conditional select") {
|
||||
TEST_SINGLE(fcsel(SReg::s30, SReg::s29, SReg::s28, Condition::CC_AL), "fcsel s30, s29, s28, al");
|
||||
TEST_SINGLE(fcsel(SReg::s30, SReg::s29, SReg::s28, Condition::CC_EQ), "fcsel s30, s29, s28, eq");
|
||||
|
||||
@@ -17,11 +17,13 @@ class TestData:
|
||||
optimal: int
|
||||
expectedinstructioncount: int
|
||||
code: bytes
|
||||
def __init__(self, Name, Optimal, ExpectedInstructionCount, Code):
|
||||
instructions: list
|
||||
def __init__(self, Name, Optimal, ExpectedInstructionCount, Code, Instructions):
|
||||
self.name = Name
|
||||
self.expectedinstructioncount = ExpectedInstructionCount
|
||||
self.optimal = Optimal
|
||||
self.code = Code
|
||||
self.instructions = Instructions
|
||||
|
||||
@property
|
||||
def Name(self):
|
||||
@@ -39,6 +41,10 @@ class TestData:
|
||||
def Code(self):
|
||||
return self.code
|
||||
|
||||
@property
|
||||
def Instructions(self):
|
||||
return self.instructions
|
||||
|
||||
TestDataMap = {}
|
||||
class HostFeatures(Flag) :
|
||||
FEATURE_ANY = 0
|
||||
@@ -48,7 +54,10 @@ class HostFeatures(Flag) :
|
||||
FEATURE_RNG = (1 << 3)
|
||||
FEATURE_FCMA = (1 << 4)
|
||||
FEATURE_CSSC = (1 << 5)
|
||||
|
||||
FEATURE_AFP = (1 << 6)
|
||||
FEATURE_RPRES = (1 << 7)
|
||||
FEATURE_FLAGM = (1 << 8)
|
||||
FEATURE_FLAGM2 = (1 << 9)
|
||||
|
||||
HostFeaturesLookup = {
|
||||
"SVE128" : HostFeatures.FEATURE_SVE128,
|
||||
@@ -57,6 +66,10 @@ HostFeaturesLookup = {
|
||||
"RNG" : HostFeatures.FEATURE_RNG,
|
||||
"FCMA" : HostFeatures.FEATURE_FCMA,
|
||||
"CSSC" : HostFeatures.FEATURE_CSSC,
|
||||
"AFP" : HostFeatures.FEATURE_AFP,
|
||||
"RPRES" : HostFeatures.FEATURE_RPRES,
|
||||
"FLAGM" : HostFeatures.FEATURE_FLAGM,
|
||||
"FLAGM2" : HostFeatures.FEATURE_FLAGM2,
|
||||
}
|
||||
|
||||
def GetHostFeatures(data):
|
||||
@@ -72,7 +85,7 @@ def GetHostFeatures(data):
|
||||
HostFeaturesData |= HostFeaturesLookup[data_key]
|
||||
return HostFeaturesData
|
||||
|
||||
def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
def parse_json_data(json_filepath, json_filename, json_data, output_binary_path):
|
||||
Bitness = 64
|
||||
EnabledHostFeatures = HostFeatures.FEATURE_ANY
|
||||
DisabledHostFeatures = HostFeatures.FEATURE_ANY
|
||||
@@ -100,6 +113,7 @@ def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
for key, items in json_data["Instructions"].items():
|
||||
ExpectedInstructionCount = 0
|
||||
Optimal = 0
|
||||
Instructions = []
|
||||
if ("ExpectedInstructionCount" in items):
|
||||
ExpectedInstructionCount = int(items["ExpectedInstructionCount"])
|
||||
|
||||
@@ -111,14 +125,23 @@ def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
if items["Skip"].upper() == "YES":
|
||||
continue
|
||||
|
||||
TestName = base64.b64encode("{}.{}".format(json_filename, key).encode("ascii")).decode("ascii")
|
||||
if "x86Insts" in items:
|
||||
Instructions = items["x86Insts"]
|
||||
else:
|
||||
# No list of instructions, only one which is the key.
|
||||
Instructions.append(key)
|
||||
TestName = base64.b64encode("{}.{}.{}".format(str(hash(json_filepath)), json_filename, key).encode("ascii")).decode("ascii")
|
||||
tmp_asm = "/tmp/{}.asm".format(TestName)
|
||||
tmp_asm_out = "/tmp/{}.asm.o".format(TestName)
|
||||
logging.info("'{}' -> '{}' -> '{}'".format(key, tmp_asm, tmp_asm_out))
|
||||
|
||||
if TestName in TestDataMap:
|
||||
sys.exit("Duplicate test name {} in tests".format(TestName))
|
||||
|
||||
with open(tmp_asm, "w") as tmp_asm_file:
|
||||
tmp_asm_file.write("BITS {};\n".format(Bitness))
|
||||
tmp_asm_file.write("{}\n".format(key))
|
||||
for Inst in Instructions:
|
||||
tmp_asm_file.write("{}\n".format(Inst))
|
||||
|
||||
Process = subprocess.Popen(["nasm", tmp_asm, "-o", tmp_asm_out])
|
||||
Process.wait()
|
||||
@@ -140,7 +163,7 @@ def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
with open(tmp_asm_out, "rb") as tmp_asm_out_file:
|
||||
binary_hex = tmp_asm_out_file.read()
|
||||
|
||||
TestDataMap[TestName] = TestData(key, Optimal, ExpectedInstructionCount, binary_hex)
|
||||
TestDataMap[TestName] = TestData(key, Optimal, ExpectedInstructionCount, binary_hex, Instructions)
|
||||
|
||||
os.remove(tmp_asm)
|
||||
os.remove(tmp_asm_out)
|
||||
@@ -161,6 +184,7 @@ def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
# uint64_t Optimal;
|
||||
# int64_t ExpectedInstructionCount;
|
||||
# uint64_t CodeSize;
|
||||
# uint64_t x86InstCount;
|
||||
# uint32_t Cookie;
|
||||
# uint8_t Code[CodeSize];
|
||||
# };
|
||||
@@ -187,6 +211,7 @@ def parse_json_data(json_filename, json_data, output_binary_path):
|
||||
MemData += struct.pack('Q', item.Optimal)
|
||||
MemData += struct.pack('q', item.ExpectedInstructionCount)
|
||||
MemData += struct.pack('Q', len(item.Code))
|
||||
MemData += struct.pack('Q', len(item.Instructions))
|
||||
MemData += struct.pack('I', 0x41424344)
|
||||
MemData += item.Code
|
||||
|
||||
@@ -218,7 +243,7 @@ def main():
|
||||
if not isinstance(json_data, dict):
|
||||
raise TypeError('JSON data must be a dict')
|
||||
|
||||
return parse_json_data(os.path.basename(json_path), json_data, output_binary_path)
|
||||
return parse_json_data(json_path, os.path.basename(json_path), json_data, output_binary_path)
|
||||
|
||||
except ValueError as ve:
|
||||
logging.error(f'JSON error: {ve}')
|
||||
|
||||
@@ -40,18 +40,17 @@ class Regs(Flag):
|
||||
REG_XMM15 = (1 << 32)
|
||||
REG_GS = (1 << 33)
|
||||
REG_FS = (1 << 34)
|
||||
REG_FLAGS = (1 << 35)
|
||||
REG_MM0 = (1 << 36)
|
||||
REG_MM1 = (1 << 37)
|
||||
REG_MM2 = (1 << 38)
|
||||
REG_MM3 = (1 << 39)
|
||||
REG_MM4 = (1 << 40)
|
||||
REG_MM5 = (1 << 41)
|
||||
REG_MM6 = (1 << 42)
|
||||
REG_MM7 = (1 << 43)
|
||||
REG_MM8 = (1 << 44)
|
||||
REG_ALL = (1 << 45) - 1
|
||||
REG_INVALID = (1 << 45)
|
||||
REG_MM0 = (1 << 35)
|
||||
REG_MM1 = (1 << 36)
|
||||
REG_MM2 = (1 << 37)
|
||||
REG_MM3 = (1 << 38)
|
||||
REG_MM4 = (1 << 39)
|
||||
REG_MM5 = (1 << 40)
|
||||
REG_MM6 = (1 << 41)
|
||||
REG_MM7 = (1 << 42)
|
||||
REG_MM8 = (1 << 43)
|
||||
REG_ALL = (1 << 44) - 1
|
||||
REG_INVALID = (1 << 44)
|
||||
|
||||
class ABI(Flag) :
|
||||
ABI_SYSTEMV = 0
|
||||
@@ -113,7 +112,6 @@ RegStringLookup = {
|
||||
"XMM15": Regs.REG_XMM15,
|
||||
"GS": Regs.REG_GS,
|
||||
"FS": Regs.REG_FS,
|
||||
"FLAGS": Regs.REG_FLAGS,
|
||||
"ALL": Regs.REG_ALL,
|
||||
"MM0": Regs.REG_MM0,
|
||||
"MM1": Regs.REG_MM1,
|
||||
|
||||
@@ -230,6 +230,7 @@ struct TestInfo {
|
||||
uint64_t Optimal;
|
||||
int64_t ExpectedInstructionCount;
|
||||
uint64_t CodeSize;
|
||||
uint64_t x86InstCount;
|
||||
uint32_t Cookie;
|
||||
uint8_t Code[];
|
||||
};
|
||||
@@ -258,7 +259,7 @@ static bool TestInstructions(FEXCore::Context::Context *CTX, FEXCore::Core::Inte
|
||||
LogMan::Msg::IFmt("Compiling instruction '{}'", CurrentTest->TestInst);
|
||||
|
||||
// Compile the INST.
|
||||
CTX->CompileRIP(Thread, CodeRIP);
|
||||
CTX->CompileRIPCount(Thread, CodeRIP, CurrentTest->x86InstCount);
|
||||
|
||||
// Go to the next test.
|
||||
CurrentTest = reinterpret_cast<TestInfo const*>(&CurrentTest->Code[CurrentTest->CodeSize]);
|
||||
@@ -462,6 +463,10 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEATURE_RNG = (1U << 3),
|
||||
FEATURE_FCMA = (1U << 4),
|
||||
FEATURE_CSSC = (1U << 5),
|
||||
FEATURE_AFP = (1U << 6),
|
||||
FEATURE_RPRES = (1U << 7),
|
||||
FEATURE_FLAGM = (1U << 8),
|
||||
FEATURE_FLAGM2 = (1U << 9),
|
||||
};
|
||||
|
||||
uint64_t SVEWidth = 0;
|
||||
@@ -486,9 +491,23 @@ int main(int argc, char **argv, char **const envp) {
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_CSSC) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLECSSC);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_AFP) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEAFP);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_RPRES) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLERPRES);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_FLAGM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEFLAGM);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_FLAGM2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEFLAGM2);
|
||||
}
|
||||
|
||||
// Always enable ARMv8.1 LSE atomics.
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEATOMICS);
|
||||
// Always enable crypto extensions.
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLECRYPTO);
|
||||
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_SVE128) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLESVE);
|
||||
@@ -508,6 +527,19 @@ int main(int argc, char **argv, char **const envp) {
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_CSSC) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLECSSC);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_AFP) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEAFP);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_RPRES) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLERPRES);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_FLAGM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEFLAGM);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_FLAGM2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEFLAGM2);
|
||||
}
|
||||
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_HOSTFEATURES, fextl::fmt::format("{}", HostFeatureControl));
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_FORCESVEWIDTH, fextl::fmt::format("{}", SVEWidth));
|
||||
|
||||
|
||||
@@ -37,6 +37,7 @@ if (NOT MINGW_BUILD)
|
||||
${PTHREAD_LIB}
|
||||
fmt::fmt
|
||||
)
|
||||
target_compile_options(${NAME} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
target_compile_definitions(${NAME} PRIVATE -DFEXLOADER_AS_INTERPRETER=${AsInterpreter})
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
|
||||
@@ -24,6 +24,7 @@
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEX::HarnessHelper {
|
||||
inline bool CompareStates(FEXCore::Core::CPUState const& State1,
|
||||
@@ -44,54 +45,11 @@ namespace FEX::HarnessHelper {
|
||||
fextl::fmt::print("{}: 0x{:016x} {} 0x{:016x}\n", Name, A, A==B ? "==" : "!=", B);
|
||||
};
|
||||
|
||||
const auto DumpFLAGs = [OutputGPRs](const fextl::string& Name, uint64_t A, uint64_t B) {
|
||||
if (!OutputGPRs) {
|
||||
return;
|
||||
}
|
||||
if (A == B) {
|
||||
return;
|
||||
}
|
||||
|
||||
static constexpr std::array<uint32_t, 17> Flags = {
|
||||
FEXCore::X86State::RFLAG_CF_LOC,
|
||||
FEXCore::X86State::RFLAG_PF_LOC,
|
||||
FEXCore::X86State::RFLAG_AF_LOC,
|
||||
FEXCore::X86State::RFLAG_ZF_LOC,
|
||||
FEXCore::X86State::RFLAG_SF_LOC,
|
||||
FEXCore::X86State::RFLAG_TF_LOC,
|
||||
FEXCore::X86State::RFLAG_IF_LOC,
|
||||
FEXCore::X86State::RFLAG_DF_LOC,
|
||||
FEXCore::X86State::RFLAG_OF_LOC,
|
||||
FEXCore::X86State::RFLAG_IOPL_LOC,
|
||||
FEXCore::X86State::RFLAG_NT_LOC,
|
||||
FEXCore::X86State::RFLAG_RF_LOC,
|
||||
FEXCore::X86State::RFLAG_VM_LOC,
|
||||
FEXCore::X86State::RFLAG_AC_LOC,
|
||||
FEXCore::X86State::RFLAG_VIF_LOC,
|
||||
FEXCore::X86State::RFLAG_VIP_LOC,
|
||||
FEXCore::X86State::RFLAG_ID_LOC,
|
||||
};
|
||||
|
||||
fextl::fmt::print("{}: 0x{:016x} {} 0x{:016x}\n", Name, A, A==B ? "==" : "!=", B);
|
||||
for (const auto Flag : Flags) {
|
||||
const auto FlagMask = uint64_t{1} << Flag;
|
||||
if ((A & FlagMask) != (B & FlagMask)) {
|
||||
fextl::fmt::print("\t{}: {} != {}\n", FEXCore::Core::GetFlagName(Flag), (A >> Flag) & 1, (B >> Flag) & 1);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const auto CheckGPRs = [&Matches, DumpGPRs](const fextl::string& Name, uint64_t A, uint64_t B){
|
||||
DumpGPRs(Name, A, B);
|
||||
Matches &= A == B;
|
||||
};
|
||||
|
||||
const auto CheckFLAGS = [&Matches, DumpFLAGs](const fextl::string& Name, uint64_t A, uint64_t B){
|
||||
DumpFLAGs(Name, A, B);
|
||||
Matches &= A == B;
|
||||
};
|
||||
|
||||
|
||||
// RIP
|
||||
if (MatchMask & 1) {
|
||||
CheckGPRs("RIP", State1.rip, State2.rip);
|
||||
@@ -135,22 +93,6 @@ namespace FEX::HarnessHelper {
|
||||
}
|
||||
MatchMask >>= 1;
|
||||
|
||||
auto CompactRFlags = [](auto Arg) -> uint32_t {
|
||||
uint32_t Res = 2;
|
||||
for (int i = 0; i < 32; ++i) {
|
||||
Res |= Arg->flags[i] << i;
|
||||
}
|
||||
return Res;
|
||||
};
|
||||
|
||||
// FLAGS
|
||||
if (MatchMask & 1) {
|
||||
uint32_t rflags1 = CompactRFlags(&State1);
|
||||
uint32_t rflags2 = CompactRFlags(&State2);
|
||||
|
||||
CheckFLAGS("FLAGS", rflags1, rflags2);
|
||||
}
|
||||
MatchMask >>= 1;
|
||||
return Matches;
|
||||
}
|
||||
|
||||
@@ -184,7 +126,7 @@ namespace FEX::HarnessHelper {
|
||||
}
|
||||
|
||||
if (BaseConfig.OptionRegDataCount > 0) {
|
||||
static constexpr std::array<uint64_t, 45> OffsetArrayAVX = {{
|
||||
static constexpr std::array<uint64_t, 44> OffsetArrayAVX = {{
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]),
|
||||
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RBX]),
|
||||
@@ -220,7 +162,6 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[15][0]),
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, flags),
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[1][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[2][0]),
|
||||
@@ -231,7 +172,7 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, mm[7][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[8][0]),
|
||||
}};
|
||||
static constexpr std::array<uint64_t, 45> OffsetArraySSE = {{
|
||||
static constexpr std::array<uint64_t, 44> OffsetArraySSE = {{
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RAX]),
|
||||
offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RBX]),
|
||||
@@ -267,7 +208,6 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[15][0]),
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, flags),
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[1][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[2][0]),
|
||||
@@ -440,17 +380,17 @@ namespace FEX::HarnessHelper {
|
||||
}
|
||||
|
||||
uint64_t StackSize() const override {
|
||||
return STACK_SIZE;
|
||||
return sysconf(_SC_PAGESIZE);
|
||||
}
|
||||
|
||||
uint64_t GetStackPointer() override {
|
||||
if (Config.Is64BitMode()) {
|
||||
return reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(STACK_SIZE)) + STACK_SIZE;
|
||||
return reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(StackSize())) + StackSize();
|
||||
}
|
||||
else {
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(reinterpret_cast<void*>(STACK_OFFSET), STACK_SIZE));
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(reinterpret_cast<void*>(STACK_OFFSET), StackSize()));
|
||||
LOGMAN_THROW_AA_FMT(Result != ~0ULL, "Stack Pointer mmap failed");
|
||||
return Result + STACK_SIZE;
|
||||
return Result + StackSize();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -466,11 +406,7 @@ namespace FEX::HarnessHelper {
|
||||
return Result;
|
||||
};
|
||||
|
||||
#ifndef _WIN32
|
||||
const auto AllocPageSize = FHU::FEX_PAGE_SIZE;
|
||||
#else
|
||||
const auto AllocPageSize = 64 * 1024;
|
||||
#endif
|
||||
const auto AllocPageSize = sysconf(_SC_PAGESIZE);
|
||||
if (LimitedSize) {
|
||||
DoMMap(0xe000'0000, AllocPageSize * 10);
|
||||
|
||||
@@ -538,7 +474,6 @@ namespace FEX::HarnessHelper {
|
||||
bool RequiresLinux() const { return Config.RequiresLinux(); }
|
||||
|
||||
private:
|
||||
constexpr static uint64_t STACK_SIZE = FHU::FEX_PAGE_SIZE;
|
||||
constexpr static uint64_t STACK_OFFSET = 0xc000'0000;
|
||||
// Zero is special case to know when we are done
|
||||
uint64_t Code_start_page = 0x1'0000;
|
||||
|
||||
@@ -416,8 +416,17 @@ fextl::string FileManager::GetEmulatedPath(const char *pathname, bool FollowSyml
|
||||
std::pair<int, const char*> FileManager::GetEmulatedFDPath(int dirfd, const char *pathname, bool FollowSymlink, FDPathTmpData &TmpFilename) {
|
||||
constexpr auto NoEntry = std::make_pair(-1, nullptr);
|
||||
|
||||
if (!pathname || // If no pathname
|
||||
pathname[0] != '/' || // If relative
|
||||
if (!pathname) {
|
||||
// No pathname.
|
||||
return NoEntry;
|
||||
}
|
||||
|
||||
if (pathname[0] == '/') {
|
||||
// If the path is absolute then dirfd is ignored.
|
||||
dirfd = AT_FDCWD;
|
||||
}
|
||||
|
||||
if (pathname[0] != '/' || // If relative
|
||||
pathname[1] == 0 || // If we are getting root
|
||||
dirfd != AT_FDCWD) { // If dirfd isn't special FDCWD
|
||||
return NoEntry;
|
||||
|
||||
@@ -160,13 +160,14 @@ namespace FEX::HLE::x32 {
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X32(clock_settime, [](FEXCore::Core::CpuStateFrame *Frame, clockid_t clockid, const timespec32 *tp) -> uint64_t {
|
||||
uint64_t Result = 0;
|
||||
if (tp) {
|
||||
const struct timespec tp64 = *tp;
|
||||
Result = ::clock_settime(clockid, &tp64);
|
||||
} else {
|
||||
Result = ::clock_settime(clockid, nullptr);
|
||||
if (!tp) {
|
||||
// clock_settime is required to pass a timespec.
|
||||
return -EFAULT;
|
||||
}
|
||||
|
||||
uint64_t Result = 0;
|
||||
const struct timespec tp64 = *tp;
|
||||
Result = ::clock_settime(clockid, &tp64);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
|
||||
@@ -128,9 +128,7 @@ void AnalysisAction::ExecuteAction() {
|
||||
|
||||
try {
|
||||
ParseInterface(context);
|
||||
if (StrictModeEnabled(context)) {
|
||||
CoverReferencedTypes(context);
|
||||
}
|
||||
CoverReferencedTypes(context);
|
||||
OnAnalysisComplete(context);
|
||||
} catch (ClangDiagnosticAsException& exception) {
|
||||
exception.Report(context.getDiagnostics());
|
||||
@@ -152,6 +150,33 @@ FindClassTemplateDeclByName(clang::DeclContext& decl_context, std::string_view s
|
||||
}
|
||||
}
|
||||
|
||||
struct TypeAnnotations {
|
||||
bool is_opaque = false;
|
||||
bool assumed_compatible = false;
|
||||
};
|
||||
|
||||
static TypeAnnotations GetTypeAnnotations(clang::ASTContext& context, clang::CXXRecordDecl* decl) {
|
||||
if (!decl->hasDefinition()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
ErrorReporter report_error { context };
|
||||
TypeAnnotations ret;
|
||||
|
||||
for (const clang::CXXBaseSpecifier& base : decl->bases()) {
|
||||
auto annotation = base.getType().getAsString();
|
||||
if (annotation == "fexgen::opaque_type") {
|
||||
ret.is_opaque = true;
|
||||
} else if (annotation == "fexgen::assume_compatible_data_layout") {
|
||||
ret.assumed_compatible = true;
|
||||
} else {
|
||||
throw report_error(base.getSourceRange().getBegin(), "Unknown type annotation");
|
||||
}
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
static ParameterAnnotations GetParameterAnnotations(clang::ASTContext& context, clang::CXXRecordDecl* decl) {
|
||||
if (!decl->hasDefinition()) {
|
||||
return {};
|
||||
@@ -161,7 +186,14 @@ static ParameterAnnotations GetParameterAnnotations(clang::ASTContext& context,
|
||||
ParameterAnnotations ret;
|
||||
|
||||
for (const clang::CXXBaseSpecifier& base : decl->bases()) {
|
||||
throw report_error(base.getSourceRange().getBegin(), "Unknown parameter annotation");
|
||||
auto annotation = base.getType().getAsString();
|
||||
if (annotation == "fexgen::ptr_passthrough") {
|
||||
ret.is_passthrough = true;
|
||||
} else if (annotation == "fexgen::assume_compatible_data_layout") {
|
||||
ret.assume_compatible = true;
|
||||
} else {
|
||||
throw report_error(base.getSourceRange().getBegin(), "Unknown parameter annotation");
|
||||
}
|
||||
}
|
||||
|
||||
return ret;
|
||||
@@ -170,6 +202,8 @@ static ParameterAnnotations GetParameterAnnotations(clang::ASTContext& context,
|
||||
void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
ErrorReporter report_error { context };
|
||||
|
||||
const std::unordered_map<unsigned, ParameterAnnotations> no_param_annotations {};
|
||||
|
||||
// TODO: Assert fex_gen_type is not declared at non-global namespaces
|
||||
if (auto template_decl = FindClassTemplateDeclByName(*context.getTranslationUnitDecl(), "fex_gen_type")) {
|
||||
for (auto* decl : template_decl->specializations()) {
|
||||
@@ -184,9 +218,14 @@ void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
type = type->getLocallyUnqualifiedSingleStepDesugaredType();
|
||||
|
||||
if (type->isFunctionPointerType() || type->isFunctionType()) {
|
||||
funcptr_types.insert(type.getTypePtr());
|
||||
thunked_funcptrs[type.getAsString()] = std::pair { type.getTypePtr(), no_param_annotations };
|
||||
} else {
|
||||
[[maybe_unused]] auto [it, inserted] = types.emplace(context.getCanonicalType(type.getTypePtr()), RepackedType { });
|
||||
const auto annotations = GetTypeAnnotations(context, decl);
|
||||
RepackedType repack_info = {
|
||||
.assumed_compatible = annotations.is_opaque || annotations.assumed_compatible,
|
||||
.pointers_only = annotations.is_opaque && !annotations.assumed_compatible,
|
||||
};
|
||||
[[maybe_unused]] auto [it, inserted] = types.emplace(context.getCanonicalType(type.getTypePtr()), repack_info);
|
||||
assert(inserted);
|
||||
}
|
||||
}
|
||||
@@ -307,9 +346,6 @@ void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
|
||||
auto check_struct_type = [&](const clang::Type* type) {
|
||||
if (type->isIncompleteType()) {
|
||||
if (!StrictModeEnabled(context)) {
|
||||
return;
|
||||
}
|
||||
throw report_error(type->getAsTagDecl()->getBeginLoc(), "Unannotated pointer with incomplete struct type; consider using an opaque_type annotation")
|
||||
.addNote(report_error(emitted_function->getNameInfo().getLoc(), "in function", clang::DiagnosticsEngine::Note))
|
||||
.addNote(report_error(template_arg_loc, "used in annotation here", clang::DiagnosticsEngine::Note));
|
||||
@@ -349,7 +385,7 @@ void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
|
||||
data.callbacks.emplace(param_idx, callback);
|
||||
if (!callback.is_stub && !callback.is_guest && !data.custom_host_impl) {
|
||||
funcptr_types.insert(context.getCanonicalType(funcptr));
|
||||
thunked_funcptrs[emitted_function->getNameAsString() + "_cb" + std::to_string(param_idx)] = std::pair { context.getCanonicalType(funcptr), no_param_annotations };
|
||||
}
|
||||
|
||||
if (data.callbacks.size() != 1) {
|
||||
@@ -358,20 +394,38 @@ void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
if (funcptr->isVariadic() && !callback.is_stub) {
|
||||
throw report_error(template_arg_loc, "Variadic callbacks are not supported");
|
||||
}
|
||||
|
||||
// Force treatment as passthrough-pointer
|
||||
data.param_annotations[param_idx].is_passthrough = true;
|
||||
} else if (param_type->isBuiltinType()) {
|
||||
// NOTE: Intentionally not using getCanonicalType here since that would turn e.g. size_t into platform-specific types
|
||||
// TODO: Still, we may want to de-duplicate some of these...
|
||||
types.emplace(param_type.getTypePtr(), RepackedType { });
|
||||
} else if (param_type->isEnumeralType()) {
|
||||
types.emplace(context.getCanonicalType(param_type.getTypePtr()), RepackedType { });
|
||||
} else if ( param_type->isStructureType()) {
|
||||
} else if ( param_type->isStructureType() &&
|
||||
!(types.contains(context.getCanonicalType(param_type.getTypePtr())) &&
|
||||
LookupType(context, param_type.getTypePtr()).assumed_compatible)) {
|
||||
check_struct_type(param_type.getTypePtr());
|
||||
types.emplace(context.getCanonicalType(param_type.getTypePtr()), RepackedType { });
|
||||
} else if (param_type->isPointerType()) {
|
||||
auto pointee_type = param_type->getPointeeType()->getLocallyUnqualifiedSingleStepDesugaredType();
|
||||
if ( pointee_type->isStructureType()) {
|
||||
if (data.param_annotations[param_idx].assume_compatible) {
|
||||
// Nothing to do
|
||||
} else if (types.contains(context.getCanonicalType(pointee_type.getTypePtr())) && LookupType(context, pointee_type.getTypePtr()).assumed_compatible) {
|
||||
// Parameter points to a type that is assumed compatible
|
||||
data.param_annotations[param_idx].assume_compatible = true;
|
||||
} else if (pointee_type->isStructureType()) {
|
||||
// Unannotated pointer to unannotated structure.
|
||||
// Append the structure type to the type list for checking data layout compatibility.
|
||||
check_struct_type(pointee_type.getTypePtr());
|
||||
types.emplace(context.getCanonicalType(pointee_type.getTypePtr()), RepackedType { });
|
||||
} else if (data.param_annotations[param_idx].is_passthrough) {
|
||||
if (!data.custom_host_impl) {
|
||||
throw report_error(param_loc, "Passthrough annotation requires custom host implementation");
|
||||
}
|
||||
|
||||
// Nothing to do
|
||||
} else if (false /* TODO: Can't check if this is unsupported until data layout analysis is complete */) {
|
||||
throw report_error(param_loc, "Unsupported parameter type")
|
||||
.addNote(report_error(emitted_function->getNameInfo().getLoc(), "in function", clang::DiagnosticsEngine::Note))
|
||||
@@ -417,7 +471,7 @@ void AnalysisAction::ParseInterface(clang::ASTContext& context) {
|
||||
|
||||
// For indirect calls, register the function signature as a function pointer type
|
||||
if (namespace_info.indirect_guest_calls) {
|
||||
funcptr_types.insert(context.getCanonicalType(emitted_function->getFunctionType()));
|
||||
thunked_funcptrs[emitted_function->getNameAsString()] = std::pair { context.getCanonicalType(emitted_function->getFunctionType()), data.param_annotations };
|
||||
}
|
||||
|
||||
thunks.push_back(std::move(data));
|
||||
@@ -439,6 +493,11 @@ void AnalysisAction::CoverReferencedTypes(clang::ASTContext& context) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (type_repack_info.assumed_compatible) {
|
||||
// If assumed compatible, we don't need the member definitions
|
||||
continue;
|
||||
}
|
||||
|
||||
for (auto* member : type->getAsStructureType()->getDecl()->fields()) {
|
||||
auto member_type = member->getType().getTypePtr();
|
||||
while (member_type->isArrayType()) {
|
||||
@@ -451,6 +510,15 @@ void AnalysisAction::CoverReferencedTypes(clang::ASTContext& context) {
|
||||
if (!member_type->isBuiltinType()) {
|
||||
member_type = context.getCanonicalType(member_type);
|
||||
}
|
||||
if (types.contains(member_type) && types.at(member_type).pointers_only) {
|
||||
if (member_type == context.getCanonicalType(member->getType().getTypePtr())) {
|
||||
throw std::runtime_error(fmt::format("\"{}\" references opaque type \"{}\" via non-pointer member \"{}\"",
|
||||
clang::QualType { type, 0 }.getAsString(),
|
||||
clang::QualType { member_type, 0 }.getAsString(),
|
||||
member->getNameAsString()));
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (member_type->isUnionType() && !types.contains(member_type)) {
|
||||
throw std::runtime_error(fmt::format("\"{}\" has unannotated member \"{}\" of union type \"{}\"",
|
||||
clang::QualType { type, 0 }.getAsString(),
|
||||
|
||||
@@ -23,6 +23,9 @@ struct ThunkedCallback : FunctionParams {
|
||||
};
|
||||
|
||||
struct ParameterAnnotations {
|
||||
bool is_passthrough = false;
|
||||
bool assume_compatible = false;
|
||||
|
||||
bool operator==(const ParameterAnnotations&) const = default;
|
||||
};
|
||||
|
||||
@@ -115,6 +118,8 @@ public:
|
||||
std::unique_ptr<clang::ASTConsumer> CreateASTConsumer(clang::CompilerInstance&, clang::StringRef /*file*/) override;
|
||||
|
||||
struct RepackedType {
|
||||
bool assumed_compatible = false; // opaque_type or assume_compatible_data_layout
|
||||
bool pointers_only = assumed_compatible; // if true, only pointers to this type may be used
|
||||
};
|
||||
|
||||
protected:
|
||||
@@ -132,7 +137,10 @@ protected:
|
||||
std::vector<ThunkedFunction> thunks;
|
||||
std::vector<ThunkedAPIFunction> thunked_api;
|
||||
|
||||
std::unordered_set<const clang::Type*> funcptr_types;
|
||||
// Set of function types for which to generate Guest->Host thunking trampolines.
|
||||
// The map key is a unique identifier that must be consistent between guest/host processing passes.
|
||||
// The map value is a pair of the function pointer's clang::Type and the mapping of parameter annotations
|
||||
std::unordered_map<std::string, std::pair<const clang::Type*, std::unordered_map<unsigned, ParameterAnnotations>>> thunked_funcptrs;
|
||||
|
||||
std::unordered_map<const clang::Type*, RepackedType> types;
|
||||
std::optional<unsigned> lib_version;
|
||||
@@ -179,11 +187,3 @@ inline std::string get_type_name(const clang::ASTContext& context, const clang::
|
||||
}
|
||||
return type_name;
|
||||
}
|
||||
|
||||
// Analysis can't process interfaces of real libraries, yet. This function
|
||||
// defines a "strict mode" to use for tests, only. Real libraries will switch
|
||||
// to strict mode once analysis is more feature-complete.
|
||||
inline bool StrictModeEnabled(clang::ASTContext& context) {
|
||||
auto filename = context.getSourceManager().getFileEntryForID(context.getSourceManager().getMainFileID())->getName();
|
||||
return filename.endswith("libfex_thunk_test_interface.cpp") || filename.endswith("gen_input.cpp");
|
||||
}
|
||||
@@ -26,6 +26,14 @@ std::unordered_map<const clang::Type*, TypeInfo> ComputeDataLayout(const clang::
|
||||
|
||||
// First, add all types directly used in function signatures of the library API to the meta set
|
||||
for (const auto& [type, type_repack_info] : types) {
|
||||
if (type_repack_info.assumed_compatible) {
|
||||
auto [_, inserted] = layout.insert(std::pair { context.getCanonicalType(type), TypeInfo {} });
|
||||
if (!inserted) {
|
||||
throw std::runtime_error("Failed to gather type metadata: Opaque type \"" + clang::QualType { type, 0 }.getAsString() + "\" already registered");
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (type->isIncompleteType()) {
|
||||
throw std::runtime_error("Cannot compute data layout of incomplete type \"" + clang::QualType { type, 0 }.getAsString() + "\". Did you forget any annotations?");
|
||||
}
|
||||
@@ -54,7 +62,7 @@ std::unordered_map<const clang::Type*, TypeInfo> ComputeDataLayout(const clang::
|
||||
|
||||
// Then, add information about members
|
||||
for (const auto& [type, type_repack_info] : types) {
|
||||
if (!type->isStructureType()) {
|
||||
if (!type->isStructureType() || type_repack_info.assumed_compatible) {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -151,9 +159,31 @@ ABI GetStableLayout(const clang::ASTContext& context, const std::unordered_map<c
|
||||
return stable_layout;
|
||||
}
|
||||
|
||||
static std::array<uint8_t, 32> GetSha256(const std::string& function_name) {
|
||||
std::array<uint8_t, 32> sha256;
|
||||
SHA256(reinterpret_cast<const unsigned char*>(function_name.data()),
|
||||
function_name.size(),
|
||||
sha256.data());
|
||||
return sha256;
|
||||
};
|
||||
|
||||
void AnalyzeDataLayoutAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
if (StrictModeEnabled(context)) {
|
||||
type_abi = GetStableLayout(context, ComputeDataLayout(context, types));
|
||||
|
||||
// Register functions that must be guest-callable through host function pointers
|
||||
for (auto funcptr_type_it = thunked_funcptrs.begin(); funcptr_type_it != thunked_funcptrs.end(); ++funcptr_type_it) {
|
||||
auto& funcptr_id = funcptr_type_it->first;
|
||||
auto& [type, param_annotations] = funcptr_type_it->second;
|
||||
auto func_type = type->getAs<clang::FunctionProtoType>();
|
||||
std::string mangled_name = clang::QualType { type, 0 }.getAsString();
|
||||
auto cb_sha256 = GetSha256("fexcallback_" + mangled_name);
|
||||
FuncPtrInfo info = { cb_sha256 };
|
||||
|
||||
info.result = func_type->getReturnType().getAsString();
|
||||
for (auto arg : func_type->getParamTypes()) {
|
||||
info.args.push_back(arg.getAsString());
|
||||
}
|
||||
type_abi.thunked_funcptrs[funcptr_id] = std::move(info);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -178,6 +208,14 @@ TypeCompatibility DataLayoutCompareAction::GetTypeCompatibility(
|
||||
}
|
||||
}
|
||||
|
||||
if (types.contains(type) && types.at(type).assumed_compatible) {
|
||||
if (types.at(type).pointers_only && !type->isPointerType()) {
|
||||
throw std::runtime_error("Tried to dereference opaque type \"" + clang::QualType { type, 0 }.getAsString() + "\" when querying data layout compatibility");
|
||||
}
|
||||
type_compat.at(type) = TypeCompatibility::Full;
|
||||
return TypeCompatibility::Full;
|
||||
}
|
||||
|
||||
auto type_name = get_type_name(context, type);
|
||||
auto& guest_info = guest_abi.at(type_name);
|
||||
auto& host_info = host_abi.at(type->isBuiltinType() ? type : context.getCanonicalType(type));
|
||||
@@ -233,12 +271,18 @@ TypeCompatibility DataLayoutCompareAction::GetTypeCompatibility(
|
||||
// * Pointer member is annotated
|
||||
// TODO: Don't restrict this to structure types. it applies to pointers to builtin types too!
|
||||
auto host_member_pointee_type = context.getCanonicalType(host_member_type->getPointeeType().getTypePtr());
|
||||
if (host_member_pointee_type->isPointerType()) {
|
||||
if (types.contains(host_member_pointee_type) && types.at(host_member_pointee_type).assumed_compatible) {
|
||||
// Pointee doesn't need repacking, but pointer needs extending on 32-bit
|
||||
member_compat.push_back(is_32bit ? TypeCompatibility::Repackable : TypeCompatibility::Full);
|
||||
} else if (host_member_pointee_type->isPointerType()) {
|
||||
// This is a nested pointer, e.g. void**
|
||||
|
||||
if (is_32bit) {
|
||||
// Nested pointers can't be repacked on 32-bit
|
||||
member_compat.push_back(TypeCompatibility::None);
|
||||
} else if (types.contains(host_member_pointee_type->getPointeeType().getTypePtr()) && types.at(host_member_pointee_type->getPointeeType().getTypePtr()).assumed_compatible) {
|
||||
// Pointers to opaque types are fine
|
||||
member_compat.push_back(TypeCompatibility::Full);
|
||||
} else {
|
||||
// Check the innermost type's compatibility on 64-bit
|
||||
auto pointee_pointee_type = host_member_pointee_type->getPointeeType().getTypePtr();
|
||||
@@ -297,6 +341,10 @@ TypeCompatibility DataLayoutCompareAction::GetTypeCompatibility(
|
||||
return compat;
|
||||
}
|
||||
|
||||
FuncPtrInfo DataLayoutCompareAction::LookupGuestFuncPtrInfo(const char* funcptr_id) {
|
||||
return guest_abi.thunked_funcptrs.at(funcptr_id);
|
||||
}
|
||||
|
||||
DataLayoutCompareActionFactory::DataLayoutCompareActionFactory(const ABI& abi) : abi(abi) {
|
||||
|
||||
}
|
||||
|
||||
@@ -84,6 +84,7 @@ struct FuncPtrInfo {
|
||||
};
|
||||
|
||||
struct ABI : std::unordered_map<std::string, TypeInfo> {
|
||||
std::unordered_map<std::string, FuncPtrInfo> thunked_funcptrs;
|
||||
int pointer_size; // in bytes
|
||||
};
|
||||
|
||||
@@ -111,6 +112,8 @@ public:
|
||||
const std::unordered_map<const clang::Type*, TypeInfo> host_abi,
|
||||
std::unordered_map<const clang::Type*, TypeCompatibility>& type_compat);
|
||||
|
||||
FuncPtrInfo LookupGuestFuncPtrInfo(const char* funcptr_id);
|
||||
|
||||
private:
|
||||
const ABI& guest_abi;
|
||||
};
|
||||
+69
-35
@@ -55,11 +55,11 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
// Compute data layout differences between host and guest
|
||||
auto type_compat = [&]() {
|
||||
std::unordered_map<const clang::Type*, TypeCompatibility> ret;
|
||||
if (StrictModeEnabled(context)) {
|
||||
const auto host_abi = ComputeDataLayout(context, types);
|
||||
for (const auto& [type, type_repack_info] : types) {
|
||||
GetTypeCompatibility(context, type, host_abi, ret);
|
||||
}
|
||||
if (!type_repack_info.pointers_only) {
|
||||
GetTypeCompatibility(context, type, host_abi, ret);
|
||||
}
|
||||
}
|
||||
return ret;
|
||||
}();
|
||||
@@ -101,14 +101,6 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
}
|
||||
};
|
||||
|
||||
auto format_struct_members = [](const FunctionParams& params, const char* indent) {
|
||||
std::string ret;
|
||||
for (std::size_t idx = 0; idx < params.param_types.size(); ++idx) {
|
||||
ret += indent + format_decl(params.param_types[idx].getUnqualifiedType(), fmt::format("a_{}", idx)) + ";\n";
|
||||
}
|
||||
return ret;
|
||||
};
|
||||
|
||||
auto format_function_params = [](const FunctionParams& params) {
|
||||
std::string ret;
|
||||
for (std::size_t idx = 0; idx < params.param_types.size(); ++idx) {
|
||||
@@ -120,8 +112,8 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
return ret;
|
||||
};
|
||||
|
||||
auto get_sha256 = [this](const std::string& function_name) {
|
||||
std::string sha256_message = libname + ":" + function_name;
|
||||
auto get_sha256 = [this](const std::string& function_name, bool include_libname) {
|
||||
std::string sha256_message = (include_libname ? libname + ":" : "") + function_name;
|
||||
std::vector<unsigned char> sha256(SHA256_DIGEST_LENGTH);
|
||||
SHA256(reinterpret_cast<const unsigned char*>(sha256_message.data()),
|
||||
sha256_message.size(),
|
||||
@@ -141,22 +133,30 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
file << "extern \"C\" {\n";
|
||||
for (auto& thunk : thunks) {
|
||||
const auto& function_name = thunk.function_name;
|
||||
auto sha256 = get_sha256(function_name);
|
||||
auto sha256 = get_sha256(function_name, true);
|
||||
fmt::print( file, "MAKE_THUNK({}, {}, \"{:#02x}\")\n",
|
||||
libname, function_name, fmt::join(sha256, ", "));
|
||||
}
|
||||
file << "}\n";
|
||||
|
||||
// Guest->Host transition points for invoking runtime host-function pointers based on their signature
|
||||
for (auto type_it = funcptr_types.begin(); type_it != funcptr_types.end(); ++type_it) {
|
||||
auto* type = *type_it;
|
||||
std::vector<std::vector<unsigned char>> sha256s;
|
||||
for (auto type_it = thunked_funcptrs.begin(); type_it != thunked_funcptrs.end(); ++type_it) {
|
||||
auto* type = type_it->second.first;
|
||||
std::string funcptr_signature = clang::QualType { type, 0 }.getAsString();
|
||||
|
||||
auto cb_sha256 = get_sha256("fexcallback_" + funcptr_signature);
|
||||
auto cb_sha256 = get_sha256("fexcallback_" + funcptr_signature, false);
|
||||
auto it = std::find(sha256s.begin(), sha256s.end(), cb_sha256);
|
||||
if (it != sha256s.end()) {
|
||||
// TODO: Avoid this ugly way of avoiding duplicates
|
||||
continue;
|
||||
} else {
|
||||
sha256s.push_back(cb_sha256);
|
||||
}
|
||||
|
||||
// Thunk used for guest-side calls to host function pointers
|
||||
file << " // " << funcptr_signature << "\n";
|
||||
auto funcptr_idx = std::distance(funcptr_types.begin(), type_it);
|
||||
auto funcptr_idx = std::distance(thunked_funcptrs.begin(), type_it);
|
||||
fmt::print( file, " MAKE_CALLBACK_THUNK(callback_{}, {}, \"{:#02x}\");\n",
|
||||
funcptr_idx, funcptr_signature, fmt::join(cb_sha256, ", "));
|
||||
}
|
||||
@@ -278,8 +278,10 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
auto cb = thunk.callbacks.find(idx);
|
||||
if (cb != thunk.callbacks.end() && cb->second.is_guest) {
|
||||
file << "fex_guest_function_ptr a_" << idx;
|
||||
} else if (thunk.param_annotations[idx].is_passthrough) {
|
||||
fmt::print(file, "guest_layout<{}> a_{}", type.getAsString(), idx);
|
||||
} else {
|
||||
file << format_decl(type, fmt::format("a_{}", idx));
|
||||
file << format_decl(type, fmt::format("a_{}", idx));
|
||||
}
|
||||
}
|
||||
// Using trailing return type as it makes handling function pointer returns much easier
|
||||
@@ -289,27 +291,29 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
// Check data layout compatibility of parameter types
|
||||
// TODO: Also check non-struct/non-pointer types
|
||||
// TODO: Also check return type
|
||||
for (size_t param_idx = 0; StrictModeEnabled(context) && param_idx != thunk.param_types.size(); ++param_idx) {
|
||||
for (size_t param_idx = 0; param_idx != thunk.param_types.size(); ++param_idx) {
|
||||
const auto& param_type = thunk.param_types[param_idx];
|
||||
if (!param_type->isPointerType() || !param_type->getPointeeType()->isStructureType()) {
|
||||
continue;
|
||||
}
|
||||
auto type = param_type->getPointeeType();
|
||||
if (type_compat.at(context.getCanonicalType(type.getTypePtr())) == TypeCompatibility::None) {
|
||||
if (!types.at(context.getCanonicalType(type.getTypePtr())).assumed_compatible && type_compat.at(context.getCanonicalType(type.getTypePtr())) == TypeCompatibility::None) {
|
||||
// TODO: Factor in "assume_compatible_layout" annotations here
|
||||
// That annotation should cause the type to be treated as TypeCompatibility::Full
|
||||
{
|
||||
if (!thunk.param_annotations[param_idx].is_passthrough) {
|
||||
throw report_error(thunk.decl->getLocation(), "Unsupported parameter type %0").AddTaggedVal(param_type);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Packed argument structs used in fexfn_unpack_*
|
||||
auto GeneratePackedArgs = [&](const auto &function_name, const auto &thunk) -> std::string {
|
||||
auto GeneratePackedArgs = [&](const auto &function_name, const ThunkedFunction &thunk) -> std::string {
|
||||
std::string struct_name = "fexfn_packed_args_" + libname + "_" + function_name;
|
||||
file << "struct " << struct_name << " {\n";
|
||||
|
||||
file << format_struct_members(thunk, " ");
|
||||
for (std::size_t idx = 0; idx < thunk.param_types.size(); ++idx) {
|
||||
fmt::print(file, " guest_layout<{}> a_{};\n", get_type_name(context, thunk.param_types[idx].getTypePtr()), idx);
|
||||
}
|
||||
if (!thunk.return_type->isVoidType()) {
|
||||
file << " " << format_decl(thunk.return_type, "rv") << ";\n";
|
||||
} else if (thunk.param_types.size() == 0) {
|
||||
@@ -330,12 +334,15 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
file << "static void fexfn_unpack_" << libname << "_" << function_name << "(" << struct_name << "* args) {\n";
|
||||
file << (thunk.return_type->isVoidType() ? " " : " args->rv = ") << function_to_call << "(";
|
||||
|
||||
for (unsigned param_idx = 0; StrictModeEnabled(context) && param_idx != thunk.param_types.size(); ++param_idx) {
|
||||
for (unsigned param_idx = 0; param_idx != thunk.param_types.size(); ++param_idx) {
|
||||
if (thunk.callbacks.contains(param_idx) && thunk.callbacks.at(param_idx).is_stub) {
|
||||
continue;
|
||||
}
|
||||
|
||||
auto& param_type = thunk.param_types[param_idx];
|
||||
const bool is_assumed_compatible = param_type->isPointerType() &&
|
||||
(thunk.param_annotations[param_idx].assume_compatible || ((param_type->getPointeeType()->isStructureType() || (param_type->getPointeeType()->isPointerType() && param_type->getPointeeType()->getPointeeType()->isStructureType())) &&
|
||||
(types.contains(context.getCanonicalType(param_type->getPointeeType()->getLocallyUnqualifiedSingleStepDesugaredType().getTypePtr())) && LookupType(context, context.getCanonicalType(param_type->getPointeeType()->getLocallyUnqualifiedSingleStepDesugaredType().getTypePtr())).assumed_compatible)));
|
||||
|
||||
std::optional<TypeCompatibility> pointee_compat;
|
||||
if (param_type->isPointerType()) {
|
||||
@@ -344,7 +351,12 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
pointee_compat = type_compat.emplace(context.getCanonicalType(param_type->getPointeeType().getTypePtr()), TypeCompatibility::Full).first->second;
|
||||
}
|
||||
|
||||
if (!param_type->isPointerType() || pointee_compat == TypeCompatibility::Full ||
|
||||
if (thunk.param_annotations[param_idx].is_passthrough) {
|
||||
// args are passed directly to function, no need to use `unpacked` wrappers
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!param_type->isPointerType() || (is_assumed_compatible || pointee_compat == TypeCompatibility::Full) ||
|
||||
param_type->getPointeeType()->isBuiltinType() /* TODO: handle size_t. Actually, properly check for data layout compatibility */) {
|
||||
// Fully compatible
|
||||
} else if (pointee_compat == TypeCompatibility::Repackable) {
|
||||
@@ -356,18 +368,22 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
|
||||
{
|
||||
auto format_param = [&](std::size_t idx) {
|
||||
std::string raw_arg = fmt::format("args->a_{}.data", idx);
|
||||
|
||||
auto cb = thunk.callbacks.find(idx);
|
||||
if (cb != thunk.callbacks.end() && cb->second.is_stub) {
|
||||
return "fexfn_unpack_" + get_callback_name(function_name, cb->first) + "_stub";
|
||||
} else if (cb != thunk.callbacks.end() && cb->second.is_guest) {
|
||||
return fmt::format("fex_guest_function_ptr {{ args->a_{} }}", idx);
|
||||
return fmt::format("fex_guest_function_ptr {{ {} }}", raw_arg);
|
||||
} else if (cb != thunk.callbacks.end()) {
|
||||
auto arg_name = fmt::format("args->a_{}", idx);
|
||||
auto arg_name = fmt::format("args->a_{}.data", idx);
|
||||
// Use comma operator to inject a function call before returning the argument
|
||||
return "(FinalizeHostTrampolineForGuestFunction(" + arg_name + "), " + arg_name + ")";
|
||||
|
||||
} else {
|
||||
} else if (thunk.param_annotations[idx].is_passthrough) {
|
||||
// Pass raw guest_layout<T*>
|
||||
return fmt::format("args->a_{}", idx);
|
||||
} else {
|
||||
return raw_arg;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -382,18 +398,36 @@ void GenerateThunkLibsAction::OnAnalysisComplete(clang::ASTContext& context) {
|
||||
file << "static ExportEntry exports[] = {\n";
|
||||
for (auto& thunk : thunks) {
|
||||
const auto& function_name = thunk.function_name;
|
||||
auto sha256 = get_sha256(function_name);
|
||||
auto sha256 = get_sha256(function_name, true);
|
||||
fmt::print( file, " {{(uint8_t*)\"\\x{:02x}\", (void(*)(void *))&fexfn_unpack_{}_{}}}, // {}:{}\n",
|
||||
fmt::join(sha256, "\\x"), libname, function_name, libname, function_name);
|
||||
}
|
||||
|
||||
// Endpoints for Guest->Host invocation of runtime host-function pointers
|
||||
for (auto& type : funcptr_types) {
|
||||
for (auto& host_funcptr_entry : thunked_funcptrs) {
|
||||
auto& [type, param_annotations] = host_funcptr_entry.second;
|
||||
std::string mangled_name = clang::QualType { type, 0 }.getAsString();
|
||||
auto cb_sha256 = get_sha256("fexcallback_" + mangled_name);
|
||||
fmt::print( file, " {{(uint8_t*)\"\\x{:02x}\", (void(*)(void *))&CallbackUnpack<{}>::ForIndirectCall}},\n",
|
||||
fmt::join(cb_sha256, "\\x"), mangled_name);
|
||||
auto info = LookupGuestFuncPtrInfo(host_funcptr_entry.first.c_str());
|
||||
|
||||
std::string annotations;
|
||||
for (int param_idx = 0; param_idx < info.args.size(); ++param_idx) {
|
||||
if (param_idx != 0) {
|
||||
annotations += ", ";
|
||||
}
|
||||
|
||||
annotations += "ParameterAnnotations {";
|
||||
if (param_annotations.contains(param_idx) && param_annotations.at(param_idx).is_passthrough) {
|
||||
annotations += ".is_passthrough=true,";
|
||||
}
|
||||
if (param_annotations.contains(param_idx) && param_annotations.at(param_idx).assume_compatible) {
|
||||
annotations += ".assume_compatible=true,";
|
||||
}
|
||||
annotations += "}";
|
||||
}
|
||||
fmt::print( file, " {{(uint8_t*)\"\\x{:02x}\", (void(*)(void *))&GuestWrapperForHostFunction<{}({})>::Call<{}>}}, // {}\n",
|
||||
fmt::join(info.sha256, "\\x"), info.result, fmt::join(info.args, ", "), annotations, host_funcptr_entry.first);
|
||||
}
|
||||
|
||||
file << " { nullptr, nullptr }\n";
|
||||
file << "};\n";
|
||||
|
||||
|
||||
@@ -17,7 +17,7 @@ endif()
|
||||
|
||||
if (CMAKE_CURRENT_SOURCE_DIR STREQUAL CMAKE_SOURCE_DIR)
|
||||
# We've been included using ExternalProject_add, so set up the actual thunk libraries to be cross-compiled
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
|
||||
# This gets passed in from the main cmake project
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
@@ -1,119 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <linux/futex.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <mutex>
|
||||
#include <unistd.h>
|
||||
|
||||
struct CrossArchEvent final {
|
||||
std::atomic<uint32_t> Futex;
|
||||
};
|
||||
|
||||
static void WaitForWorkFunc(CrossArchEvent *Event) {
|
||||
|
||||
// Wait for Futex value to become 1
|
||||
while (true) {
|
||||
|
||||
// First step compare it with 1 already and see if we can early out
|
||||
uint32_t One = 1;
|
||||
if (Event->Futex.compare_exchange_strong(One, 0)) {
|
||||
return;
|
||||
}
|
||||
|
||||
int Op = FUTEX_WAIT | FUTEX_PRIVATE_FLAG;
|
||||
[[maybe_unused]] int Res = syscall(SYS_futex,
|
||||
&Event->Futex,
|
||||
Op,
|
||||
nullptr, // Timeout
|
||||
nullptr, // Addr
|
||||
0);
|
||||
}
|
||||
}
|
||||
|
||||
static void NotifyWorkFunc(CrossArchEvent *Event) {
|
||||
uint32_t Zero = 0;
|
||||
if (Event->Futex.compare_exchange_strong(Zero, 1)) {
|
||||
int Op = FUTEX_WAKE | FUTEX_PRIVATE_FLAG;
|
||||
syscall(SYS_futex,
|
||||
&Event->Futex,
|
||||
Op,
|
||||
nullptr, // Timeout
|
||||
nullptr, // Addr
|
||||
0);
|
||||
}
|
||||
}
|
||||
|
||||
//class CrossArchEvent final {
|
||||
// private:
|
||||
// /**
|
||||
// * @brief Literally just an atomic bool that we are using for this class
|
||||
// */
|
||||
// class Flag final {
|
||||
// public:
|
||||
// bool TestAndSet(bool SetValue = true) {
|
||||
// bool Expected = !SetValue;
|
||||
// return Value.compare_exchange_strong(Expected, SetValue);
|
||||
// }
|
||||
//
|
||||
// bool TestAndClear() {
|
||||
// return TestAndSet(false);
|
||||
// }
|
||||
//
|
||||
// bool Get() {
|
||||
// return Value.load();
|
||||
// }
|
||||
//
|
||||
// private:
|
||||
// std::atomic_bool Value {false};
|
||||
// };
|
||||
//
|
||||
// public:
|
||||
// ~CrossArchEvent() {
|
||||
// NotifyAll();
|
||||
// }
|
||||
// void NotifyOne() {
|
||||
// if (FlagObject.TestAndSet()) {
|
||||
// std::lock_guard<std::mutex> lk(MutexObject);
|
||||
// CondObject.notify_one();
|
||||
// }
|
||||
// }
|
||||
//
|
||||
// void NotifyAll() {
|
||||
// if (FlagObject.TestAndSet()) {
|
||||
// std::lock_guard<std::mutex> lk(MutexObject);
|
||||
// CondObject.notify_all();
|
||||
// }
|
||||
// }
|
||||
//
|
||||
// void Wait() {
|
||||
// // Have we signaled before we started waiting?
|
||||
// if (FlagObject.TestAndClear())
|
||||
// return;
|
||||
//
|
||||
// std::unique_lock<std::mutex> lk(MutexObject);
|
||||
// CondObject.wait(lk, [this]{ return FlagObject.TestAndClear(); });
|
||||
// }
|
||||
//
|
||||
// template<class Rep, class Period>
|
||||
// bool WaitFor(std::chrono::duration<Rep, Period> const& time) {
|
||||
// // Have we signaled before we started waiting?
|
||||
// if (FlagObject.TestAndClear())
|
||||
// return true;
|
||||
//
|
||||
// std::unique_lock<std::mutex> lk(MutexObject);
|
||||
// bool DidSignal = CondObject.wait_for(lk, time, [this]{ return FlagObject.TestAndClear(); });
|
||||
// return DidSignal;
|
||||
// }
|
||||
// void BusyWaitForWork() {
|
||||
// while (!FlagObject.Get());
|
||||
// }
|
||||
//
|
||||
// private:
|
||||
// std::mutex MutexObject;
|
||||
// std::condition_variable CondObject;
|
||||
// Flag FlagObject;
|
||||
//};
|
||||
|
||||
|
||||
@@ -13,4 +13,20 @@ struct callback_annotation_base {
|
||||
struct callback_stub : callback_annotation_base {};
|
||||
struct callback_guest : callback_annotation_base {};
|
||||
|
||||
struct type_annotation_base { bool prevent_multiple; };
|
||||
|
||||
// Pointers to types annotated with this will be passed through without change
|
||||
struct opaque_type : type_annotation_base {};
|
||||
|
||||
// Function parameter annotation.
|
||||
// Pointers are passed through to host (extending to 64-bit if needed) without modifying the pointee.
|
||||
// The type passed to Host will be guest_layout<pointee_type>*.
|
||||
struct ptr_passthrough {};
|
||||
|
||||
// Type / Function parameter annotation.
|
||||
// Assume objects of the given type are compatible across architectures,
|
||||
// even if the generator can't automatically prove this. For pointers, this refers to the pointee type.
|
||||
// NOTE: In contrast to opaque_type, this allows for non-pointer members with the annotated type to be repacked automatically.
|
||||
struct assume_compatible_data_layout : type_annotation_base {};
|
||||
|
||||
} // namespace fexgen
|
||||
@@ -10,7 +10,7 @@
|
||||
#ifdef __clang__
|
||||
#define THUNK_ABI __fastcall
|
||||
#else
|
||||
#define THUNK_ABI [[gnu::fastcall]]
|
||||
#define THUNK_ABI __attribute__((fastcall))
|
||||
#endif
|
||||
#endif
|
||||
|
||||
@@ -75,8 +75,8 @@ MAKE_THUNK(fex, allocate_host_trampoline_for_guest_function, "0x9b, 0xb2, 0xf4,
|
||||
|
||||
inline void LinkAddressToFunction(uintptr_t addr, uintptr_t target) {
|
||||
struct args_t {
|
||||
uintptr_t original_callee;
|
||||
uintptr_t target_addr; // Function to call when branching to replaced_addr
|
||||
uint64_t original_callee;
|
||||
uint64_t target_addr; // Function to call when branching to replaced_addr
|
||||
};
|
||||
args_t args = { addr, target };
|
||||
fexthunks_fex_link_address_to_function(&args);
|
||||
@@ -126,14 +126,14 @@ inline Result CallHostFunction(Args... args) {
|
||||
// Use mm0 to pass in host_addr (chosen to avoid conflicts with vectorcall).
|
||||
// Note this register overlaps the x87 st(0) register (used to return float values),
|
||||
// so applications that expect this register to be preserved could run into problems.
|
||||
register uintptr_t host_addr asm ("mm0");
|
||||
asm volatile("" : "=r" (host_addr));
|
||||
uintptr_t host_addr; \
|
||||
asm volatile("movd %%mm0, %0" : "=r" (host_addr));
|
||||
#endif
|
||||
#else
|
||||
uintptr_t host_addr = 0;
|
||||
#endif
|
||||
|
||||
PackedArguments<Result, Args..., uintptr_t> packed_args = {
|
||||
PackedArguments<Result, Args..., uint64_t> packed_args = {
|
||||
args...,
|
||||
host_addr
|
||||
// Return value not explicitly initialized since an initializer would fail to compile for the void case
|
||||
@@ -162,20 +162,20 @@ inline void MakeHostFunctionGuestCallable(THUNK_ABI Result (*host_func)(Args...)
|
||||
}
|
||||
|
||||
template<typename Target>
|
||||
inline Target *AllocateHostTrampolineForGuestFunction(void (*GuestUnpacker)(uintptr_t, void*), Target *GuestTarget) {
|
||||
inline Target* AllocateHostTrampolineForGuestFunction(void THUNK_ABI (*GuestUnpacker)(uintptr_t, void*), Target *GuestTarget) {
|
||||
if (!GuestTarget) {
|
||||
return nullptr;
|
||||
return 0;
|
||||
}
|
||||
|
||||
struct {
|
||||
uintptr_t GuestUnpacker;
|
||||
uintptr_t GuestTarget;
|
||||
uintptr_t rv;
|
||||
uint64_t GuestUnpacker;
|
||||
uint64_t GuestTarget;
|
||||
uint64_t rv;
|
||||
} argsrv = { (uintptr_t)GuestUnpacker, (uintptr_t)GuestTarget };
|
||||
|
||||
fexthunks_fex_allocate_host_trampoline_for_guest_function((void*)&argsrv);
|
||||
|
||||
return (Target *)argsrv.rv;
|
||||
return (Target*)argsrv.rv;
|
||||
}
|
||||
|
||||
template<typename F>
|
||||
@@ -183,7 +183,7 @@ struct CallbackUnpack;
|
||||
|
||||
template<typename Result, typename... Args>
|
||||
struct CallbackUnpack<Result(Args...)> {
|
||||
static void Unpack(uintptr_t cb, void* argsv) {
|
||||
static void THUNK_ABI Unpack(uintptr_t cb, void* argsv) {
|
||||
using fn_t = Result(Args...);
|
||||
auto callback = reinterpret_cast<fn_t*>(cb);
|
||||
auto args = reinterpret_cast<PackedArguments<Result, Args...>*>(argsv);
|
||||
|
||||
@@ -103,6 +103,17 @@ struct GuestcallInfo {
|
||||
asm volatile("mov %0, x11" : "=r" (target_variable))
|
||||
#endif
|
||||
|
||||
struct ParameterAnnotations {
|
||||
bool is_passthrough = false;
|
||||
bool assume_compatible = false;
|
||||
};
|
||||
|
||||
// Placeholder type to indicate the given data is in guest-layout
|
||||
template<typename T>
|
||||
struct guest_layout {
|
||||
T data;
|
||||
};
|
||||
|
||||
template<typename>
|
||||
struct CallbackUnpack;
|
||||
|
||||
@@ -120,9 +131,28 @@ struct CallbackUnpack<Result(Args...)> {
|
||||
return packed_args.rv;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
static void ForIndirectCall(void* argsv) {
|
||||
auto args = reinterpret_cast<PackedArguments<Result, Args..., uintptr_t>*>(argsv);
|
||||
template<ParameterAnnotations Annotation, typename T>
|
||||
auto Projection(guest_layout<T>& data) {
|
||||
if constexpr (Annotation.is_passthrough) {
|
||||
return data;
|
||||
} else {
|
||||
return data.data;
|
||||
}
|
||||
}
|
||||
|
||||
template<typename>
|
||||
struct GuestWrapperForHostFunction;
|
||||
|
||||
template<typename Result, typename... Args>
|
||||
struct GuestWrapperForHostFunction<Result(Args...)> {
|
||||
// Host functions called from Guest
|
||||
template<ParameterAnnotations... Annotations>
|
||||
static void Call(void* argsv) {
|
||||
static_assert(sizeof...(Annotations) == sizeof...(Args));
|
||||
|
||||
auto args = reinterpret_cast<PackedArguments<Result, guest_layout<Args>..., uintptr_t>*>(argsv);
|
||||
constexpr auto CBIndex = sizeof...(Args);
|
||||
uintptr_t cb;
|
||||
static_assert(CBIndex <= 18 || CBIndex == 23);
|
||||
@@ -168,8 +198,15 @@ struct CallbackUnpack<Result(Args...)> {
|
||||
cb = args->a23;
|
||||
}
|
||||
|
||||
auto callback = reinterpret_cast<Result(*)(Args..., uintptr_t)>(cb);
|
||||
Invoke(callback, *args);
|
||||
// This is almost the same type as "Result func(Args..., uintptr_t)", but
|
||||
// individual parameters annotated as passthrough are replaced by guest_layout<GuestArgs>
|
||||
auto callback = reinterpret_cast<Result(*)(std::conditional_t<Annotations.is_passthrough, guest_layout<Args>, Args>..., uintptr_t)>(cb);
|
||||
|
||||
auto f = [&callback](guest_layout<Args>... args, uintptr_t target) -> Result {
|
||||
// Fold over each of Annotations, Args, and args. This will match up the elements in triplets.
|
||||
return callback(Projection<Annotations, Args>(args)..., target);
|
||||
};
|
||||
Invoke(f, *args);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -113,8 +113,8 @@ T&& operator,(T&& t, Regularize) {
|
||||
return std::forward<T>(t);
|
||||
}
|
||||
|
||||
template<typename Result, typename... Args>
|
||||
void Invoke(Result(*func)(Args...), PackedArguments<Result, Args...>& args) {
|
||||
template<typename Result, typename... Args, typename Func>
|
||||
void Invoke(Func&& func, PackedArguments<Result, Args...>& args) requires(std::is_invocable_r_v<Result, Func, Args...>) {
|
||||
constexpr auto NumArgs = sizeof...(Args);
|
||||
static_assert(NumArgs <= 19 || NumArgs == 24);
|
||||
|
||||
|
||||
@@ -7,6 +7,10 @@ struct fex_gen_config {
|
||||
unsigned version = 1;
|
||||
};
|
||||
|
||||
// Function, parameter index, parameter type [optional]
|
||||
template<auto, int, typename = void>
|
||||
struct fex_gen_param {};
|
||||
|
||||
template<> struct fex_gen_config<eglBindAPI> {};
|
||||
template<> struct fex_gen_config<eglChooseConfig> {};
|
||||
template<> struct fex_gen_config<eglDestroyContext> {};
|
||||
@@ -23,4 +27,7 @@ template<> struct fex_gen_config<eglCreateWindowSurface> {};
|
||||
template<> struct fex_gen_config<eglGetCurrentContext> {};
|
||||
template<> struct fex_gen_config<eglGetCurrentDisplay> {};
|
||||
template<> struct fex_gen_config<eglGetCurrentSurface> {};
|
||||
|
||||
// EGLNativeDisplayType is a pointer to opaque data (wl_display/(X)Display/...)
|
||||
template<> struct fex_gen_config<eglGetDisplay> {};
|
||||
template<> struct fex_gen_param<eglGetDisplay, 0, EGLNativeDisplayType> : fexgen::assume_compatible_data_layout {};
|
||||
@@ -11,6 +11,8 @@
|
||||
#undef GL_ARB_viewport_array
|
||||
#include "glcorearb.h"
|
||||
|
||||
#include <type_traits>
|
||||
|
||||
template<auto>
|
||||
struct fex_gen_config {
|
||||
unsigned version = 1;
|
||||
@@ -18,6 +20,25 @@ struct fex_gen_config {
|
||||
|
||||
template<> struct fex_gen_config<glXGetProcAddress> : fexgen::custom_guest_entrypoint, fexgen::returns_guest_pointer {};
|
||||
|
||||
template<typename>
|
||||
struct fex_gen_type {};
|
||||
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<GLXContext>> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<GLXFBConfig>> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<GLsync>> : fexgen::opaque_type {};
|
||||
|
||||
// NOTE: These should be opaque, but actually aren't because the respective libraries aren't thunked
|
||||
template<> struct fex_gen_type<_cl_context> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_cl_event> : fexgen::opaque_type {};
|
||||
|
||||
// Opaque for the purpose of libGL
|
||||
template<> struct fex_gen_type<_XDisplay> : fexgen::opaque_type {};
|
||||
|
||||
#ifndef IS_32BIT_THUNK
|
||||
// TODO: These are largely compatible, *but* contain function pointer members that need adjustment!
|
||||
template<> struct fex_gen_type<XExtData> : fexgen::assume_compatible_data_layout {};
|
||||
#endif
|
||||
|
||||
// Symbols queryable through glXGetProcAddr
|
||||
namespace internal {
|
||||
template<auto>
|
||||
|
||||
@@ -14,6 +14,8 @@ extern "C" {
|
||||
#include <X11/Xproto.h>
|
||||
#include <X11/XKBlib.h>
|
||||
|
||||
#include <X11/extensions/extutil.h>
|
||||
|
||||
// Include Xlibint.h and undefine some of its macros that clash with the standard library
|
||||
#include <X11/Xlibint.h>
|
||||
#undef min
|
||||
|
||||
@@ -20,12 +20,20 @@ extern "C" {
|
||||
#undef min
|
||||
#undef max
|
||||
|
||||
#define XTRANS_SEND_FDS 1
|
||||
#include <X11/Xtrans/Xtransint.h>
|
||||
|
||||
#include <X11/Xutil.h>
|
||||
#include <X11/Xregion.h>
|
||||
#include <X11/Xresource.h>
|
||||
|
||||
#include <X11/Xproto.h>
|
||||
|
||||
#include <X11/extensions/extutil.h>
|
||||
#include <X11/extensions/XKBstr.h>
|
||||
#include <X11/extensions/XKBgeom.h>
|
||||
|
||||
#include <X11/Xtrans/Xtransint.h>
|
||||
}
|
||||
|
||||
#include "common/Host.h"
|
||||
|
||||
@@ -10,11 +10,15 @@ extern "C" {
|
||||
|
||||
#include <X11/ImUtil.h>
|
||||
|
||||
#include <X11/extensions/extutil.h>
|
||||
#include <X11/extensions/Xext.h>
|
||||
#include <X11/extensions/XKBgeom.h>
|
||||
|
||||
#include <X11/Xlibint.h>
|
||||
#include <X11/XKBlib.h>
|
||||
|
||||
#include <X11/Xtrans/Xtransint.h>
|
||||
|
||||
#include <X11/Xutil.h>
|
||||
}
|
||||
|
||||
@@ -28,6 +32,10 @@ struct fex_gen_config {
|
||||
template<typename>
|
||||
struct fex_gen_type {};
|
||||
|
||||
// Function, parameter index, parameter type [optional]
|
||||
template<auto, int, typename = void>
|
||||
struct fex_gen_param {};
|
||||
|
||||
template<> struct fex_gen_type<XID(Display*)> {}; // XDisplay::resource_alloc
|
||||
|
||||
// NOTE: only indirect calls to this are allowed
|
||||
@@ -54,6 +62,60 @@ template<> struct fex_gen_type<unsigned long(XImage*, int, int)> {}; // XIm
|
||||
template<> struct fex_gen_type<int(XImage*, int, int, unsigned long)> {}; // XImage::f.put_pixel
|
||||
template<> struct fex_gen_type<int(XImage*, long)> {}; // XImage::f.add_pixel
|
||||
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<XIC>> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<XIM>> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<XOC>> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<XOM>> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XrmHashBucketRec> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XkbInfoRec> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XContextDB> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XDisplayAtoms> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XLockInfo> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XIMFilter> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XKeytrans> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_X11XCBPrivate> : fexgen::opaque_type {};
|
||||
|
||||
|
||||
#ifndef IS_32BIT_THUNK
|
||||
// This has a public definition but is used as an opaque type in most APIs
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<Region>> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// Union types
|
||||
template<> struct fex_gen_type<XEvent> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<xEvent> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<XkbDoodadRec> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<xReply> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// Union type of all sorts of different X objects... Further, XEHeadOfExtensionList casts this to XExtData**.
|
||||
// This is likely highly unsafe to assume compatible.
|
||||
template<> struct fex_gen_type<XEDataObject> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// Linked-list types
|
||||
template<> struct fex_gen_type<_XtransConnFd> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XkbDeviceLedChanges> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XQEvent> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// Contains nontrivial circular pointer relationships
|
||||
template<> struct fex_gen_type<XkbOverlayRec> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// TODO: These are largely compatible, *but* contain function pointer members that need adjustment!
|
||||
template<> struct fex_gen_type<_XConnWatchInfo> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<XExtData> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XLockPtrs> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XFreeFuncRec> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XExtension> : fexgen::assume_compatible_data_layout {}; // Also contains linked-list pointers
|
||||
template<> struct fex_gen_type<_XConnectionInfo> : fexgen::assume_compatible_data_layout {}; // Also contains linked-list pointers
|
||||
template<> struct fex_gen_type<_XInternalAsync> : fexgen::assume_compatible_data_layout {}; // Also contains linked-list pointers
|
||||
template<> struct fex_gen_type<_XDisplay> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// TODO: This contains a nested struct type of function pointer members
|
||||
template<> struct fex_gen_type<XImage> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<XExtensionHooks> : fexgen::assume_compatible_data_layout {};
|
||||
#endif
|
||||
|
||||
// Union type (each member is defined in terms of char members)
|
||||
template<> struct fex_gen_type<XkbAction> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// Xlibint
|
||||
template<> struct fex_gen_config<_XGetRequest> {};
|
||||
template<> struct fex_gen_config<_XFlushGCCache> {};
|
||||
|
||||
@@ -35,6 +35,8 @@ extern "C" {
|
||||
#include <X11/extensions/syncconst.h>
|
||||
#include <X11/extensions/syncproto.h>
|
||||
//#include <X11/extensions/XTest.h>
|
||||
#undef min
|
||||
#undef max
|
||||
|
||||
|
||||
#include "common/Guest.h"
|
||||
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include <X11/Xlib.h>
|
||||
#include <X11/Xlibint.h>
|
||||
#include <X11/Xregion.h>
|
||||
#include <X11/Xutil.h>
|
||||
#include <X11/Xproto.h>
|
||||
#include <X11/extensions/Xext.h>
|
||||
@@ -38,6 +39,12 @@ extern "C" {
|
||||
#include <X11/extensions/syncproto.h>
|
||||
//#include <X11/extensions/XTest.h>
|
||||
|
||||
#define XTRANS_SEND_FDS 1
|
||||
#include <X11/Xtrans/Xtransint.h>
|
||||
|
||||
#undef min
|
||||
#undef max
|
||||
|
||||
#include "common/Host.h"
|
||||
#include <dlfcn.h>
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
#include <X11/Xlib.h>
|
||||
#include <X11/Xlibint.h>
|
||||
#include <X11/Xregion.h>
|
||||
#include <X11/extensions/Xext.h>
|
||||
#include <X11/extensions/dpms.h>
|
||||
#include <X11/extensions/XEVI.h>
|
||||
@@ -16,12 +17,51 @@
|
||||
extern "C" {
|
||||
#include <X11/extensions/extutil.h>
|
||||
}
|
||||
#include <X11/Xtrans/Xtransint.h>
|
||||
|
||||
template<auto>
|
||||
struct fex_gen_config {
|
||||
unsigned version = 6;
|
||||
};
|
||||
|
||||
template<typename>
|
||||
struct fex_gen_type {};
|
||||
|
||||
template<> struct fex_gen_type<_X11XCBPrivate> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XContextDB> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XDisplayAtoms> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XIMFilter> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XkbInfoRec> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XKeytrans> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XLockInfo> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<_XrmHashBucketRec> : fexgen::opaque_type {};
|
||||
|
||||
#ifndef IS_32BIT_THUNK
|
||||
// Union types
|
||||
template<> struct fex_gen_type<XEvent> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<xEvent> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// Linked-list types
|
||||
template<> struct fex_gen_type<_XtransConnFd> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XQEvent> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// TODO: These are largely compatible, *but* contain function pointer members that need adjustment!
|
||||
template<> struct fex_gen_type<_XConnectionInfo> : fexgen::assume_compatible_data_layout {}; // Also contains linked-list pointers
|
||||
template<> struct fex_gen_type<_XConnWatchInfo> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XDisplay> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<XExtData> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<XExtDisplayInfo> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XExtension> : fexgen::assume_compatible_data_layout {}; // Also contains linked-list pointers
|
||||
template<> struct fex_gen_type<_XInternalAsync> : fexgen::assume_compatible_data_layout {}; // Also contains linked-list pointers
|
||||
|
||||
// Contains nontrivial circular pointer relationships
|
||||
template<> struct fex_gen_type<Screen> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// TODO: This contains a nested struct type of function pointer members
|
||||
template<> struct fex_gen_type<XImage> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<XExtensionHooks> : fexgen::assume_compatible_data_layout {};
|
||||
#endif
|
||||
|
||||
template<> struct fex_gen_config<DPMSCapable> {};
|
||||
template<> struct fex_gen_config<DPMSDisable> {};
|
||||
template<> struct fex_gen_config<DPMSEnable> {};
|
||||
|
||||
@@ -6,9 +6,14 @@ $end_info$
|
||||
|
||||
#include <stdio.h>
|
||||
|
||||
extern "C" {
|
||||
#include <X11/Xlib.h>
|
||||
#include <X11/Xlibint.h>
|
||||
#include <X11/Xutil.h>
|
||||
#include <X11/extensions/Xfixes.h>
|
||||
}
|
||||
#undef min
|
||||
#undef max
|
||||
|
||||
#include "common/Host.h"
|
||||
#include <dlfcn.h>
|
||||
|
||||
@@ -1,12 +1,24 @@
|
||||
#include <common/GeneratorInterface.h>
|
||||
|
||||
extern "C" {
|
||||
#include <X11/extensions/Xfixes.h>
|
||||
#include <X11/Xlibint.h>
|
||||
}
|
||||
|
||||
template<auto>
|
||||
struct fex_gen_config {
|
||||
unsigned version = 3;
|
||||
};
|
||||
|
||||
template<typename>
|
||||
struct fex_gen_type {};
|
||||
|
||||
#ifndef IS_32BIT_THUNK
|
||||
// TODO: These are largely compatible, *but* contain function pointer members that need adjustment!
|
||||
template<> struct fex_gen_type<XExtData> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XDisplay> : fexgen::assume_compatible_data_layout {};
|
||||
#endif
|
||||
|
||||
template<> struct fex_gen_config<XFixesGetCursorName> {};
|
||||
template<> struct fex_gen_config<XFixesQueryExtension> {};
|
||||
template<> struct fex_gen_config<XFixesQueryVersion> {};
|
||||
|
||||
@@ -6,9 +6,15 @@ $end_info$
|
||||
|
||||
#include <stdio.h>
|
||||
|
||||
extern "C" {
|
||||
#include <X11/Xlib.h>
|
||||
#include <X11/Xlibint.h>
|
||||
#include <X11/Xregion.h>
|
||||
#include <X11/Xutil.h>
|
||||
#include <X11/extensions/Xrender.h>
|
||||
}
|
||||
#undef min
|
||||
#undef max
|
||||
|
||||
#include "common/Host.h"
|
||||
#include <dlfcn.h>
|
||||
|
||||
@@ -2,11 +2,28 @@
|
||||
|
||||
#include <X11/extensions/Xrender.h>
|
||||
|
||||
#include <type_traits>
|
||||
|
||||
template<auto>
|
||||
struct fex_gen_config {
|
||||
unsigned version = 1;
|
||||
};
|
||||
|
||||
template<typename>
|
||||
struct fex_gen_type {};
|
||||
|
||||
// Struct with multi-dimensional array member. Compatible data layout across all architectures
|
||||
template<> struct fex_gen_type<_XTransform> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
#ifndef IS_32BIT_THUNK
|
||||
// This has a public definition but is used as an opaque type in most APIs
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<Region>> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// TODO: These are largely compatible, *but* contain function pointer members that need adjustment!
|
||||
template<> struct fex_gen_type<XExtData> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<_XDisplay> : fexgen::assume_compatible_data_layout {};
|
||||
#endif
|
||||
|
||||
template<> struct fex_gen_config<XRenderCreateAnimCursor> {};
|
||||
template<> struct fex_gen_config<XRenderCreateCursor> {};
|
||||
template<> struct fex_gen_config<XRenderCreateGlyphSet> {};
|
||||
|
||||
@@ -3,11 +3,99 @@
|
||||
#include <alsa/asoundlib.h>
|
||||
#include <alsa/version.h>
|
||||
|
||||
#include <type_traits>
|
||||
|
||||
template<auto>
|
||||
struct fex_gen_config {
|
||||
unsigned version = 2;
|
||||
};
|
||||
|
||||
// Function, parameter index, parameter type [optional]
|
||||
template<auto, int, typename = void>
|
||||
struct fex_gen_param {};
|
||||
|
||||
template<typename>
|
||||
struct fex_gen_type {};
|
||||
|
||||
template<> struct fex_gen_type<snd_pcm_status_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_shm_area> : fexgen::opaque_type {};
|
||||
|
||||
template<> struct fex_gen_type<snd_async_handler_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<std::remove_pointer_t<snd_config_iterator_t>> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_config_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_config_update_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_ctl_card_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_ctl_elem_id_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_ctl_elem_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_ctl_elem_list_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_ctl_elem_value_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_ctl_event_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_ctl_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_devname_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_hctl_elem_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_hctl_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_hwdep_dsp_image_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_hwdep_dsp_status_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_hwdep_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_hwdep_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_input_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_midi_event_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_mixer_class_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_mixer_elem_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_mixer_selem_id_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_mixer_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_output_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_pcm_access_mask_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_pcm_format_mask_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_pcm_hook_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_pcm_hw_params_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_pcm_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_pcm_scope_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_pcm_subformat_mask_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_pcm_sw_params_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_pcm_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_rawmidi_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_rawmidi_params_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_rawmidi_status_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_rawmidi_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_sctl_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_client_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_client_pool_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_ev_ext_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_port_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_port_subscribe_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_query_subscribe_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_queue_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_queue_status_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_queue_tempo_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_queue_timer_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_remove_events_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_system_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_seq_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timer_ginfo_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timer_gparams_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timer_gstatus_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timer_id_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timer_info_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timer_params_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timer_query_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timer_status_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timer_t> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<snd_timestamp_t> : fexgen::opaque_type {};
|
||||
|
||||
template<> struct fex_gen_type<FILE> : fexgen::opaque_type {};
|
||||
|
||||
// Union types with compatible data layout
|
||||
template<> struct fex_gen_type<snd_pcm_sync_id_t> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<snd_seq_timestamp> : fexgen::assume_compatible_data_layout {};
|
||||
// Has anonymous union member
|
||||
template<> struct fex_gen_type<snd_seq_event> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
#ifndef IS_32BIT_THUNK
|
||||
// TODO: Convert vtable
|
||||
template<> struct fex_gen_type<snd_pcm_scope_ops_t> : fexgen::assume_compatible_data_layout {};
|
||||
#endif
|
||||
|
||||
template<> struct fex_gen_config<snd_asoundlib_version> {};
|
||||
#if SND_LIB_VERSION < ((1 << 16) | (2 << 8) | (6))
|
||||
// Exists on 1.2.6
|
||||
|
||||
@@ -7,6 +7,21 @@ struct fex_gen_config {
|
||||
unsigned version = 2;
|
||||
};
|
||||
|
||||
template<typename>
|
||||
struct fex_gen_type {};
|
||||
|
||||
#ifndef IS_32BIT_THUNK
|
||||
// Union types with compatible data layout
|
||||
template<> struct fex_gen_type<drmDevice> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// Anonymous sub-structs
|
||||
template<> struct fex_gen_type<drmStatsT> : fexgen::assume_compatible_data_layout {};
|
||||
|
||||
// TODO: Convert vtable
|
||||
template<> struct fex_gen_type<drmServerInfo> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<drmEventContext> : fexgen::assume_compatible_data_layout {};
|
||||
#endif
|
||||
|
||||
size_t FEX_usable_size(void*);
|
||||
void FEX_free_on_host(void*);
|
||||
|
||||
|
||||
@@ -10,4 +10,22 @@ extern "C" {
|
||||
|
||||
uint32_t GetDoubledValue(uint32_t);
|
||||
|
||||
|
||||
/// Interfaces used to test opaque_type and assume_compatible_data_layout annotations
|
||||
|
||||
struct OpaqueType;
|
||||
|
||||
OpaqueType* MakeOpaqueType(uint32_t data);
|
||||
uint32_t ReadOpaqueTypeData(OpaqueType*);
|
||||
void DestroyOpaqueType(OpaqueType*);
|
||||
|
||||
union UnionType {
|
||||
uint32_t a;
|
||||
int32_t b;
|
||||
uint8_t c[4];
|
||||
};
|
||||
|
||||
UnionType MakeUnionType(uint8_t a, uint8_t b, uint8_t c, uint8_t d);
|
||||
uint32_t GetUnionTypeA(UnionType*);
|
||||
|
||||
}
|
||||
@@ -6,4 +6,28 @@ uint32_t GetDoubledValue(uint32_t input) {
|
||||
return 2 * input;
|
||||
}
|
||||
|
||||
struct OpaqueType {
|
||||
uint32_t data;
|
||||
};
|
||||
|
||||
OpaqueType* MakeOpaqueType(uint32_t data) {
|
||||
return new OpaqueType { data };
|
||||
}
|
||||
|
||||
uint32_t ReadOpaqueTypeData(OpaqueType* value) {
|
||||
return value->data;
|
||||
}
|
||||
|
||||
void DestroyOpaqueType(OpaqueType* value) {
|
||||
delete value;
|
||||
}
|
||||
|
||||
UnionType MakeUnionType(uint8_t a, uint8_t b, uint8_t c, uint8_t d) {
|
||||
return UnionType { .c = { a, b, c, d } };
|
||||
}
|
||||
|
||||
uint32_t GetUnionTypeA(UnionType* value) {
|
||||
return value->a;
|
||||
}
|
||||
|
||||
} // extern "C"
|
||||
@@ -12,3 +12,12 @@ template<auto, int, typename>
|
||||
struct fex_gen_param {};
|
||||
|
||||
template<> struct fex_gen_config<GetDoubledValue> {};
|
||||
|
||||
template<> struct fex_gen_type<OpaqueType> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_config<MakeOpaqueType> {};
|
||||
template<> struct fex_gen_config<ReadOpaqueTypeData> {};
|
||||
template<> struct fex_gen_config<DestroyOpaqueType> {};
|
||||
|
||||
template<> struct fex_gen_type<UnionType> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<MakeUnionType> {};
|
||||
template<> struct fex_gen_config<GetUnionTypeA> {};
|
||||
@@ -4,6 +4,8 @@ tags: thunklibs|Vulkan
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#define VK_USE_64_BIT_PTR_DEFINES 0
|
||||
|
||||
#define VK_USE_PLATFORM_XLIB_XRANDR_EXT
|
||||
#define VK_USE_PLATFORM_XLIB_KHR
|
||||
#define VK_USE_PLATFORM_XCB_KHR
|
||||
|
||||
@@ -4,6 +4,8 @@ tags: thunklibs|Vulkan
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#define VK_USE_64_BIT_PTR_DEFINES 0
|
||||
|
||||
#define VK_USE_PLATFORM_XLIB_XRANDR_EXT
|
||||
#define VK_USE_PLATFORM_XLIB_KHR
|
||||
#define VK_USE_PLATFORM_XCB_KHR
|
||||
@@ -60,7 +62,7 @@ static VkBool32 DummyVkDebugReportCallback(VkDebugReportFlagsEXT, VkDebugReportO
|
||||
return VK_FALSE;
|
||||
}
|
||||
|
||||
static VkResult FEXFN_IMPL(vkCreateInstance)(const VkInstanceCreateInfo* a_0, const VkAllocationCallbacks* a_1, VkInstance* a_2) {
|
||||
static VkResult FEXFN_IMPL(vkCreateInstance)(const VkInstanceCreateInfo* a_0, const VkAllocationCallbacks* a_1, guest_layout<VkInstance*> a_2) {
|
||||
const VkInstanceCreateInfo* vk_struct_base = a_0;
|
||||
for (const VkBaseInStructure* vk_struct = reinterpret_cast<const VkBaseInStructure*>(vk_struct_base); vk_struct->pNext; vk_struct = vk_struct->pNext) {
|
||||
// Override guest callbacks used for VK_EXT_debug_report
|
||||
@@ -74,11 +76,11 @@ static VkResult FEXFN_IMPL(vkCreateInstance)(const VkInstanceCreateInfo* a_0, co
|
||||
}
|
||||
}
|
||||
|
||||
return LDR_PTR(vkCreateInstance)(vk_struct_base, nullptr, a_2);
|
||||
return LDR_PTR(vkCreateInstance)(vk_struct_base, nullptr, a_2.data);
|
||||
}
|
||||
|
||||
static VkResult FEXFN_IMPL(vkCreateDevice)(VkPhysicalDevice a_0, const VkDeviceCreateInfo* a_1, const VkAllocationCallbacks* a_2, VkDevice* a_3){
|
||||
return LDR_PTR(vkCreateDevice)(a_0, a_1, nullptr, a_3);
|
||||
static VkResult FEXFN_IMPL(vkCreateDevice)(VkPhysicalDevice a_0, const VkDeviceCreateInfo* a_1, const VkAllocationCallbacks* a_2, guest_layout<VkDevice*> a_3){
|
||||
return LDR_PTR(vkCreateDevice)(a_0, a_1, nullptr, a_3.data);
|
||||
}
|
||||
|
||||
static VkResult FEXFN_IMPL(vkAllocateMemory)(VkDevice a_0, const VkMemoryAllocateInfo* a_1, const VkAllocationCallbacks* a_2, VkDeviceMemory* a_3){
|
||||
@@ -91,8 +93,8 @@ static void FEXFN_IMPL(vkFreeMemory)(VkDevice a_0, VkDeviceMemory a_1, const VkA
|
||||
LDR_PTR(vkFreeMemory)(a_0, a_1, nullptr);
|
||||
}
|
||||
|
||||
static VkResult FEXFN_IMPL(vkCreateDebugReportCallbackEXT)(VkInstance a_0, const VkDebugReportCallbackCreateInfoEXT* a_1, const VkAllocationCallbacks* a_2, VkDebugReportCallbackEXT* a_3) {
|
||||
VkDebugReportCallbackCreateInfoEXT overridden_callback = *a_1;
|
||||
static VkResult FEXFN_IMPL(vkCreateDebugReportCallbackEXT)(VkInstance a_0, guest_layout<const VkDebugReportCallbackCreateInfoEXT*> a_1, const VkAllocationCallbacks* a_2, VkDebugReportCallbackEXT* a_3) {
|
||||
VkDebugReportCallbackCreateInfoEXT overridden_callback = *a_1.data;
|
||||
overridden_callback.pfnCallback = DummyVkDebugReportCallback;
|
||||
(void*&)LDR_PTR(vkCreateDebugReportCallbackEXT) = (void*)LDR_PTR(vkGetInstanceProcAddr)(a_0, "vkCreateDebugReportCallbackEXT");
|
||||
return LDR_PTR(vkCreateDebugReportCallbackEXT)(a_0, &overridden_callback, nullptr, a_3);
|
||||
@@ -103,6 +105,21 @@ static void FEXFN_IMPL(vkDestroyDebugReportCallbackEXT)(VkInstance a_0, VkDebugR
|
||||
LDR_PTR(vkDestroyDebugReportCallbackEXT)(a_0, a_1, nullptr);
|
||||
}
|
||||
|
||||
extern "C" VkBool32 DummyVkDebugUtilsMessengerCallback(
|
||||
VkDebugUtilsMessageSeverityFlagBitsEXT, VkDebugUtilsMessageTypeFlagsEXT,
|
||||
const VkDebugUtilsMessengerCallbackDataEXT*, void*) {
|
||||
return VK_FALSE;
|
||||
}
|
||||
|
||||
static VkResult FEXFN_IMPL(vkCreateDebugUtilsMessengerEXT)(
|
||||
VkInstance_T* a_0, guest_layout<const VkDebugUtilsMessengerCreateInfoEXT*> a_1,
|
||||
const VkAllocationCallbacks* a_2, VkDebugUtilsMessengerEXT* a_3) {
|
||||
VkDebugUtilsMessengerCreateInfoEXT overridden_callback = *a_1.data;
|
||||
overridden_callback.pfnUserCallback = DummyVkDebugUtilsMessengerCallback;
|
||||
(void*&)LDR_PTR(vkCreateDebugUtilsMessengerEXT) = (void*)LDR_PTR(vkGetInstanceProcAddr)(a_0, "vkCreateDebugUtilsMessengerEXT");
|
||||
return LDR_PTR(vkCreateDebugUtilsMessengerEXT)(a_0, &overridden_callback, nullptr, a_3);
|
||||
}
|
||||
|
||||
static PFN_vkVoidFunction FEXFN_IMPL(vkGetDeviceProcAddr)(VkDevice a_0, const char* a_1) {
|
||||
// Just return the host facing function pointer
|
||||
// The guest will handle mapping if this exists
|
||||
|
||||
@@ -1,10 +1,19 @@
|
||||
#include <common/GeneratorInterface.h>
|
||||
|
||||
#include <type_traits>
|
||||
|
||||
template<auto>
|
||||
struct fex_gen_config {
|
||||
unsigned version = 1;
|
||||
};
|
||||
|
||||
// Some of Vulkan's handle types are so-called "non-dispatchable handles".
|
||||
// On 64-bit, these are defined as dedicated types by default, which makes
|
||||
// annotating these handle types unnecessarily complicated. Instead, setting
|
||||
// the following define will make the Vulkan headers alias all handle types
|
||||
// to uint64_t.
|
||||
#define VK_USE_64_BIT_PTR_DEFINES 0
|
||||
|
||||
#define VK_USE_PLATFORM_XLIB_XRANDR_EXT
|
||||
#define VK_USE_PLATFORM_XLIB_KHR
|
||||
#define VK_USE_PLATFORM_XCB_KHR
|
||||
@@ -14,25 +23,75 @@ struct fex_gen_config {
|
||||
template<> struct fex_gen_config<vkGetDeviceProcAddr> : fexgen::custom_host_impl, fexgen::custom_guest_entrypoint, fexgen::returns_guest_pointer {};
|
||||
template<> struct fex_gen_config<vkGetInstanceProcAddr> : fexgen::custom_host_impl, fexgen::custom_guest_entrypoint, fexgen::returns_guest_pointer {};
|
||||
|
||||
template<typename>
|
||||
struct fex_gen_type {};
|
||||
|
||||
// So-called "dispatchable" handles are represented as opaque pointers.
|
||||
// In addition to marking them as such, API functions that create these objects
|
||||
// need special care since they wrap these handles in another pointer, which
|
||||
// the thunk generator can't automatically handle.
|
||||
//
|
||||
// So-called "non-dispatchable" handles don't need this extra treatment, since
|
||||
// they are uint64_t IDs on both 32-bit and 64-bit systems.
|
||||
template<> struct fex_gen_type<VkCommandBuffer_T> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<VkDevice_T> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<VkInstance_T> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<VkPhysicalDevice_T> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<VkQueue_T> : fexgen::opaque_type {};
|
||||
|
||||
// Mark union types with compatible layout as such
|
||||
// TODO: These may still have different alignment requirements!
|
||||
template<> struct fex_gen_type<VkClearValue> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<VkClearColorValue> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<VkPipelineExecutableStatisticValueKHR> : fexgen::assume_compatible_data_layout {};
|
||||
#ifndef IS_32BIT_THUNK
|
||||
template<> struct fex_gen_type<VkAccelerationStructureGeometryDataKHR> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<VkDescriptorDataEXT> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<VkDeviceOrHostAddressKHR> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<VkDeviceOrHostAddressConstKHR> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_type<VkPerformanceValueDataINTEL> : fexgen::assume_compatible_data_layout {};
|
||||
#endif
|
||||
|
||||
#ifndef IS_32BIT_THUNK
|
||||
// The pNext member of this is a pointer to another VkBaseOutStructure, so data layout compatibility can't be inferred automatically
|
||||
template<> struct fex_gen_type<VkBaseOutStructure> : fexgen::assume_compatible_data_layout {};
|
||||
#endif
|
||||
|
||||
|
||||
// TODO: Should not be opaque, but it's usually NULL anyway. Supporting the contained function pointers will need more work.
|
||||
template<> struct fex_gen_type<VkAllocationCallbacks> : fexgen::opaque_type {};
|
||||
|
||||
// X11 interop
|
||||
template<> struct fex_gen_type<Display> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<xcb_connection_t> : fexgen::opaque_type {};
|
||||
|
||||
// Wayland interop
|
||||
template<> struct fex_gen_type<wl_display> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<wl_surface> : fexgen::opaque_type {};
|
||||
|
||||
namespace internal {
|
||||
|
||||
// Function, parameter index, parameter type [optional]
|
||||
template<auto, int, typename = void>
|
||||
struct fex_gen_param {};
|
||||
|
||||
template<auto>
|
||||
struct fex_gen_config : fexgen::generate_guest_symtable, fexgen::indirect_guest_calls {
|
||||
};
|
||||
|
||||
template<> struct fex_gen_config<vkCreateInstance> : fexgen::custom_host_impl {};
|
||||
template<> struct fex_gen_param<vkCreateInstance, 2, VkInstance*> : fexgen::ptr_passthrough {};
|
||||
template<> struct fex_gen_config<vkDestroyInstance> {};
|
||||
template<> struct fex_gen_config<vkEnumeratePhysicalDevices> {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceFeatures> {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceFormatProperties> {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceImageFormatProperties> {};
|
||||
// TODO: Output parameter must repack on exit!
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceProperties> {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceQueueFamilyProperties> {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceMemoryProperties> {};
|
||||
// Manually implemented
|
||||
// template<> struct fex_gen_config<vkGetInstanceProcAddr> {};
|
||||
// template<> struct fex_gen_config<vkGetDeviceProcAddr> {};
|
||||
template<> struct fex_gen_config<vkCreateDevice> : fexgen::custom_host_impl {};
|
||||
template<> struct fex_gen_param<vkCreateDevice, 3, VkDevice*> : fexgen::ptr_passthrough {};
|
||||
template<> struct fex_gen_config<vkDestroyDevice> {};
|
||||
template<> struct fex_gen_config<vkEnumerateInstanceExtensionProperties> {};
|
||||
template<> struct fex_gen_config<vkEnumerateDeviceExtensionProperties> {};
|
||||
@@ -71,6 +130,7 @@ template<> struct fex_gen_config<vkResetEvent> {};
|
||||
template<> struct fex_gen_config<vkCreateQueryPool> {};
|
||||
template<> struct fex_gen_config<vkDestroyQueryPool> {};
|
||||
template<> struct fex_gen_config<vkGetQueryPoolResults> {};
|
||||
template<> struct fex_gen_param<vkGetQueryPoolResults, 5, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkCreateBuffer> {};
|
||||
template<> struct fex_gen_config<vkDestroyBuffer> {};
|
||||
template<> struct fex_gen_config<vkCreateBufferView> {};
|
||||
@@ -182,6 +242,7 @@ template<> struct fex_gen_config<vkDestroySamplerYcbcrConversion> {};
|
||||
template<> struct fex_gen_config<vkCreateDescriptorUpdateTemplate> {};
|
||||
template<> struct fex_gen_config<vkDestroyDescriptorUpdateTemplate> {};
|
||||
template<> struct fex_gen_config<vkUpdateDescriptorSetWithTemplate> {};
|
||||
template<> struct fex_gen_param<vkUpdateDescriptorSetWithTemplate, 3, const void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalBufferProperties> {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalFenceProperties> {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalSemaphoreProperties> {};
|
||||
@@ -293,9 +354,11 @@ template<> struct fex_gen_config<vkImportSemaphoreFdKHR> {};
|
||||
template<> struct fex_gen_config<vkGetSemaphoreFdKHR> {};
|
||||
template<> struct fex_gen_config<vkCmdPushDescriptorSetKHR> {};
|
||||
template<> struct fex_gen_config<vkCmdPushDescriptorSetWithTemplateKHR> {};
|
||||
template<> struct fex_gen_param<vkCmdPushDescriptorSetWithTemplateKHR, 4, const void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkCreateDescriptorUpdateTemplateKHR> {};
|
||||
template<> struct fex_gen_config<vkDestroyDescriptorUpdateTemplateKHR> {};
|
||||
template<> struct fex_gen_config<vkUpdateDescriptorSetWithTemplateKHR> {};
|
||||
template<> struct fex_gen_param<vkUpdateDescriptorSetWithTemplateKHR, 3, const void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkCreateRenderPass2KHR> {};
|
||||
template<> struct fex_gen_config<vkCmdBeginRenderPass2KHR> {};
|
||||
template<> struct fex_gen_config<vkCmdNextSubpass2KHR> {};
|
||||
@@ -341,6 +404,8 @@ template<> struct fex_gen_config<vkDeferredOperationJoinKHR> {};
|
||||
template<> struct fex_gen_config<vkGetPipelineExecutablePropertiesKHR> {};
|
||||
template<> struct fex_gen_config<vkGetPipelineExecutableStatisticsKHR> {};
|
||||
template<> struct fex_gen_config<vkGetPipelineExecutableInternalRepresentationsKHR> {};
|
||||
template<> struct fex_gen_config<vkMapMemory2KHR> {};
|
||||
template<> struct fex_gen_config<vkUnmapMemory2KHR> {};
|
||||
template<> struct fex_gen_config<vkCmdSetEvent2KHR> {};
|
||||
template<> struct fex_gen_config<vkCmdResetEvent2KHR> {};
|
||||
template<> struct fex_gen_config<vkCmdWaitEvents2KHR> {};
|
||||
@@ -359,7 +424,13 @@ template<> struct fex_gen_config<vkCmdTraceRaysIndirect2KHR> {};
|
||||
template<> struct fex_gen_config<vkGetDeviceBufferMemoryRequirementsKHR> {};
|
||||
template<> struct fex_gen_config<vkGetDeviceImageMemoryRequirementsKHR> {};
|
||||
template<> struct fex_gen_config<vkGetDeviceImageSparseMemoryRequirementsKHR> {};
|
||||
template<> struct fex_gen_config<vkCmdBindIndexBuffer2KHR> {};
|
||||
template<> struct fex_gen_config<vkGetRenderingAreaGranularityKHR> {};
|
||||
template<> struct fex_gen_config<vkGetDeviceImageSubresourceLayoutKHR> {};
|
||||
template<> struct fex_gen_config<vkGetImageSubresourceLayout2KHR> {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR> {};
|
||||
template<> struct fex_gen_config<vkCreateDebugReportCallbackEXT> : fexgen::custom_host_impl {};
|
||||
template<> struct fex_gen_param<vkCreateDebugReportCallbackEXT, 1, const VkDebugReportCallbackCreateInfoEXT*> : fexgen::ptr_passthrough {};
|
||||
template<> struct fex_gen_config<vkDestroyDebugReportCallbackEXT> : fexgen::custom_host_impl {};
|
||||
template<> struct fex_gen_config<vkDebugReportMessageEXT> {};
|
||||
template<> struct fex_gen_config<vkDebugMarkerSetObjectTagEXT> {};
|
||||
@@ -383,6 +454,7 @@ template<> struct fex_gen_config<vkGetImageViewAddressNVX> {};
|
||||
template<> struct fex_gen_config<vkCmdDrawIndirectCountAMD> {};
|
||||
template<> struct fex_gen_config<vkCmdDrawIndexedIndirectCountAMD> {};
|
||||
template<> struct fex_gen_config<vkGetShaderInfoAMD> {};
|
||||
template<> struct fex_gen_param<vkGetShaderInfoAMD, 5, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalImageFormatPropertiesNV> {};
|
||||
template<> struct fex_gen_config<vkCmdBeginConditionalRenderingEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdEndConditionalRenderingEXT> {};
|
||||
@@ -396,6 +468,8 @@ template<> struct fex_gen_config<vkGetSwapchainCounterEXT> {};
|
||||
template<> struct fex_gen_config<vkGetRefreshCycleDurationGOOGLE> {};
|
||||
template<> struct fex_gen_config<vkGetPastPresentationTimingGOOGLE> {};
|
||||
template<> struct fex_gen_config<vkCmdSetDiscardRectangleEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetDiscardRectangleEnableEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetDiscardRectangleModeEXT> {};
|
||||
template<> struct fex_gen_config<vkSetHdrMetadataEXT> {};
|
||||
template<> struct fex_gen_config<vkSetDebugUtilsObjectNameEXT> {};
|
||||
template<> struct fex_gen_config<vkSetDebugUtilsObjectTagEXT> {};
|
||||
@@ -405,7 +479,8 @@ template<> struct fex_gen_config<vkQueueInsertDebugUtilsLabelEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdBeginDebugUtilsLabelEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdEndDebugUtilsLabelEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdInsertDebugUtilsLabelEXT> {};
|
||||
template<> struct fex_gen_config<vkCreateDebugUtilsMessengerEXT> {};
|
||||
template<> struct fex_gen_config<vkCreateDebugUtilsMessengerEXT> : fexgen::custom_host_impl {};
|
||||
template<> struct fex_gen_param<vkCreateDebugUtilsMessengerEXT, 1, const VkDebugUtilsMessengerCreateInfoEXT*> : fexgen::ptr_passthrough {};
|
||||
template<> struct fex_gen_config<vkDestroyDebugUtilsMessengerEXT> {};
|
||||
template<> struct fex_gen_config<vkSubmitDebugUtilsMessageEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetSampleLocationsEXT> {};
|
||||
@@ -415,6 +490,7 @@ template<> struct fex_gen_config<vkCreateValidationCacheEXT> {};
|
||||
template<> struct fex_gen_config<vkDestroyValidationCacheEXT> {};
|
||||
template<> struct fex_gen_config<vkMergeValidationCachesEXT> {};
|
||||
template<> struct fex_gen_config<vkGetValidationCacheDataEXT> {};
|
||||
template<> struct fex_gen_param<vkGetValidationCacheDataEXT, 3, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkCmdBindShadingRateImageNV> {};
|
||||
template<> struct fex_gen_config<vkCmdSetViewportShadingRatePaletteNV> {};
|
||||
template<> struct fex_gen_config<vkCmdSetCoarseSampleOrderNV> {};
|
||||
@@ -427,19 +503,25 @@ template<> struct fex_gen_config<vkCmdCopyAccelerationStructureNV> {};
|
||||
template<> struct fex_gen_config<vkCmdTraceRaysNV> {};
|
||||
template<> struct fex_gen_config<vkCreateRayTracingPipelinesNV> {};
|
||||
template<> struct fex_gen_config<vkGetRayTracingShaderGroupHandlesKHR> {};
|
||||
template<> struct fex_gen_param<vkGetRayTracingShaderGroupHandlesKHR, 5, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkGetRayTracingShaderGroupHandlesNV> {};
|
||||
template<> struct fex_gen_param<vkGetRayTracingShaderGroupHandlesNV, 5, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkGetAccelerationStructureHandleNV> {};
|
||||
template<> struct fex_gen_param<vkGetAccelerationStructureHandleNV, 3, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkCmdWriteAccelerationStructuresPropertiesNV> {};
|
||||
template<> struct fex_gen_config<vkCompileDeferredNV> {};
|
||||
template<> struct fex_gen_config<vkGetMemoryHostPointerPropertiesEXT> {};
|
||||
template<> struct fex_gen_param<vkGetMemoryHostPointerPropertiesEXT, 2, const void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkCmdWriteBufferMarkerAMD> {};
|
||||
template<> struct fex_gen_config<vkGetPhysicalDeviceCalibrateableTimeDomainsEXT> {};
|
||||
template<> struct fex_gen_config<vkGetCalibratedTimestampsEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdDrawMeshTasksNV> {};
|
||||
template<> struct fex_gen_config<vkCmdDrawMeshTasksIndirectNV> {};
|
||||
template<> struct fex_gen_config<vkCmdDrawMeshTasksIndirectCountNV> {};
|
||||
template<> struct fex_gen_config<vkCmdSetExclusiveScissorEnableNV> {};
|
||||
template<> struct fex_gen_config<vkCmdSetExclusiveScissorNV> {};
|
||||
template<> struct fex_gen_config<vkCmdSetCheckpointNV> {};
|
||||
template<> struct fex_gen_param<vkCmdSetCheckpointNV, 1, const void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkGetQueueCheckpointDataNV> {};
|
||||
template<> struct fex_gen_config<vkInitializePerformanceApiINTEL> {};
|
||||
template<> struct fex_gen_config<vkUninitializePerformanceApiINTEL> {};
|
||||
@@ -470,6 +552,11 @@ template<> struct fex_gen_config<vkCmdSetDepthCompareOpEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetDepthBoundsTestEnableEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetStencilTestEnableEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetStencilOpEXT> {};
|
||||
template<> struct fex_gen_config<vkCopyMemoryToImageEXT> {};
|
||||
template<> struct fex_gen_config<vkCopyImageToMemoryEXT> {};
|
||||
template<> struct fex_gen_config<vkCopyImageToImageEXT> {};
|
||||
template<> struct fex_gen_config<vkTransitionImageLayoutEXT> {};
|
||||
template<> struct fex_gen_config<vkGetImageSubresourceLayout2EXT> {};
|
||||
template<> struct fex_gen_config<vkReleaseSwapchainImagesEXT> {};
|
||||
template<> struct fex_gen_config<vkGetGeneratedCommandsMemoryRequirementsNV> {};
|
||||
template<> struct fex_gen_config<vkCmdPreprocessGeneratedCommandsNV> {};
|
||||
@@ -477,6 +564,7 @@ template<> struct fex_gen_config<vkCmdExecuteGeneratedCommandsNV> {};
|
||||
template<> struct fex_gen_config<vkCmdBindPipelineShaderGroupNV> {};
|
||||
template<> struct fex_gen_config<vkCreateIndirectCommandsLayoutNV> {};
|
||||
template<> struct fex_gen_config<vkDestroyIndirectCommandsLayoutNV> {};
|
||||
template<> struct fex_gen_config<vkCmdSetDepthBias2EXT> {};
|
||||
template<> struct fex_gen_config<vkAcquireDrmDisplayEXT> {};
|
||||
template<> struct fex_gen_config<vkGetDrmDisplayEXT> {};
|
||||
template<> struct fex_gen_config<vkCreatePrivateDataSlotEXT> {};
|
||||
@@ -495,7 +583,6 @@ template<> struct fex_gen_config<vkGetImageViewOpaqueCaptureDescriptorDataEXT> {
|
||||
template<> struct fex_gen_config<vkGetSamplerOpaqueCaptureDescriptorDataEXT> {};
|
||||
template<> struct fex_gen_config<vkGetAccelerationStructureOpaqueCaptureDescriptorDataEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetFragmentShadingRateEnumNV> {};
|
||||
template<> struct fex_gen_config<vkGetImageSubresourceLayout2EXT> {};
|
||||
template<> struct fex_gen_config<vkGetDeviceFaultInfoEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetVertexInputEXT> {};
|
||||
template<> struct fex_gen_config<vkGetDeviceSubpassShadingMaxWorkgroupSizeHUAWEI> {};
|
||||
@@ -519,6 +606,7 @@ template<> struct fex_gen_config<vkCopyMicromapEXT> {};
|
||||
template<> struct fex_gen_config<vkCopyMicromapToMemoryEXT> {};
|
||||
template<> struct fex_gen_config<vkCopyMemoryToMicromapEXT> {};
|
||||
template<> struct fex_gen_config<vkWriteMicromapsPropertiesEXT> {};
|
||||
template<> struct fex_gen_param<vkWriteMicromapsPropertiesEXT, 5, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkCmdCopyMicromapEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdCopyMicromapToMemoryEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdCopyMemoryToMicromapEXT> {};
|
||||
@@ -534,6 +622,9 @@ template<> struct fex_gen_config<vkCmdCopyMemoryIndirectNV> {};
|
||||
template<> struct fex_gen_config<vkCmdCopyMemoryToImageIndirectNV> {};
|
||||
template<> struct fex_gen_config<vkCmdDecompressMemoryNV> {};
|
||||
template<> struct fex_gen_config<vkCmdDecompressMemoryIndirectCountNV> {};
|
||||
template<> struct fex_gen_config<vkGetPipelineIndirectMemoryRequirementsNV> {};
|
||||
template<> struct fex_gen_config<vkCmdUpdatePipelineIndirectBufferNV> {};
|
||||
template<> struct fex_gen_config<vkGetPipelineIndirectDeviceAddressNV> {};
|
||||
template<> struct fex_gen_config<vkCmdSetTessellationDomainOriginEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetDepthClampEnableEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdSetPolygonModeEXT> {};
|
||||
@@ -572,8 +663,13 @@ template<> struct fex_gen_config<vkCreateOpticalFlowSessionNV> {};
|
||||
template<> struct fex_gen_config<vkDestroyOpticalFlowSessionNV> {};
|
||||
template<> struct fex_gen_config<vkBindOpticalFlowSessionImageNV> {};
|
||||
template<> struct fex_gen_config<vkCmdOpticalFlowExecuteNV> {};
|
||||
template<> struct fex_gen_config<vkCreateShadersEXT> {};
|
||||
template<> struct fex_gen_config<vkDestroyShaderEXT> {};
|
||||
template<> struct fex_gen_config<vkGetShaderBinaryDataEXT> {};
|
||||
template<> struct fex_gen_config<vkCmdBindShadersEXT> {};
|
||||
template<> struct fex_gen_config<vkGetFramebufferTilePropertiesQCOM> {};
|
||||
template<> struct fex_gen_config<vkGetDynamicRenderingTilePropertiesQCOM> {};
|
||||
template<> struct fex_gen_config<vkCmdSetAttachmentFeedbackLoopEnableEXT> {};
|
||||
template<> struct fex_gen_config<vkCreateAccelerationStructureKHR> {};
|
||||
template<> struct fex_gen_config<vkDestroyAccelerationStructureKHR> {};
|
||||
template<> struct fex_gen_config<vkCmdBuildAccelerationStructuresKHR> {};
|
||||
@@ -583,6 +679,7 @@ template<> struct fex_gen_config<vkCopyAccelerationStructureKHR> {};
|
||||
template<> struct fex_gen_config<vkCopyAccelerationStructureToMemoryKHR> {};
|
||||
template<> struct fex_gen_config<vkCopyMemoryToAccelerationStructureKHR> {};
|
||||
template<> struct fex_gen_config<vkWriteAccelerationStructuresPropertiesKHR> {};
|
||||
template<> struct fex_gen_param<vkWriteAccelerationStructuresPropertiesKHR, 5, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkCmdCopyAccelerationStructureKHR> {};
|
||||
template<> struct fex_gen_config<vkCmdCopyAccelerationStructureToMemoryKHR> {};
|
||||
template<> struct fex_gen_config<vkCmdCopyMemoryToAccelerationStructureKHR> {};
|
||||
@@ -593,6 +690,7 @@ template<> struct fex_gen_config<vkGetAccelerationStructureBuildSizesKHR> {};
|
||||
template<> struct fex_gen_config<vkCmdTraceRaysKHR> {};
|
||||
template<> struct fex_gen_config<vkCreateRayTracingPipelinesKHR> {};
|
||||
template<> struct fex_gen_config<vkGetRayTracingCaptureReplayShaderGroupHandlesKHR> {};
|
||||
template<> struct fex_gen_param<vkGetRayTracingCaptureReplayShaderGroupHandlesKHR, 5, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<vkCmdTraceRaysIndirectKHR> {};
|
||||
template<> struct fex_gen_config<vkGetRayTracingShaderGroupStackSizeKHR> {};
|
||||
template<> struct fex_gen_config<vkCmdSetRayTracingPipelineStackSizeKHR> {};
|
||||
|
||||
@@ -46,7 +46,7 @@ static void WaylandFinalizeHostTrampolineForGuestListener(void (*callback)()) {
|
||||
}
|
||||
|
||||
extern "C" int fexfn_impl_libwayland_client_wl_proxy_add_listener(struct wl_proxy *proxy,
|
||||
void (**callback)(void), void *data) {
|
||||
guest_layout<void (**)(void)> callback_raw, void* data) {
|
||||
auto guest_interface = ((wl_proxy_private*)proxy)->interface;
|
||||
|
||||
for (int i = 0; i < guest_interface->event_count; ++i) {
|
||||
@@ -60,99 +60,101 @@ extern "C" int fexfn_impl_libwayland_client_wl_proxy_add_listener(struct wl_prox
|
||||
// ? just indicates that the argument may be null, so it doesn't change the signature
|
||||
signature.erase(std::remove(signature.begin(), signature.end(), '?'), signature.end());
|
||||
|
||||
auto callback = callback_raw.data[i];
|
||||
|
||||
if (signature == "") {
|
||||
// E.g. xdg_toplevel::close
|
||||
WaylandFinalizeHostTrampolineForGuestListener<>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<>(callback);
|
||||
} else if (signature == "a") {
|
||||
// E.g. xdg_toplevel::wm_capabilities
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'a'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'a'>(callback);
|
||||
} else if (signature == "hu") {
|
||||
// E.g. zwp_linux_dmabuf_feedback_v1::format_table
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'h', 'u'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'h', 'u'>(callback);
|
||||
} else if (signature == "i") {
|
||||
// E.g. wl_output_listener::scale
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i'>(callback);
|
||||
} else if (signature == "if") {
|
||||
// E.g. wl_touch_listener::orientation
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'f'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'f'>(callback);
|
||||
} else if (signature == "iff") {
|
||||
// E.g. wl_touch_listener::shape
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'f', 'f'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'f', 'f'>(callback);
|
||||
} else if (signature == "ii") {
|
||||
// E.g. xdg_toplevel::configure_bounds
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'i'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'i'>(callback);
|
||||
} else if (signature == "iia") {
|
||||
// E.g. xdg_toplevel::configure
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'i', 'a'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'i', 'a'>(callback);
|
||||
} else if (signature == "iiiiissi") {
|
||||
// E.g. wl_output_listener::geometry
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'i', 'i', 'i', 'i', 's', 's', 'i'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'i', 'i', 'i', 'i', 'i', 's', 's', 'i'>(callback);
|
||||
} else if (signature == "n") {
|
||||
// E.g. wl_data_device_listener::data_offer
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'n'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'n'>(callback);
|
||||
} else if (signature == "o") {
|
||||
// E.g. wl_data_device_listener::selection
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'o'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'o'>(callback);
|
||||
} else if (signature == "u") {
|
||||
// E.g. wl_registry::global_remove
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u'>(callback);
|
||||
} else if (signature == "uff") {
|
||||
// E.g. wl_pointer_listener::motion
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'f', 'f'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'f', 'f'>(callback);
|
||||
} else if (signature == "uhu") {
|
||||
// E.g. wl_keyboard_listener::keymap
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'h', 'u'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'h', 'u'>(callback);
|
||||
} else if (signature == "ui") {
|
||||
// E.g. wl_pointer_listener::axis_discrete
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'i'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'i'>(callback);
|
||||
} else if (signature == "uiff") {
|
||||
// E.g. wl_touch_listener::motion
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'i', 'f', 'f'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'i', 'f', 'f'>(callback);
|
||||
} else if (signature == "uiii") {
|
||||
// E.g. wl_output_listener::mode
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'i', 'i', 'i'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'i', 'i', 'i'>(callback);
|
||||
} else if (signature == "uo") {
|
||||
// E.g. wl_pointer_listener::leave
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'o'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'o'>(callback);
|
||||
} else if (signature == "uoa") {
|
||||
// E.g. wl_keyboard_listener::enter
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'o', 'a'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'o', 'a'>(callback);
|
||||
} else if (signature == "uoff") {
|
||||
// E.g. wl_pointer_listener::enter
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'o', 'f', 'f'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'o', 'f', 'f'>(callback);
|
||||
} else if (signature == "uoffo") {
|
||||
// E.g. wl_data_device_listener::enter
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'o', 'f', 'f', 'o'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'o', 'f', 'f', 'o'>(callback);
|
||||
} else if (signature == "usu") {
|
||||
// E.g. wl_registry::global
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 's', 'u'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 's', 'u'>(callback);
|
||||
} else if (signature == "uu") {
|
||||
// E.g. wl_pointer_listener::axis_stop
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u'>(callback);
|
||||
} else if (signature == "uuf") {
|
||||
// E.g. wl_pointer_listener::axis
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'f'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'f'>(callback);
|
||||
} else if (signature == "uui") {
|
||||
// E.g. wl_touch_listener::up
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'i'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'i'>(callback);
|
||||
} else if (signature == "uuoiff") {
|
||||
// E.g. wl_touch_listener::down
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'o', 'i', 'f', 'f'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'o', 'i', 'f', 'f'>(callback);
|
||||
} else if (signature == "uuu") {
|
||||
// E.g. zwp_linux_dmabuf_v1::modifier
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'u'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'u'>(callback);
|
||||
} else if (signature == "uuuu") {
|
||||
// E.g. wl_pointer_listener::button
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'u', 'u'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'u', 'u'>(callback);
|
||||
} else if (signature == "uuuuu") {
|
||||
// E.g. wl_keyboard_listener::modifiers
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'u', 'u', 'u'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'u', 'u', 'u', 'u', 'u'>(callback);
|
||||
} else if (signature == "s") {
|
||||
// E.g. wl_seat::name
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'s'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'s'>(callback);
|
||||
} else if (signature == "sii") {
|
||||
// E.g. zwp_text_input_v3::preedit_string
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'s', 'i', 'i'>(callback[i]);
|
||||
WaylandFinalizeHostTrampolineForGuestListener<'s', 'i', 'i'>(callback);
|
||||
} else {
|
||||
fprintf(stderr, "TODO: Unknown wayland event signature descriptor %s\n", signature.data());
|
||||
std::abort();
|
||||
@@ -160,7 +162,7 @@ extern "C" int fexfn_impl_libwayland_client_wl_proxy_add_listener(struct wl_prox
|
||||
}
|
||||
|
||||
// Pass the original function pointer table to the host wayland library. This ensures the table is valid until the listener is unregistered.
|
||||
return fexldr_ptr_libwayland_client_wl_proxy_add_listener(proxy, callback, data);
|
||||
return fexldr_ptr_libwayland_client_wl_proxy_add_listener(proxy, callback_raw.data, data);
|
||||
}
|
||||
|
||||
wl_interface* fexfn_impl_libwayland_client_fex_wl_exchange_interface_pointer(wl_interface* guest_interface, char const* name) {
|
||||
|
||||
@@ -10,6 +10,20 @@ struct fex_gen_config {
|
||||
template<typename>
|
||||
struct fex_gen_type {};
|
||||
|
||||
// Function, parameter index, parameter type [optional]
|
||||
template<auto, int, typename = void>
|
||||
struct fex_gen_param {};
|
||||
|
||||
template<> struct fex_gen_type<wl_display> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<wl_proxy> : fexgen::opaque_type {};
|
||||
template<> struct fex_gen_type<wl_interface> : fexgen::opaque_type {};
|
||||
|
||||
template<> struct fex_gen_type<wl_event_queue> : fexgen::opaque_type {};
|
||||
|
||||
// Passed over Wayland's wire protocol for some functions
|
||||
template<> struct fex_gen_type<wl_array> {};
|
||||
|
||||
|
||||
template<> struct fex_gen_config<wl_proxy_destroy> : fexgen::custom_guest_entrypoint {};
|
||||
|
||||
template<> struct fex_gen_config<wl_display_cancel_read> {};
|
||||
@@ -31,6 +45,10 @@ template<> struct fex_gen_config<wl_display_get_fd> {};
|
||||
template<> struct fex_gen_config<wl_event_queue_destroy> {};
|
||||
|
||||
template<> struct fex_gen_config<wl_proxy_add_listener> : fexgen::custom_host_impl, fexgen::custom_guest_entrypoint {};
|
||||
// Callback table
|
||||
template<> struct fex_gen_param<wl_proxy_add_listener, 1, void(**)()> : fexgen::ptr_passthrough {};
|
||||
// User-provided data pointer (not used in caller-provided callback)
|
||||
template<> struct fex_gen_param<wl_proxy_add_listener, 2, void*> : fexgen::assume_compatible_data_layout {};
|
||||
template<> struct fex_gen_config<wl_proxy_create> {};
|
||||
template<> struct fex_gen_config<wl_proxy_create_wrapper> {};
|
||||
template<> struct fex_gen_config<wl_proxy_get_tag> {};
|
||||
@@ -38,6 +56,7 @@ template<> struct fex_gen_config<wl_proxy_get_user_data> {};
|
||||
template<> struct fex_gen_config<wl_proxy_get_version> {};
|
||||
template<> struct fex_gen_config<wl_proxy_set_queue> {};
|
||||
template<> struct fex_gen_config<wl_proxy_set_tag> {};
|
||||
// TODO: This has a void* parameter. Why does 32-bit accept this without annotations?
|
||||
template<> struct fex_gen_config<wl_proxy_set_user_data> {};
|
||||
template<> struct fex_gen_config<wl_proxy_wrapper_destroy> {};
|
||||
|
||||
|
||||
Loaded 100 of 209 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user