mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 01:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7d3090f782 | ||
|
|
435dbdd014 | ||
|
|
d2b6e371e2 | ||
|
|
902d433a00 | ||
|
|
f11c8c747e | ||
|
|
47be6ad720 | ||
|
|
12326d7e71 | ||
|
|
3648ee97e8 | ||
|
|
e3f402a38a | ||
|
|
2aeff14e45 | ||
|
|
986c769f52 | ||
|
|
79a7afe7e1 | ||
|
|
7325cb09be | ||
|
|
c40675c6d0 | ||
|
|
1644d185c4 | ||
|
|
079a69dde2 | ||
|
|
c724742243 | ||
|
|
6788bcf264 | ||
|
|
6b43daccf0 | ||
|
|
67a7d4b352 | ||
|
|
aad2c969b7 | ||
|
|
0df84d3844 | ||
|
|
e320e1de79 | ||
|
|
63eb604b6c | ||
|
|
517a1836ee | ||
|
|
b1e54d803b | ||
|
|
4956ac23e5 | ||
|
|
59f85d6b7d | ||
|
|
d3dcc740b8 | ||
|
|
dd63548cb1 | ||
|
|
7962f40065 | ||
|
|
f3b0b41254 | ||
|
|
47557a6184 | ||
|
|
47bc73575e | ||
|
|
72b2ff88ae | ||
|
|
b37aae728f | ||
|
|
4e5c18113a | ||
|
|
ae3a4ae197 | ||
|
|
e2f973fe93 | ||
|
|
af8770005b | ||
|
|
0206479a08 | ||
|
|
0ddd46ba44 | ||
|
|
e2f0cf41d8 | ||
|
|
ff887adb19 | ||
|
|
638329278c | ||
|
|
08f451d3bc | ||
|
|
83d8317dae | ||
|
|
37d83d9375 | ||
|
|
2ea5cc3f41 | ||
|
|
4b418a81aa | ||
|
|
64a6da9aee | ||
|
|
0a41f470f8 | ||
|
|
229151f878 | ||
|
|
5006e02498 | ||
|
|
c747be4d8b | ||
|
|
e0c7419858 | ||
|
|
3aa38d12bc | ||
|
|
56a6e143ce | ||
|
|
cfafd8929a | ||
|
|
2005933518 | ||
|
|
177542e673 | ||
|
|
9c871c51c9 | ||
|
|
48d71752e2 | ||
|
|
82040bae3e | ||
|
|
555e8bd2d1 | ||
|
|
a0ba36eb1b | ||
|
|
69a3da4f31 | ||
|
|
ec502b99d8 | ||
|
|
c55fb3b5b8 | ||
|
|
1405755a31 | ||
|
|
06b7f3b232 | ||
|
|
7a82b07bfd | ||
|
|
54b178d151 | ||
|
|
26f0e5017f | ||
|
|
9243efc241 | ||
|
|
c91c6efc87 | ||
|
|
9bbdb35711 | ||
|
|
7723c47be4 | ||
|
|
eb41dac33f | ||
|
|
d119a5e7dc | ||
|
|
37eaff93cb | ||
|
|
ca1fb3f851 | ||
|
|
393f7d8702 | ||
|
|
19df181f65 | ||
|
|
d06e98ed84 | ||
|
|
2339831f6d | ||
|
|
93def42061 | ||
|
|
2790f89949 | ||
|
|
c3e9c3f70f | ||
|
|
4bc4d785af | ||
|
|
0309bc3d90 | ||
|
|
1523142540 | ||
|
|
5743282944 | ||
|
|
9d936a8989 | ||
|
|
1e95181b34 | ||
|
|
d85f5b4398 | ||
|
|
ed6d7092a2 | ||
|
|
a164315b7d | ||
|
|
e8084e3f11 | ||
|
|
17ebb38ebd | ||
|
|
fd7b749e47 | ||
|
|
1167d0354a | ||
|
|
6efbc2ba0b | ||
|
|
f2e8f580a4 | ||
|
|
5a24152672 | ||
|
|
fd208da77b | ||
|
|
cad4cb9b64 | ||
|
|
ea263166c5 | ||
|
|
3f66c6754d | ||
|
|
242f17eca9 | ||
|
|
7403789d8b | ||
|
|
ae84606da2 | ||
|
|
3f8f886f65 | ||
|
|
56a4f11ddb | ||
|
|
1695d742b1 | ||
|
|
5efc3212d2 | ||
|
|
82510eb452 | ||
|
|
7917f43720 | ||
|
|
58e0e1a03a | ||
|
|
8b469313f9 | ||
|
|
5f8a7a18f0 | ||
|
|
b3fe634c8b | ||
|
|
6f395885cb | ||
|
|
8cfbc44482 | ||
|
|
4731a5ace4 | ||
|
|
8201228c0d | ||
|
|
4170d9d50c | ||
|
|
3af19acdda | ||
|
|
cb9eb00339 | ||
|
|
98516b539e | ||
|
|
c18deb0d0f | ||
|
|
aa9ae27932 | ||
|
|
f9623af073 | ||
|
|
df9141ee7b | ||
|
|
5d6609a6b1 | ||
|
|
82e014cf5c | ||
|
|
4b652eb91b | ||
|
|
f18599d097 | ||
|
|
480308c021 | ||
|
|
89a13cd5fd | ||
|
|
794a11833d | ||
|
|
200f59ab70 | ||
|
|
91bbce5a96 | ||
|
|
af04940466 | ||
|
|
e431bfe0fa | ||
|
|
3397eb9b80 | ||
|
|
b5ac608fc5 | ||
|
|
8b01071e05 | ||
|
|
a9403db9a5 | ||
|
|
d96e48e246 | ||
|
|
86f363d202 | ||
|
|
208e9c31ea | ||
|
|
516ce6b15c | ||
|
|
7813c7aaa7 | ||
|
|
f32fe56bb3 | ||
|
|
27d8d57984 | ||
|
|
184c9522e1 | ||
|
|
f79de3c3f9 |
No files matched your search
+1
-1
@@ -57,7 +57,7 @@ promote:
|
||||
- arm64
|
||||
- aarch64
|
||||
rules:
|
||||
- if: '$PROMOTE_BRANCH'
|
||||
- if: $PROMOTE_BRANCH && $CI_COMMIT_BRANCH == 'main'
|
||||
before_script:
|
||||
- apt-get -y update
|
||||
- apt-get install -y tmux curl
|
||||
|
||||
+20
-6
@@ -367,7 +367,10 @@ include(LinkerGC)
|
||||
|
||||
## Externals ##
|
||||
|
||||
find_package(unordered_dense QUIET CONFIG)
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
find_package(unordered_dense QUIET CONFIG)
|
||||
endif()
|
||||
|
||||
if (NOT unordered_dense_FOUND)
|
||||
add_subdirectory(External/unordered_dense)
|
||||
endif()
|
||||
@@ -378,8 +381,10 @@ if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ZYDIS)
|
||||
find_package(Zycore 1.5 MODULE QUIET)
|
||||
find_package(Zydis 4.0 MODULE QUIET)
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
find_package(Zycore 1.5 MODULE QUIET)
|
||||
find_package(Zydis 4.0 MODULE QUIET)
|
||||
endif()
|
||||
|
||||
if (TARGET Zydis::Zydis AND TARGET Zycore::Zycore)
|
||||
message(STATUS "Using system Zydis")
|
||||
@@ -400,7 +405,7 @@ find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
|
||||
if (NOT CMAKE_CROSSCOMPILING)
|
||||
if (NOT CMAKE_CROSSCOMPILING AND NOT BUILD_STEAM_SUPPORT)
|
||||
find_package(xxhash MODULE QUIET)
|
||||
endif()
|
||||
|
||||
@@ -428,7 +433,7 @@ else ()
|
||||
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
|
||||
endif()
|
||||
|
||||
if (MINGW)
|
||||
if (MINGW OR BUILD_STEAM_SUPPORT)
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
else()
|
||||
@@ -440,7 +445,10 @@ else()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
find_package(range-v3 QUIET)
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
find_package(range-v3 QUIET)
|
||||
endif()
|
||||
|
||||
if (NOT range-v3_FOUND)
|
||||
add_subdirectory(External/range-v3/)
|
||||
target_compile_definitions(range-v3 INTERFACE RANGES_DISABLE_DEPRECATED_WARNINGS)
|
||||
@@ -609,6 +617,12 @@ if (BUILD_TESTING)
|
||||
execute_process(COMMAND "nproc" OUTPUT_STRIP_TRAILING_WHITESPACE OUTPUT_VARIABLE TEST_JOB_COUNT)
|
||||
endif()
|
||||
set(TEST_JOB_FLAG "-j${TEST_JOB_COUNT}")
|
||||
|
||||
# Runs the whole test suite, with large pages enabled to reduce the cost of fork() in ctest.
|
||||
add_custom_target(tests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}"
|
||||
USES_TERMINAL
|
||||
COMMAND ${CMAKE_COMMAND} -E env GLIBC_TUNABLES=glibc.malloc.hugetlb=1 ctest "--progress" "--timeout" "302" ${TEST_JOB_FLAG})
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/SoftFloat-3e/)
|
||||
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 450bd22322...ee2ec5fd83.
+3
-3
@@ -308,9 +308,9 @@ pygithub==2.6.1 \
|
||||
--hash=sha256:6f2fa6d076ccae475f9fc392cc6cdbd54db985d4f69b8833a28397de75ed6ca3 \
|
||||
--hash=sha256:b5c035392991cca63959e9453286b41b54d83bf2de2daa7d7ff7e4312cebf3bf
|
||||
# via -r requirements_formatting.txt.in
|
||||
pyjwt==2.13.0 \
|
||||
--hash=sha256:41571c89ca91598c79e8ef18a2d07367d4810fbbd6f637794879baf1b7703423 \
|
||||
--hash=sha256:66adcc2aff09b3f1bbd95fc1e1577df8ac8723c978552fd43304c8a290ac5728
|
||||
pyjwt==2.15.1 \
|
||||
--hash=sha256:42d59d631f7768a1028a64c7ff581a9bf7519804daf91fc5b6c56e30eec5e193 \
|
||||
--hash=sha256:4f259e80cdfb6b3fc18a7de51fd1ef9ec79652f25019bae68975ca2468a34df8
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
|
||||
@@ -7,4 +7,4 @@ requests>=2.33.0
|
||||
idna>=3.15
|
||||
certifi>=2024.7.4
|
||||
PyNaCl>=1.6.2
|
||||
PyJWT>=2.13.0
|
||||
PyJWT>=2.15.1
|
||||
Vendored
+1
-1
Submodule External/vixl updated: 20bccdbe04...585d860b12.
+2
-2
@@ -8,8 +8,8 @@ This project aims to provide a fast and functional x86-64 emulation library that
|
||||
* Support a tiered recompiler to allow for fast runtime performance
|
||||
* Support offline compilation and offline tooling for inspection and performance analysis
|
||||
* Support threaded emulation. Including emulating x86-64's strong memory model on weak memory model architectures
|
||||
* Support a significant portion of the x86-64 instruction space.
|
||||
* Including MMX, SSE, SSE2, SSE3, SSSE3, and SSE4*
|
||||
* Support a majority of the x86-64 instruction space.
|
||||
* Including MMX, SSE, SSE2, SSE3, SSSE3, SSE4*, AVX, AVX2, F16C, and AVX-VNNI
|
||||
* Support fallback routines for uncommonly used x86-64 instructions
|
||||
* Including x87 and 3DNow!
|
||||
* Only support userspace emulation.
|
||||
|
||||
@@ -4,6 +4,7 @@ set(FEXCORE_BASE_SRCS
|
||||
Interface/Config/Config.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/FileLoading.cpp
|
||||
Utils/FileUtils.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
Utils/SpinWaitLock.cpp
|
||||
@@ -261,6 +262,10 @@ endfunction()
|
||||
# Build FEXCore_Base static library
|
||||
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base PUBLIC ${LIBS})
|
||||
if (MINGW)
|
||||
target_link_libraries(FEXCore_Base PUBLIC ntdll)
|
||||
endif()
|
||||
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
|
||||
@@ -15,14 +15,9 @@ JITSymbols::~JITSymbols() {
|
||||
}
|
||||
}
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
void JITSymbols::InitFile(uint32_t ProcessPID) {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
#ifdef __ANDROID__
|
||||
// Android simpleperf looks in /data/local/tmp instead of /tmp
|
||||
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
|
||||
#else
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
#endif
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", ProcessPID);
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
}
|
||||
|
||||
|
||||
@@ -35,7 +35,7 @@ public:
|
||||
JITSymbols();
|
||||
~JITSymbols();
|
||||
|
||||
void InitFile();
|
||||
void InitFile(uint32_t ProcessPID);
|
||||
void RegisterNamedRegion(const void* HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void RegisterJITSpace(const void* HostAddr, uint32_t CodeSize);
|
||||
|
||||
|
||||
@@ -91,7 +91,11 @@
|
||||
"ENABLESSE4A": "enablesse4a",
|
||||
"DISABLESSE4A": "disablesse4a",
|
||||
"ENABLEMOPS": "enablemops",
|
||||
"DISABLEMOPS": "disablemops"
|
||||
"DISABLEMOPS": "disablemops",
|
||||
"ENABLEI8MM": "enablei8mm",
|
||||
"DISABLEI8MM": "disablei8mm",
|
||||
"ENABLEDOTPROD": "enabledotprod",
|
||||
"DISABLEDOTPROD": "disabledotprod"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -116,7 +120,9 @@
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
|
||||
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it",
|
||||
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it"
|
||||
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it",
|
||||
"\t{enable,disable}i8mm: Will force enable or disable i8mm even if the host doesn't support it",
|
||||
"\t{enable,disable}dotprod: Will force enable or disable dotprod even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
@@ -135,6 +141,14 @@
|
||||
"Hides hybrid CPU core arrangement."
|
||||
]
|
||||
},
|
||||
"SoftwareRNG": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"AffectsCodeGen": "true",
|
||||
"Desc": [
|
||||
"Emulates RDRAND and RDSEED in software when the host does not implement FEAT_RNG."
|
||||
]
|
||||
},
|
||||
"CPUFeatureRegisters": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
@@ -184,6 +198,14 @@
|
||||
"Attempt to cache anonymous code"
|
||||
]
|
||||
},
|
||||
"DiskCacheMemorySize": {
|
||||
"Type": "uint32",
|
||||
"Default": "0",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"How much memory, if any, to use as a cache for the cache"
|
||||
]
|
||||
},
|
||||
"DiskCachePath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
@@ -199,6 +221,23 @@
|
||||
"Desc": [
|
||||
"Optional list of extra read-only disk cache DBs to consider"
|
||||
]
|
||||
},
|
||||
"DiskCacheMaxFileSize": {
|
||||
"Type": "uint64",
|
||||
"Default": "1073741824",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Size limit on the main cache file - index not included. Default 1G"
|
||||
]
|
||||
},
|
||||
"DiskCachePruneStaleEntries": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"AffectsCodeGen": "false",
|
||||
"Desc": [
|
||||
"Allows FEX to prune stale disk cache entries on startup.",
|
||||
"Frees space by removing cache entries associated with old FEX versions."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
|
||||
@@ -4,13 +4,13 @@
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/DiskCache.h"
|
||||
#include "Interface/Core/SharedCodeBufferManager.h"
|
||||
#include <Interface/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/DiskCache.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
@@ -83,7 +83,6 @@ public:
|
||||
FEX_CONFIG_OPT(EnableLazyCodeCaching, ENABLELAZYCODECACHINGWIP);
|
||||
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
|
||||
|
||||
uint64_t ComputeCodeMapId(std::string_view Filename, int FD) override;
|
||||
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
|
||||
|
||||
fextl::unique_ptr<MappedCodeCacheFile> LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo&, uint64_t FileStartVA) override;
|
||||
@@ -369,10 +368,16 @@ public:
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(SoftwareRNG, SOFTWARERNG);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
|
||||
} Config;
|
||||
|
||||
bool SoftwareRNGEnabled() const {
|
||||
return Config.SoftwareRNG() && HostRNGAvailable;
|
||||
}
|
||||
bool HostRNGAvailable {};
|
||||
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex CodeInvalidationMutex {};
|
||||
|
||||
uint32_t StrictSplitLockMutex {};
|
||||
|
||||
@@ -16,6 +16,7 @@ namespace CPU {
|
||||
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
|
||||
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
|
||||
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
|
||||
{0x0706'0504'0302'0100ULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_INCREMENTAL_U8_INDEX
|
||||
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT
|
||||
{0x0000'0000'8000'0000ULL, 0x0000'0000'8000'0000ULL}, // NAMED_VECTOR_PADDSUBPS_INVERT_UPPER
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'0000ULL}, // NAMED_VECTOR_PADDSUBPD_INVERT
|
||||
|
||||
@@ -458,6 +458,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(GetCPUID() << 24); // Local APIC ID
|
||||
|
||||
const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled();
|
||||
|
||||
Res.ecx = (1 << 0) | // SSE3
|
||||
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
|
||||
(1 << 2) | // DS area supports 64bit layout
|
||||
@@ -488,7 +490,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
(SupportsAVX() << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 28) | // AVX
|
||||
(SupportsAVX() << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(SupportsRAND << 30) | // RDRAND
|
||||
(Hypervisor << 31);
|
||||
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
@@ -648,6 +650,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) const {
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
|
||||
// AVX-VNNI is only advertised when the CPU supports I8MM or Dot Product.
|
||||
// Without these features the implementation is so slow that it is likely
|
||||
// to harm performance.
|
||||
const uint32_t SupportsAVXVNNI = SupportsAVX() && (CTX->HostFeatures.SupportsI8MM || CTX->HostFeatures.SupportsDotProd);
|
||||
|
||||
if (Leaf == 0) {
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t SUPPORTS_RDPID = 1;
|
||||
@@ -664,40 +672,44 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
|
||||
const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT;
|
||||
|
||||
const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled();
|
||||
|
||||
// Number of subfunctions
|
||||
Res.eax = 0x0;
|
||||
Res.ebx = (1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(SupportsAVX() << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(SupportsAVX() << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(SupportsAVX() << 8) | // BMI2
|
||||
(SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
(0 << 12) | // Intel resource directory technology Monitoring
|
||||
(1 << 13) | // Deprecates FPU CS and DS
|
||||
(0 << 14) | // Intel MPX
|
||||
(0 << 15) | // Intel Resource Directory Technology Allocation
|
||||
(0 << 16) | // AVX512-F
|
||||
(0 << 17) | // AVX512-DQ
|
||||
(CTX->HostFeatures.SupportsRAND << 18) | // RDSEED
|
||||
(1 << 19) | // ADCX and ADOX instructions
|
||||
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
|
||||
(0 << 21) | // AVX512-IFMA
|
||||
(0 << 22) | // PCOMMIT (deprecated?)
|
||||
(1 << 23) | // CLFLUSHOPT instruction
|
||||
(1 << 24) | // CLWB instruction
|
||||
(0 << 25) | // Intel processor trace
|
||||
(0 << 26) | // AVX512-PF
|
||||
(0 << 27) | // AVX512-ER
|
||||
(0 << 28) | // AVX512-CD
|
||||
(Features.SHA << 29) | // SHA instructions
|
||||
(0 << 30) | // AVX512-BW
|
||||
(0 << 31); // AVX512-VL
|
||||
// TODO: For now, subfunction 1 only exposes AVX-VNNI so we make it conditional
|
||||
// on AVX-VNNI support. We should revisit this if/when we add more to this leaf.
|
||||
Res.eax = SupportsAVXVNNI;
|
||||
Res.ebx = (1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(SupportsAVX() << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(SupportsAVX() << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(SupportsAVX() << 8) | // BMI2
|
||||
(SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
(0 << 12) | // Intel resource directory technology Monitoring
|
||||
(1 << 13) | // Deprecates FPU CS and DS
|
||||
(0 << 14) | // Intel MPX
|
||||
(0 << 15) | // Intel Resource Directory Technology Allocation
|
||||
(0 << 16) | // AVX512-F
|
||||
(0 << 17) | // AVX512-DQ
|
||||
(SupportsRAND << 18) | // RDSEED
|
||||
(1 << 19) | // ADCX and ADOX instructions
|
||||
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
|
||||
(0 << 21) | // AVX512-IFMA
|
||||
(0 << 22) | // PCOMMIT (deprecated?)
|
||||
(1 << 23) | // CLFLUSHOPT instruction
|
||||
(1 << 24) | // CLWB instruction
|
||||
(0 << 25) | // Intel processor trace
|
||||
(0 << 26) | // AVX512-PF
|
||||
(0 << 27) | // AVX512-ER
|
||||
(0 << 28) | // AVX512-CD
|
||||
(Features.SHA << 29) | // SHA instructions
|
||||
(0 << 30) | // AVX512-BW
|
||||
(0 << 31); // AVX512-VL
|
||||
|
||||
Res.ecx = (1 << 0) | // PREFETCHWT1
|
||||
(0 << 1) | // AVX512VBMI
|
||||
@@ -765,38 +777,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 30) | // Arch capabilities - MSR module specific
|
||||
(0 << 31); // SSBD - Speculative Store Bypass Disable
|
||||
} else if (Leaf == 1) {
|
||||
Res.eax = (0U << 0) | // SHA512
|
||||
(0U << 1) | // SM3
|
||||
(0U << 2) | // SM4
|
||||
(0U << 3) | // RAO_INT
|
||||
(0U << 4) | // AVX_VNNI
|
||||
(0U << 5) | // AVX512_BF16
|
||||
(0U << 6) | // LASS (Linear Address Space Separation)
|
||||
(0U << 7) | // CMPCCXADD
|
||||
(0U << 8) | // ARCH_PERFMON_EXT
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // FAST_REP_MOVSB
|
||||
(0U << 11) | // FAST_REP_STOSB
|
||||
(0U << 12) | // FAST_REP_CMPSB_SCASB
|
||||
(0U << 13) | // Reserved
|
||||
(0U << 14) | // Reserved
|
||||
(0U << 15) | // Reserved
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // FRED (Flexible Return and Event Delivery)
|
||||
(0U << 18) | // LKGS (Load into Kernel GS Base)
|
||||
(0U << 19) | // WRMSRNS
|
||||
(0U << 20) | // NMI_SRC
|
||||
(0U << 21) | // AMX_FP16
|
||||
(0U << 22) | // HRESET
|
||||
(0U << 23) | // AVX_IFMA
|
||||
(0U << 24) | // Reserved
|
||||
(0U << 25) | // Reserved
|
||||
(0U << 26) | // LAM (Linear Address Masking)
|
||||
(0U << 27) | // MSRLIST
|
||||
(0U << 28) | // Reserved
|
||||
(0U << 29) | // Reserved
|
||||
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
|
||||
(0U << 31); // MOVRS
|
||||
Res.eax = (0U << 0) | // SHA512
|
||||
(0U << 1) | // SM3
|
||||
(0U << 2) | // SM4
|
||||
(0U << 3) | // RAO_INT
|
||||
(SupportsAVXVNNI << 4) | // AVX_VNNI
|
||||
(0U << 5) | // AVX512_BF16
|
||||
(0U << 6) | // LASS (Linear Address Space Separation)
|
||||
(0U << 7) | // CMPCCXADD
|
||||
(0U << 8) | // ARCH_PERFMON_EXT
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // FAST_REP_MOVSB
|
||||
(0U << 11) | // FAST_REP_STOSB
|
||||
(0U << 12) | // FAST_REP_CMPSB_SCASB
|
||||
(0U << 13) | // Reserved
|
||||
(0U << 14) | // Reserved
|
||||
(0U << 15) | // Reserved
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // FRED (Flexible Return and Event Delivery)
|
||||
(0U << 18) | // LKGS (Load into Kernel GS Base)
|
||||
(0U << 19) | // WRMSRNS
|
||||
(0U << 20) | // NMI_SRC
|
||||
(0U << 21) | // AMX_FP16
|
||||
(0U << 22) | // HRESET
|
||||
(0U << 23) | // AVX_IFMA
|
||||
(0U << 24) | // Reserved
|
||||
(0U << 25) | // Reserved
|
||||
(0U << 26) | // LAM (Linear Address Masking)
|
||||
(0U << 27) | // MSRLIST
|
||||
(0U << 28) | // Reserved
|
||||
(0U << 29) | // Reserved
|
||||
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
|
||||
(0U << 31); // MOVRS
|
||||
|
||||
// Bits 4-31 currently reserved.
|
||||
Res.ebx = (0U << 0) | // PPIN
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/crc32.h"
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/Utils/TypeDefines.h"
|
||||
@@ -272,16 +273,6 @@ CodeCache::CodeCache(ContextImpl& CTX_)
|
||||
: CTX(CTX_) {}
|
||||
CodeCache::~CodeCache() = default;
|
||||
|
||||
uint64_t CodeCache::ComputeCodeMapId(std::string_view Filename, int FD) {
|
||||
if (Filename.empty()) {
|
||||
return 0xffff'ffff'ffff'ffff;
|
||||
}
|
||||
|
||||
// For now, we just use the file path as an identifier.
|
||||
// TODO: Ensure the hash is unique enough to distinguish executables while remaining independent of the installation location
|
||||
return XXH3_64bits(Filename.data(), Filename.size());
|
||||
}
|
||||
|
||||
struct CodeCacheHeader {
|
||||
std::array<char, 4> Magic = ExpectedMagic;
|
||||
// Version history:
|
||||
@@ -538,30 +529,37 @@ ApplyRIPMoveRelocation(ContextImpl& CTX, uint64_t GuestRIP, uint8_t RegisterInde
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Pointer, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableDataRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
|
||||
uint64_t Value = 0;
|
||||
memcpy(&Value, reinterpret_cast<const void*>(SiteAddress), ValueSize);
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Value, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
static inline int64_t ReadLiveGuestDisplacement(uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
static inline int64_t ReadLiveGuestData(uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
uint64_t Raw = 0;
|
||||
memcpy(&Raw, reinterpret_cast<const void*>(SiteAddress), ValueSize);
|
||||
// manual sign-extension from guest live bytes
|
||||
// 1/2 sizes not permitted in DetectDataMasks currently
|
||||
if (ValueSize == 4) {
|
||||
if (ValueSize == 1) {
|
||||
return (int8_t)Raw;
|
||||
} else if (ValueSize == 2) {
|
||||
return (int16_t)Raw;
|
||||
} else if (ValueSize == 4) {
|
||||
return (int32_t)Raw;
|
||||
} else {
|
||||
return (int64_t)Raw;
|
||||
}
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableDataRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), ReadLiveGuestData(SiteAddress, ValueSize),
|
||||
CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableRIPLiteralRelocation(uint64_t SiteAddress, uint8_t ValueSize, CPU::Arm64Emitter& Emitter) {
|
||||
Emitter.dc64(SiteAddress + ValueSize + ReadLiveGuestDisplacement(SiteAddress, ValueSize));
|
||||
Emitter.dc64(SiteAddress + ValueSize + ReadLiveGuestData(SiteAddress, ValueSize));
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableRIPMoveRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
|
||||
const uint64_t Target = SiteAddress + ValueSize + ReadLiveGuestDisplacement(SiteAddress, ValueSize);
|
||||
const uint64_t Target = SiteAddress + ValueSize + ReadLiveGuestData(SiteAddress, ValueSize);
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Target, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
static inline void ApplyPatchableCRCMoveRelocation(uint64_t SiteAddress, uint8_t ValueSize, uint8_t RegisterIndex, CPU::Arm64Emitter& Emitter) {
|
||||
const uint64_t Target = FEXCore::Utils::crc32(reinterpret_cast<const uint8_t*>(SiteAddress), ValueSize);
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(RegisterIndex), Target, CPU::Arm64Emitter::PadType::DOPAD);
|
||||
}
|
||||
|
||||
@@ -599,6 +597,11 @@ bool CodeCache::ApplyPackedCodeRelocations(uint64_t GuestEntry, std::span<std::b
|
||||
Reloc.PatchableData.RegisterIndex, Emitter);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_CRC_MOVE: {
|
||||
ApplyPatchableCRCMoveRelocation(GuestEntry + Reloc.PatchableData.SiteOffset, Reloc.PatchableData.ValueSize,
|
||||
Reloc.PatchableData.RegisterIndex, Emitter);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unknown packed relocation type {}", ToUnderlying((CPU::RelocationTypes)Reloc.Type));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,6 +30,7 @@ $end_info$
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
#include "Utils/crc32.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
#include "Utils/variable_length_integer.h"
|
||||
|
||||
@@ -76,9 +77,6 @@ $end_info$
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
#include <arm_acle.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
@@ -92,7 +90,7 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
|
||||
if (Config.BlockJITNaming() || Config.GlobalJITNaming() || Config.LibraryJITNaming()) {
|
||||
// Only initialize symbols file if enabled. Ensures we don't pollute /tmp with empty files.
|
||||
Symbols.InitFile();
|
||||
Symbols.InitFile(Features.ProcessPID);
|
||||
}
|
||||
|
||||
uint64_t FrequencyCounter = FEXCore::GetCycleCounterFrequency();
|
||||
@@ -107,6 +105,14 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
#ifndef _WIN32
|
||||
// Check if the kernel supports getrandom().
|
||||
uint64_t Probe {};
|
||||
HostRNGAvailable = FHU::Syscalls::getrandom(&Probe, sizeof(Probe), 0) == sizeof(Probe);
|
||||
#else
|
||||
HostRNGAvailable = true;
|
||||
#endif
|
||||
|
||||
DiskCache.Init(this);
|
||||
}
|
||||
|
||||
@@ -358,11 +364,6 @@ bool ContextImpl::InitCore() {
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
SignalDelegation->SetConfig(Dispatcher->MakeSignalDelegatorConfig());
|
||||
|
||||
#if defined(_WIN32) && !defined(ARCHITECTURE_arm64ec)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
|
||||
if (Config.GdbServer) {
|
||||
// If gdbserver is enabled then this needs to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
@@ -528,6 +529,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
bool HasCustomIR {};
|
||||
|
||||
bool WantsDiskCachePatching = DiskCache.IsReadingDiskCache() || DiskCache.IsWritingDiskCache();
|
||||
|
||||
if (HasCustomIRHandlers.load(std::memory_order_relaxed)) {
|
||||
std::shared_lock lk(CustomIRMutex);
|
||||
auto Handler = CustomIRHandlers.find(GuestRIP);
|
||||
@@ -642,27 +645,11 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
auto ExistingCodePtr = reinterpret_cast<uint8_t*>(Block.Entry + BlockInstructionsLength);
|
||||
auto InstAddressReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
|
||||
|
||||
auto crc32 = [](const uint8_t* Ptr, size_t Size) -> uint32_t {
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
uint32_t Result {};
|
||||
#define do_crc(type, suffix) \
|
||||
while (Size >= sizeof(type)) { \
|
||||
Result = __crc32##suffix(Result, *reinterpret_cast<const type*>(Ptr)); \
|
||||
Ptr += sizeof(type); \
|
||||
Size -= sizeof(type); \
|
||||
}
|
||||
do_crc(uint64_t, d);
|
||||
do_crc(uint32_t, w);
|
||||
do_crc(uint16_t, h);
|
||||
do_crc(uint8_t, b);
|
||||
return Result;
|
||||
#else
|
||||
// Unsupported on non-arm.
|
||||
return 0;
|
||||
#endif
|
||||
};
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(
|
||||
Thread->OpDispatcher->Constant(crc32(ExistingCodePtr, DecodedInfo->InstSize)), InstAddressReg, DecodedInfo->InstSize);
|
||||
auto Value = FEXCore::Utils::crc32(ExistingCodePtr, DecodedInfo->InstSize);
|
||||
auto CRC = WantsDiskCachePatching ?
|
||||
Thread->OpDispatcher->_PatchableGuestCRC(IR::OpSize::i64Bit, Value, (int64_t)ExistingCodePtr, DecodedInfo->InstSize) :
|
||||
Thread->OpDispatcher->Constant(Value);
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CRC, InstAddressReg, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->CondJump(CodeChanged);
|
||||
|
||||
@@ -674,11 +661,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
|
||||
// Generate a relocatable entry for invalidation purposes.
|
||||
auto EntryReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, 0);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry(EntryReg);
|
||||
|
||||
// Exit the function at this instruction after invalidation.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
auto EntryToInvalidate = Thread->OpDispatcher->_EntrypointOffset(GPRSize, 0);
|
||||
auto NewRIP = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
|
||||
// Invalidate and exit the function
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry(EntryToInvalidate, NewRIP);
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -773,10 +759,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
#endif
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
|
||||
IR::IREmitter* IREmitter = Thread->OpDispatcher.get();
|
||||
|
||||
auto ShouldDump = Thread->OpDispatcher->ShouldDumpIR();
|
||||
|
||||
@@ -2,18 +2,22 @@
|
||||
|
||||
#define XXH_STATIC_LINKING_ONLY
|
||||
|
||||
#include "FEXCore/Utils/TypeDefines.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "FEXCore/Config/Config.h"
|
||||
#include "FEXCore/fextl/string.h"
|
||||
#include "FEXHeaderUtils/Filesystem.h"
|
||||
#include "FEXCore/Core/DiskCache.h"
|
||||
#include "FEXCore/Core/DiskCacheFileMapper.h"
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "FEXCore/HLE/SyscallHandler.h"
|
||||
#include "FEXCore/Utils/File.h"
|
||||
#include "FEXCore/fextl/memory.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/DiskCache.h>
|
||||
#include <FEXCore/Core/DiskCacheFileMapper.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/FileUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <atomic>
|
||||
@@ -269,37 +273,16 @@ namespace DiskCache {
|
||||
}
|
||||
}
|
||||
|
||||
bool IndexedDB::StoreCacheBlob(const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey, std::span<const uint8_t> Blob, Index& Index,
|
||||
std::mutex& IndexMutex, std::span<const uint8_t> IndexBlob) {
|
||||
bool IndexedDB::StoreCacheBlob(const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey, std::span<const uint8_t> Blob,
|
||||
MesaFOZ::mesa_index_db_file_entry& IndexEntry, std::span<const uint8_t> IndexBlob) {
|
||||
if (ReadOnly) {
|
||||
// shouldn't happen
|
||||
return false;
|
||||
}
|
||||
|
||||
{
|
||||
std::lock_guard Guard(IndexMutex);
|
||||
auto IndexIt = Index.find(LookupKey);
|
||||
bool Dupe = false;
|
||||
if (IndexIt != Index.end()) {
|
||||
if (IndexIt->second.MoreEntries.get()) {
|
||||
for (auto& [Key, Elem] : *IndexIt->second.MoreEntries) {
|
||||
if (XXH128_isEqual(Elem.GuestHash, *(const XXH128_hash_t*)(IndexBlob.data()))) {
|
||||
Dupe = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!Dupe && XXH128_isEqual(IndexIt->second.MainEntry.GuestHash, *(const XXH128_hash_t*)(IndexBlob.data()))) {
|
||||
Dupe = true;
|
||||
}
|
||||
// could happen if it's seen again while in flight in the store queue
|
||||
if (Dupe) {
|
||||
return true;
|
||||
}
|
||||
if (IndexIt->second.MoreEntries.get() && IndexIt->second.MoreEntries->size() >= LOOKUP_KEY_MAX_BUCKET_DEPTH) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
if (MaxSizeReached || CacheFileSize + Blob.size() >= MaxFileSize) {
|
||||
MaxSizeReached = true;
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!CacheFOZ.Lock(STORE_LOCK_TIMEOUT_MS) || !IndexFOZ.Lock(STORE_LOCK_TIMEOUT_MS)) {
|
||||
@@ -317,10 +300,10 @@ namespace DiskCache {
|
||||
return false;
|
||||
}
|
||||
|
||||
MesaFOZ::mesa_index_db_file_entry IndexEntry {.hash = LookupKey,
|
||||
.size = (uint32_t)Blob.size(),
|
||||
.last_access_time = 0, // todo..
|
||||
.cache_db_file_offset = BlobOffset};
|
||||
IndexEntry = {.hash = LookupKey,
|
||||
.size = (uint32_t)Blob.size(),
|
||||
.last_access_time = 0, // todo..
|
||||
.cache_db_file_offset = BlobOffset};
|
||||
|
||||
std::span<const uint8_t> IndexBlobChunks[] = {
|
||||
{(const uint8_t*)&IndexEntry, sizeof(IndexEntry)},
|
||||
@@ -341,22 +324,6 @@ namespace DiskCache {
|
||||
CacheFileSize = BlobOffset + Blob.size();
|
||||
}
|
||||
|
||||
const IndexExtraBlobHeader* IndexBlobHeader = reinterpret_cast<const IndexExtraBlobHeader*>(IndexBlob.data());
|
||||
|
||||
struct IndexEntry NewEntry {this, BlobOffset, (uint32_t)Blob.size(), IndexBlobHeader->GuestSize, IndexBlobHeader->GuestHash};
|
||||
NewEntry.GuestExtents.resize(IndexBlobHeader->GuestExtentsCount);
|
||||
memcpy(NewEntry.GuestExtents.data(), reinterpret_cast<const uint32_t*>(IndexBlob.data() + sizeof(IndexExtraBlobHeader)),
|
||||
IndexBlobHeader->GuestExtentsCount * sizeof(uint32_t));
|
||||
std::lock_guard Guard(IndexMutex);
|
||||
auto It = Index.find(LookupKey);
|
||||
if (It == Index.end()) {
|
||||
Index.emplace(LookupKey, IndexCacheHead {std::move(NewEntry), IndexBlobHeader->GuestFootprint, nullptr});
|
||||
} else {
|
||||
if (!It->second.MoreEntries.get()) {
|
||||
It->second.MoreEntries = fextl::make_unique<fextl::multimap<uint64_t, struct IndexEntry>>();
|
||||
}
|
||||
It->second.MoreEntries->insert({IndexBlobHeader->GuestFootprint, std::move(NewEntry)});
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -393,6 +360,39 @@ namespace DiskCache {
|
||||
FileMapper = Func;
|
||||
}
|
||||
|
||||
static inline void PruneStaleEntries(std::string_view CacheBase, std::string_view MachineBucketHash) {
|
||||
FEXCORE_PROFILE_SCOPED("DiskCache::PruneStaleEntries");
|
||||
FEX_CONFIG_OPT(DiskCachePruneStaleEntries, DISKCACHEPRUNESTALEENTRIES);
|
||||
|
||||
if (!DiskCachePruneStaleEntries) {
|
||||
return;
|
||||
}
|
||||
|
||||
struct SimpleCapture {
|
||||
std::string_view CacheBase, MachineBucketHash;
|
||||
} const SimpleCapture {
|
||||
.CacheBase = CacheBase,
|
||||
.MachineBucketHash = MachineBucketHash,
|
||||
};
|
||||
|
||||
FEXCore::FileUtils::WalkDirectory(
|
||||
CacheBase,
|
||||
[](std::string_view name, bool is_dir, const void* user_data) {
|
||||
if (!is_dir) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto capture = reinterpret_cast<const struct SimpleCapture*>(user_data);
|
||||
|
||||
// Current behaviour is to remove entries that no longer match the MachineBucketHash.
|
||||
// This means that if the Disk Cache version no longer matches, or the FEXCore::HostFeatures differ, then they get removed.
|
||||
if (name != capture->MachineBucketHash) {
|
||||
FEXCore::FileUtils::RecursiveRemoveDirectory(fextl::fmt::format("{}/{}", capture->CacheBase, name));
|
||||
}
|
||||
},
|
||||
&SimpleCapture);
|
||||
}
|
||||
|
||||
void DiskCache::Init(FEXCore::Context::ContextImpl* CTX) {
|
||||
this->CTX = CTX;
|
||||
|
||||
@@ -402,30 +402,43 @@ namespace DiskCache {
|
||||
|
||||
fextl::string SerializedConfig = FEXCore::Config::SerializeForCache();
|
||||
|
||||
const auto HostFeatureHash = CTX->HostFeatures.HashForCaching();
|
||||
|
||||
struct __attribute__((packed)) {
|
||||
uint16_t FormatVersion;
|
||||
uint8_t Is64BitMode;
|
||||
uint64_t HostFeaturesHash;
|
||||
} BucketHeader = {FormatVersion, CTX->Config.Is64BitMode, CTX->HostFeatures.HashForCaching()};
|
||||
} MachineBucketData = {FormatVersion, HostFeatureHash.HostFeaturesHash};
|
||||
|
||||
fextl::vector<uint8_t> BucketBytes;
|
||||
BucketBytes.resize(sizeof(BucketHeader) + SerializedConfig.size());
|
||||
memcpy(BucketBytes.data(), &BucketHeader, sizeof(BucketHeader));
|
||||
memcpy(BucketBytes.data() + sizeof(BucketHeader), SerializedConfig.data(), SerializedConfig.size());
|
||||
BucketHash = XXH3_128bits(BucketBytes.data(), BucketBytes.size());
|
||||
fextl::vector<uint8_t> BucketBytes(sizeof(MachineBucketData) + (sizeof(uint8_t) * 2) + SerializedConfig.size());
|
||||
memcpy(BucketBytes.data(), &MachineBucketData, sizeof(MachineBucketData));
|
||||
|
||||
// 64-bit mode and HostType is in the ProcessBucket hash only instead of the MachineBucketHash
|
||||
// These effect code-gen, but they aren't part of the MachineBucketHash, as it comes from Process state.
|
||||
BucketBytes[sizeof(MachineBucketData)] = CTX->Config.Is64BitMode;
|
||||
BucketBytes[sizeof(MachineBucketData) + 1] = FEXCore::ToUnderlying(HostFeatureHash.HostType);
|
||||
memcpy(BucketBytes.data() + sizeof(MachineBucketData) + 2, SerializedConfig.data(), SerializedConfig.size());
|
||||
|
||||
uint64_t MachineBucketHash = XXH3_64bits(BucketBytes.data(), sizeof(MachineBucketData));
|
||||
uint64_t ProcessBucketHash = XXH3_64bits(BucketBytes.data() + sizeof(MachineBucketData), 2 + SerializedConfig.size());
|
||||
BucketHash.high64 = MachineBucketHash;
|
||||
BucketHash.low64 = ProcessBucketHash;
|
||||
|
||||
fextl::string BasePath = BasePathOverride();
|
||||
if (BasePath.empty()) {
|
||||
BasePath = FEXCore::Config::GetCacheDirectory() + "DiskCache/";
|
||||
BasePath += fextl::fmt::format("{:016x}{:016x}", BucketHash.high64, BucketHash.low64) + "/";
|
||||
}
|
||||
|
||||
const auto MachineBucketHashAsString = fextl::fmt::format("{:016x}", MachineBucketHash);
|
||||
PruneStaleEntries(BasePath, MachineBucketHashAsString);
|
||||
|
||||
BasePath += MachineBucketHashAsString + "/";
|
||||
FHU::Filesystem::CreateDirectories(BasePath);
|
||||
|
||||
if (!MapDiskCacheFiles) {
|
||||
FileMapper = nullptr;
|
||||
}
|
||||
|
||||
fextl::string RWDBBasePath = BasePath + "RWCacheDB";
|
||||
const auto RWDBBasePath = fextl::fmt::format("{}RWCacheDB_{:016x}", BasePath, ProcessBucketHash);
|
||||
OpenCacheDB(RWDBBasePath, false);
|
||||
|
||||
if (RWCacheDB && !FoundMetadata) {
|
||||
@@ -433,7 +446,8 @@ namespace DiskCache {
|
||||
MesaFOZ::foz_payload_key MetadataKey;
|
||||
memset(MetadataKey.bytes, 0xFF, sizeof(MetadataKey));
|
||||
IndexExtraBlobHeader MetaDataHeader = {};
|
||||
RWCacheDB->StoreCacheBlob(MetadataKey, ~0, {BucketBytes.data(), BucketBytes.size()}, Index, IndexLock,
|
||||
MesaFOZ::mesa_index_db_file_entry IndexEntry = {};
|
||||
RWCacheDB->StoreCacheBlob(MetadataKey, ~0, {BucketBytes.data(), BucketBytes.size()}, IndexEntry,
|
||||
{reinterpret_cast<uint8_t*>(&MetaDataHeader), sizeof(MetaDataHeader)});
|
||||
Index.erase(~0);
|
||||
}
|
||||
@@ -459,7 +473,7 @@ namespace DiskCache {
|
||||
|
||||
if (IsWritingDiskCache()) {
|
||||
FEXCore::Threads::Flags WriterThreadFlags = {.LowPriority = true, .Internal = true};
|
||||
Writer = fextl::make_unique<WorkQueueThread>(WriterThreadFlags);
|
||||
Writer = fextl::make_unique<WorkQueueThread>(WriterThreadFlags, "FEX:DiskCache");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -486,6 +500,49 @@ namespace DiskCache {
|
||||
return XXH3_64bits(&BlobKeyBytes, sizeof(BlobKeyBytes));
|
||||
}
|
||||
|
||||
struct DiskCache::PruneMemoryLRUWorkItem final : WorkQueueThread::WorkItem {
|
||||
DiskCache* Self;
|
||||
PruneMemoryLRUWorkItem(DiskCache* Self)
|
||||
: Self(Self) {}
|
||||
void Run() override {
|
||||
if (Self->MemoryLRUCurrentSize <= Self->MemoryLRUMaxSize + Self->MemoryLRUEvictThreshold) {
|
||||
return;
|
||||
}
|
||||
std::lock_guard IndexGuard(Self->IndexLock);
|
||||
std::lock_guard LRUGuard(Self->MemoryLRULock);
|
||||
if (Self->MemoryLRU.empty()) {
|
||||
return;
|
||||
}
|
||||
auto Last = std::prev(Self->MemoryLRU.end());
|
||||
|
||||
while (Self->MemoryLRUCurrentSize > Self->MemoryLRUMaxSize && !Self->MemoryLRU.empty()) {
|
||||
auto LastKey = *Last;
|
||||
auto IndexEntry = Self->LookupLocked(LastKey.LookupKey, LastKey.GuestHash, LastKey.GuestFootprint);
|
||||
bool AtFront = (Last == Self->MemoryLRU.begin());
|
||||
if (IndexEntry && IndexEntry->MemoryBlob.use_count() > 1) {
|
||||
// being read rn, keep moving
|
||||
if (AtFront) {
|
||||
break;
|
||||
}
|
||||
Last--;
|
||||
continue;
|
||||
}
|
||||
if (IndexEntry) {
|
||||
IndexEntry->MemoryBlob.reset();
|
||||
IndexEntry->LRUEntry.reset();
|
||||
}
|
||||
Self->MemoryLRUCurrentSize -= LastKey.Size;
|
||||
auto Deleted = Last;
|
||||
if (AtFront) {
|
||||
Self->MemoryLRU.erase(Deleted);
|
||||
break;
|
||||
}
|
||||
Last--;
|
||||
Self->MemoryLRU.erase(Deleted);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<CodeHitData> DiskCache::Lookup(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region,
|
||||
uint64_t GuestRIP, std::optional<uint64_t>& GuestCodeKey) {
|
||||
if (!IsReadingDiskCache()) {
|
||||
@@ -546,6 +603,8 @@ namespace DiskCache {
|
||||
fextl::multimap<uint64_t, IndexEntry>::iterator MoreEntriesIt;
|
||||
fextl::multimap<uint64_t, IndexEntry>* MapPointer = nullptr;
|
||||
IndexEntry* EntryUnderReview;
|
||||
fextl::shared_ptr<fextl::vector<uint8_t>> BlobRef;
|
||||
std::optional<fextl::list<MemoryLRUKey>::iterator> LRUIter;
|
||||
uint64_t CurrentFootprint = 0;
|
||||
{
|
||||
std::lock_guard Guard(IndexLock);
|
||||
@@ -563,9 +622,13 @@ namespace DiskCache {
|
||||
}
|
||||
if (MapPointer && MoreEntriesIt != It->second.MoreEntries->end()) {
|
||||
EntryUnderReview = &MoreEntriesIt->second;
|
||||
BlobRef = EntryUnderReview->MemoryBlob;
|
||||
LRUIter = EntryUnderReview->LRUEntry;
|
||||
CurrentFootprint = MoreEntriesIt->first;
|
||||
} else {
|
||||
EntryUnderReview = &MainEntry;
|
||||
BlobRef = EntryUnderReview->MemoryBlob;
|
||||
LRUIter = EntryUnderReview->LRUEntry;
|
||||
CurrentFootprint = MainEntryFootprint;
|
||||
TriedMainEntry = true;
|
||||
}
|
||||
@@ -582,6 +645,8 @@ namespace DiskCache {
|
||||
// if the current footprint is also the main entry's footprint, give main entry a shot next
|
||||
if (!TriedMainEntry && MoreEntriesIt->first == MainEntryFootprint) {
|
||||
EntryUnderReview = &MainEntry;
|
||||
BlobRef = EntryUnderReview->MemoryBlob;
|
||||
LRUIter = EntryUnderReview->LRUEntry;
|
||||
CurrentFootprint = MainEntryFootprint;
|
||||
TriedMainEntry = true;
|
||||
} else {
|
||||
@@ -594,17 +659,27 @@ namespace DiskCache {
|
||||
break;
|
||||
} else {
|
||||
EntryUnderReview = &MainEntry;
|
||||
BlobRef = EntryUnderReview->MemoryBlob;
|
||||
LRUIter = EntryUnderReview->LRUEntry;
|
||||
CurrentFootprint = MainEntryFootprint;
|
||||
TriedMainEntry = true;
|
||||
}
|
||||
}
|
||||
if (!EntryUnderReview && MapPointer && MoreEntriesIt != MapPointer->end()) {
|
||||
EntryUnderReview = &MoreEntriesIt->second;
|
||||
BlobRef = EntryUnderReview->MemoryBlob;
|
||||
LRUIter = EntryUnderReview->LRUEntry;
|
||||
CurrentFootprint = MoreEntriesIt->first;
|
||||
}
|
||||
}
|
||||
|
||||
Advance = true;
|
||||
|
||||
// entry not backed by anything right now - todo prune..
|
||||
if (!EntryUnderReview->DB && !BlobRef) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// do we have enough room in our live code to even hash GuestSize worth?
|
||||
if (Available < EntryUnderReview->GuestSize) {
|
||||
continue;
|
||||
@@ -635,7 +710,7 @@ namespace DiskCache {
|
||||
break;
|
||||
} else if (Validation) {
|
||||
fextl::vector<uint8_t> GuestCode(Entry.GuestSize);
|
||||
if (Entry.Size >= Entry.GuestSize && Entry.DB->ReadCacheBlob(Entry.Offset + Entry.Size - Entry.GuestSize, GuestCode)) {
|
||||
if (Entry.Size >= Entry.GuestSize && Entry.DB && Entry.DB->ReadCacheBlob(Entry.Offset + Entry.Size - Entry.GuestSize, GuestCode)) {
|
||||
const uint8_t* CachedGuest = GuestCode.data();
|
||||
const uint8_t* LiveGuest = reinterpret_cast<const uint8_t*>(GuestRIP);
|
||||
uint64_t DiffCount = 0;
|
||||
@@ -699,7 +774,17 @@ namespace DiskCache {
|
||||
HitData.Blob.resize(GuestPages.size() * sizeof(uint64_t) + EntrySizeWithoutGuestCode);
|
||||
memcpy(HitData.Blob.data(), GuestPages.data(), GuestPages.size() * sizeof(uint64_t));
|
||||
uint32_t BlobOffset = GuestPages.size() * sizeof(uint64_t);
|
||||
if (!Entry.DB->ReadCacheBlob(Entry.Offset, {HitData.Blob.data() + BlobOffset, EntrySizeWithoutGuestCode})) {
|
||||
bool FoundInLRU = false;
|
||||
if (BlobRef && BlobRef->size() >= EntrySizeWithoutGuestCode) {
|
||||
FoundInLRU = true;
|
||||
memcpy(HitData.Blob.data() + BlobOffset, BlobRef->data(), EntrySizeWithoutGuestCode);
|
||||
if (LRUIter) {
|
||||
std::lock_guard Guard(MemoryLRULock);
|
||||
// avoided disk by nabbing from lru, bump to front
|
||||
// LogMan::Msg::IFmt("lru hit! {}", MemoryLRUCurrentSize);
|
||||
MemoryLRU.splice(MemoryLRU.begin(), MemoryLRU, *LRUIter);
|
||||
}
|
||||
} else if (!Entry.DB->ReadCacheBlob(Entry.Offset, {HitData.Blob.data() + BlobOffset, EntrySizeWithoutGuestCode})) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
@@ -720,6 +805,29 @@ namespace DiskCache {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
bool StoreInMemory = !FoundInLRU;
|
||||
if (EntrySizeWithoutGuestCode > MemoryLRUMaxSize) {
|
||||
StoreInMemory = false;
|
||||
}
|
||||
|
||||
if (StoreInMemory) {
|
||||
auto NewBlobRef = fextl::make_shared<fextl::vector<uint8_t>>(EntrySizeWithoutGuestCode);
|
||||
memcpy(NewBlobRef->data(), HitData.Blob.data() + BlobOffset, EntrySizeWithoutGuestCode);
|
||||
std::lock_guard Guard(IndexLock);
|
||||
auto CurrentIndexEntry = LookupLocked(LookupKey, Header.GuestHash, LastFootprintHashed);
|
||||
if (CurrentIndexEntry && !CurrentIndexEntry->MemoryBlob) {
|
||||
std::lock_guard LRUGuard(MemoryLRULock);
|
||||
MemoryLRU.push_front({LookupKey, Header.GuestHash, LastFootprintHashed, EntrySizeWithoutGuestCode});
|
||||
CurrentIndexEntry->LRUEntry = MemoryLRU.begin();
|
||||
CurrentIndexEntry->MemoryBlob = NewBlobRef;
|
||||
MemoryLRUCurrentSize += EntrySizeWithoutGuestCode;
|
||||
}
|
||||
}
|
||||
|
||||
if (StoreInMemory && MemoryLRUCurrentSize > MemoryLRUMaxSize + MemoryLRUEvictThreshold) {
|
||||
Writer->QueueWork(fextl::make_unique<PruneMemoryLRUWorkItem>(this));
|
||||
}
|
||||
|
||||
HitData.HostCode = {HitData.Blob.data() + BlobOffset, Header.HostSize};
|
||||
BlobOffset += Header.HostSize;
|
||||
HitData.EntryPointRIPs = {reinterpret_cast<uint64_t*>(HitData.Blob.data() + BlobOffset), Header.EntryPointCount};
|
||||
@@ -785,23 +893,81 @@ namespace DiskCache {
|
||||
}
|
||||
}
|
||||
|
||||
IndexEntry* DiskCache::LookupLocked(const uint64_t LookupKey, const XXH128_hash_t& GuestHash, const uint64_t GuestFootprint) {
|
||||
auto It = Index.find(LookupKey);
|
||||
if (It != Index.end()) {
|
||||
if (XXH128_isEqual(It->second.MainEntry.GuestHash, GuestHash)) {
|
||||
return &It->second.MainEntry;
|
||||
}
|
||||
if (It->second.MoreEntries.get()) {
|
||||
auto Range = It->second.MoreEntries->equal_range(GuestFootprint);
|
||||
for (auto MoreEntriesIt = Range.first; MoreEntriesIt != Range.second; MoreEntriesIt++) {
|
||||
if (XXH128_isEqual(MoreEntriesIt->second.GuestHash, GuestHash)) {
|
||||
return &MoreEntriesIt->second;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
struct DiskCache::CacheStoreWorkItem final : WorkQueueThread::WorkItem {
|
||||
DiskCache* Self;
|
||||
IndexedDB* DB;
|
||||
MesaFOZ::foz_payload_key UniqueKey;
|
||||
uint64_t LookupKey;
|
||||
fextl::vector<uint8_t> Blob;
|
||||
std::span<uint8_t> Blob;
|
||||
fextl::vector<uint8_t> IndexBlob;
|
||||
bool StoreDisk;
|
||||
CacheStoreWorkItem(DiskCache* Self, IndexedDB* DB, const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey,
|
||||
fextl::vector<uint8_t>&& Blob, fextl::vector<uint8_t>&& IndexBlob)
|
||||
std::span<uint8_t> Blob, fextl::vector<uint8_t>&& IndexBlob, bool StoreDisk)
|
||||
: Self(Self)
|
||||
, DB(DB)
|
||||
, UniqueKey(UniqueKey)
|
||||
, LookupKey(LookupKey)
|
||||
, Blob(std::move(Blob))
|
||||
, IndexBlob(std::move(IndexBlob)) {}
|
||||
, Blob(Blob)
|
||||
, IndexBlob(std::move(IndexBlob))
|
||||
, StoreDisk(StoreDisk) {}
|
||||
void Run() override {
|
||||
DB->StoreCacheBlob(UniqueKey, LookupKey, Blob, Self->Index, Self->IndexLock, IndexBlob);
|
||||
struct MesaFOZ::mesa_index_db_file_entry IndexHeader;
|
||||
bool DiskSuccess = !StoreDisk || DB->StoreCacheBlob(UniqueKey, LookupKey, Blob, IndexHeader, IndexBlob);
|
||||
|
||||
bool KeepEntryInMemory = true;
|
||||
// todo possible other lru condition here like entry size?
|
||||
if (Blob.size() > Self->MemoryLRUMaxSize) {
|
||||
KeepEntryInMemory = false;
|
||||
} else {
|
||||
Self->MemoryLRUCurrentSize += Blob.size();
|
||||
}
|
||||
|
||||
const IndexExtraBlobHeader* IndexAfterHeader = reinterpret_cast<const IndexExtraBlobHeader*>(IndexBlob.data());
|
||||
|
||||
fextl::list<MemoryLRUKey>::iterator NewLRUEntry;
|
||||
if (KeepEntryInMemory) {
|
||||
std::lock_guard Guard(Self->MemoryLRULock);
|
||||
Self->MemoryLRU.push_front({LookupKey, IndexAfterHeader->GuestHash, IndexAfterHeader->GuestFootprint, (uint32_t)Blob.size()});
|
||||
NewLRUEntry = Self->MemoryLRU.begin();
|
||||
}
|
||||
{
|
||||
std::lock_guard Guard(Self->IndexLock);
|
||||
auto IndexEntry = Self->LookupLocked(LookupKey, IndexAfterHeader->GuestHash, IndexAfterHeader->GuestFootprint);
|
||||
LOGMAN_THROW_A_FMT(IndexEntry != nullptr, "Stored Index entry not found?");
|
||||
if (IndexEntry) {
|
||||
if (StoreDisk && DiskSuccess) {
|
||||
IndexEntry->DB = DB;
|
||||
IndexEntry->Offset = IndexHeader.cache_db_file_offset;
|
||||
}
|
||||
if (!KeepEntryInMemory) {
|
||||
IndexEntry->MemoryBlob.reset();
|
||||
} else {
|
||||
IndexEntry->LRUEntry = NewLRUEntry;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (KeepEntryInMemory && Self->MemoryLRUCurrentSize > Self->MemoryLRUMaxSize + Self->MemoryLRUEvictThreshold) {
|
||||
Self->Writer->QueueWork(fextl::make_unique<PruneMemoryLRUWorkItem>(Self));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -825,6 +991,11 @@ namespace DiskCache {
|
||||
if (Target >= Region->BeginVA && Target < Region->EndVA) {
|
||||
continue;
|
||||
}
|
||||
// let it through if it's inside the same ELF image? (like bss)
|
||||
if (Region->FileInfo.MappedSize && Target >= Region->FileStartVA && Target < Region->FileStartVA + Region->FileInfo.MappedSize) {
|
||||
continue;
|
||||
}
|
||||
|
||||
auto TargetSection = CTX->SyscallHandler->LookupExecutableFileSection(Thread, Target);
|
||||
if (!TargetSection || TargetSection->FileInfo.FileId != Region->FileInfo.FileId) {
|
||||
// we don't know where it's pointing, so we don't know how to encode the offset, so we can't cache atm
|
||||
@@ -843,61 +1014,6 @@ namespace DiskCache {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::set<uint64_t> DataMaskAddresses;
|
||||
|
||||
fextl::vector<uint32_t> ExactGuestCodeExtents;
|
||||
uint64_t CurStartExtent = 0, CurEndExtent = 0;
|
||||
const Frontend::Decoder::DecodedBlocks* LastBlock = nullptr;
|
||||
for (auto& SubBlock : DecodedBlockInfo->Blocks) {
|
||||
if (SubBlock.BlockStatus != Frontend::Decoder::DecodedBlockStatus::SUCCESS) {
|
||||
return false;
|
||||
}
|
||||
if (!CurStartExtent) {
|
||||
CurStartExtent = SubBlock.Entry;
|
||||
CurEndExtent = SubBlock.Entry + SubBlock.Size;
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(SubBlock.Entry >= CurEndExtent, "DecodedBlocks not sorted or overlapping?");
|
||||
if (SubBlock.Entry == CurEndExtent) {
|
||||
CurEndExtent = SubBlock.Entry + SubBlock.Size;
|
||||
} else {
|
||||
ExactGuestCodeExtents.push_back(CurStartExtent - GuestRIP);
|
||||
ExactGuestCodeExtents.push_back(CurEndExtent - CurStartExtent);
|
||||
CurStartExtent = SubBlock.Entry;
|
||||
CurEndExtent = SubBlock.Entry + SubBlock.Size;
|
||||
}
|
||||
}
|
||||
// split extents according to data masks as well
|
||||
for (auto& Mask : SubBlock.DataMasks) {
|
||||
if (Mask.FieldAddress > CurStartExtent) {
|
||||
ExactGuestCodeExtents.push_back(CurStartExtent - GuestRIP);
|
||||
ExactGuestCodeExtents.push_back(Mask.FieldAddress - CurStartExtent);
|
||||
}
|
||||
CurStartExtent = Mask.FieldAddress + Mask.ValueSize;
|
||||
|
||||
DataMaskAddresses.insert(Mask.FieldAddress);
|
||||
}
|
||||
LastBlock = &SubBlock;
|
||||
}
|
||||
if (LastBlock && (CurStartExtent != GuestRIP || CurEndExtent != GuestRIP + GuestCode.size())) {
|
||||
ExactGuestCodeExtents.push_back(CurStartExtent - GuestRIP);
|
||||
ExactGuestCodeExtents.push_back(CurEndExtent - CurStartExtent);
|
||||
}
|
||||
|
||||
if (ExactGuestCodeExtents.size() == 0) {
|
||||
ExactGuestCodeExtents.reserve(2);
|
||||
ExactGuestCodeExtents.push_back(0);
|
||||
ExactGuestCodeExtents.push_back(GuestCode.size());
|
||||
}
|
||||
|
||||
uint64_t GuestFootprint = XXH3_64bits(ExactGuestCodeExtents.data(), ExactGuestCodeExtents.size() * sizeof(uint32_t));
|
||||
|
||||
// if (ExactGuestCodeExtents.size()) {
|
||||
// LogMan::Msg::IFmt("store! length {:d}", GuestCode.size());
|
||||
// for(uint32_t i = 0; i < ExactGuestCodeExtents.size(); i+=2 ) {
|
||||
// LogMan::Msg::IFmt("extent {} {}", ExactGuestCodeExtents[i], ExactGuestCodeExtents[i]+ExactGuestCodeExtents[i+1]);
|
||||
// }
|
||||
// }
|
||||
|
||||
const uint32_t EntryPointCount = (uint32_t)CompiledCode.EntryPoints.size();
|
||||
|
||||
const size_t HeaderOffset = 0;
|
||||
@@ -922,15 +1038,6 @@ namespace DiskCache {
|
||||
.ThunkRelocCount = ThunkRelocCount,
|
||||
};
|
||||
|
||||
{
|
||||
XXH3_state_t HashState;
|
||||
XXH3_128bits_reset(&HashState);
|
||||
for (uint32_t i = 0; i < ExactGuestCodeExtents.size(); i += 2) {
|
||||
XXH3_128bits_update(&HashState, GuestCode.data() + ExactGuestCodeExtents[i], ExactGuestCodeExtents[i + 1]);
|
||||
}
|
||||
Header.GuestHash = XXH3_128bits_digest(&HashState);
|
||||
}
|
||||
memcpy(BlobData + HeaderOffset, &Header, sizeof(Header));
|
||||
memcpy(BlobData + HostCodeOffset, CompiledCode.BlockBegin, CompiledCode.Size);
|
||||
|
||||
// pack and relocate entrypoints
|
||||
@@ -943,6 +1050,8 @@ namespace DiskCache {
|
||||
EntryIdx++;
|
||||
}
|
||||
|
||||
fextl::set<uint64_t> DataMasksConsumed;
|
||||
|
||||
// pack relocations
|
||||
auto* SmallRelocs = reinterpret_cast<BlobSmallRelocation*>(BlobData + SmallRelocsOffset);
|
||||
auto* ThunkRelocs = reinterpret_cast<BlobThunkRelocation*>(BlobData + ThunkRelocsOffset);
|
||||
@@ -979,7 +1088,8 @@ namespace DiskCache {
|
||||
}
|
||||
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE:
|
||||
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE:
|
||||
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
|
||||
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL:
|
||||
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_CRC_MOVE: {
|
||||
// same data for all, relative vs. not and register vs. literal will depend on type on apply
|
||||
BlobSmallRelocation SmallReloc = {};
|
||||
SmallReloc.Offset = Reloc.Header.Offset;
|
||||
@@ -989,12 +1099,8 @@ namespace DiskCache {
|
||||
SmallReloc.PatchableData.SiteOffset = uint32_t(Reloc.GuestPatchableData.SiteAddress - GuestRIP);
|
||||
SmallRelocs[SmallIdx++] = SmallReloc;
|
||||
|
||||
// mark the corresponding data mask consumed - we might not find one if they got removed due to the smc workaround
|
||||
// the hash will just fail on lookup later
|
||||
auto It = DataMaskAddresses.find(Reloc.GuestPatchableData.SiteAddress);
|
||||
if (It != DataMaskAddresses.end()) {
|
||||
DataMaskAddresses.erase(It);
|
||||
}
|
||||
// mark the corresponding data mask consumed
|
||||
DataMasksConsumed.insert(Reloc.GuestPatchableData.SiteAddress);
|
||||
|
||||
break;
|
||||
}
|
||||
@@ -1009,18 +1115,109 @@ namespace DiskCache {
|
||||
}
|
||||
}
|
||||
|
||||
if (!DataMaskAddresses.empty()) {
|
||||
LogMan::Msg::IFmt("DiskCache: DataMask unaccounted for! {:x}", GuestCodeKey);
|
||||
// this would mean we omitted contents in the hash that we're not going to patch, which would be loading corrupt code
|
||||
return false;
|
||||
fextl::vector<uint32_t> ExactGuestCodeExtents;
|
||||
uint64_t CurStartExtent = 0, CurEndExtent = 0;
|
||||
const Frontend::Decoder::DecodedBlocks* LastBlock = nullptr;
|
||||
for (auto& SubBlock : DecodedBlockInfo->Blocks) {
|
||||
if (SubBlock.BlockStatus != Frontend::Decoder::DecodedBlockStatus::SUCCESS) {
|
||||
return false;
|
||||
}
|
||||
if (!CurStartExtent) {
|
||||
CurStartExtent = SubBlock.Entry;
|
||||
CurEndExtent = SubBlock.Entry + SubBlock.Size;
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(SubBlock.Entry >= CurEndExtent, "DecodedBlocks not sorted or overlapping?");
|
||||
if (SubBlock.Entry == CurEndExtent) {
|
||||
CurEndExtent = SubBlock.Entry + SubBlock.Size;
|
||||
} else {
|
||||
ExactGuestCodeExtents.push_back(CurStartExtent - GuestRIP);
|
||||
ExactGuestCodeExtents.push_back(CurEndExtent - CurStartExtent);
|
||||
CurStartExtent = SubBlock.Entry;
|
||||
CurEndExtent = SubBlock.Entry + SubBlock.Size;
|
||||
}
|
||||
}
|
||||
// split extents according to data masks as well
|
||||
for (auto& Mask : SubBlock.DataMasks) {
|
||||
if (Mask.Type == Frontend::Decoder::DataMaskType::NOP || DataMasksConsumed.find(Mask.FieldAddress) != DataMasksConsumed.end()) {
|
||||
if (Mask.FieldAddress > CurStartExtent) {
|
||||
ExactGuestCodeExtents.push_back(CurStartExtent - GuestRIP);
|
||||
ExactGuestCodeExtents.push_back(Mask.FieldAddress - CurStartExtent);
|
||||
}
|
||||
CurStartExtent = Mask.FieldAddress + Mask.ValueSize;
|
||||
}
|
||||
}
|
||||
LastBlock = &SubBlock;
|
||||
}
|
||||
if (LastBlock && (CurStartExtent != GuestRIP || CurEndExtent != GuestRIP + GuestCode.size())) {
|
||||
ExactGuestCodeExtents.push_back(CurStartExtent - GuestRIP);
|
||||
ExactGuestCodeExtents.push_back(CurEndExtent - CurStartExtent);
|
||||
}
|
||||
|
||||
memcpy(BlobData + GuestCodeOffset, GuestCode.data(), GuestCode.size());
|
||||
if (ExactGuestCodeExtents.size() == 0) {
|
||||
ExactGuestCodeExtents.reserve(2);
|
||||
ExactGuestCodeExtents.push_back(0);
|
||||
ExactGuestCodeExtents.push_back(GuestCode.size());
|
||||
}
|
||||
|
||||
uint64_t GuestFootprint = XXH3_64bits(ExactGuestCodeExtents.data(), ExactGuestCodeExtents.size() * sizeof(uint32_t));
|
||||
|
||||
// if (ExactGuestCodeExtents.size()) {
|
||||
// LogMan::Msg::IFmt("store! length {:d}", GuestCode.size());
|
||||
// for(uint32_t i = 0; i < ExactGuestCodeExtents.size(); i+=2 ) {
|
||||
// LogMan::Msg::IFmt("extent {} {}", ExactGuestCodeExtents[i], ExactGuestCodeExtents[i]+ExactGuestCodeExtents[i+1]);
|
||||
// }
|
||||
// }
|
||||
{
|
||||
XXH3_state_t HashState;
|
||||
XXH3_128bits_reset(&HashState);
|
||||
for (uint32_t i = 0; i < ExactGuestCodeExtents.size(); i += 2) {
|
||||
XXH3_128bits_update(&HashState, GuestCode.data() + ExactGuestCodeExtents[i], ExactGuestCodeExtents[i + 1]);
|
||||
}
|
||||
Header.GuestHash = XXH3_128bits_digest(&HashState);
|
||||
}
|
||||
memcpy(BlobData + HeaderOffset, &Header, sizeof(Header));
|
||||
|
||||
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, GuestRIP);
|
||||
uint64_t LookupKey =
|
||||
MakeLookupKey(Thread, GuestCodeKey, RangeInfo.Writable, GuestRIP == CTX->GetMonoBackPatcherBlock().load(std::memory_order_relaxed));
|
||||
|
||||
// blob done, publish to index as in-memory for now, flush to disk below
|
||||
auto BlobRef = fextl::make_shared<fextl::vector<uint8_t>>(std::move(Blob));
|
||||
struct IndexEntry NewEntry {nullptr, 0, (uint32_t)TotalSize, Header.GuestSize, Header.GuestHash, BlobRef};
|
||||
NewEntry.GuestExtents = ExactGuestCodeExtents;
|
||||
{
|
||||
std::lock_guard Guard(IndexLock);
|
||||
auto It = Index.find(LookupKey);
|
||||
if (It == Index.end()) {
|
||||
Index.emplace(LookupKey, IndexCacheHead {std::move(NewEntry), GuestFootprint, nullptr});
|
||||
} else {
|
||||
bool Dupe = false;
|
||||
if (It->second.MoreEntries.get()) {
|
||||
for (auto& [Key, Elem] : *It->second.MoreEntries) {
|
||||
if (XXH128_isEqual(Elem.GuestHash, Header.GuestHash)) {
|
||||
Dupe = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!Dupe && XXH128_isEqual(It->second.MainEntry.GuestHash, Header.GuestHash)) {
|
||||
Dupe = true;
|
||||
}
|
||||
// could happen if it's seen again while in flight in the store queue
|
||||
if (Dupe) {
|
||||
return true;
|
||||
}
|
||||
if (!It->second.MoreEntries.get()) {
|
||||
It->second.MoreEntries = fextl::make_unique<fextl::multimap<uint64_t, struct IndexEntry>>();
|
||||
} else if (It->second.MoreEntries->size() >= LOOKUP_KEY_MAX_BUCKET_DEPTH) {
|
||||
return false;
|
||||
}
|
||||
It->second.MoreEntries->insert({GuestFootprint, std::move(NewEntry)});
|
||||
}
|
||||
}
|
||||
|
||||
memcpy(BlobData + GuestCodeOffset, GuestCode.data(), GuestCode.size());
|
||||
|
||||
MesaFOZ::foz_payload_key Key = {};
|
||||
{
|
||||
XXH3_state_t HashState;
|
||||
@@ -1039,8 +1236,15 @@ namespace DiskCache {
|
||||
memcpy(IndexBlob.data(), &IndexBlobHeader, sizeof(IndexExtraBlobHeader));
|
||||
memcpy(IndexBlob.data() + sizeof(IndexExtraBlobHeader), ExactGuestCodeExtents.data(), ExactGuestCodeExtents.size() * sizeof(uint32_t));
|
||||
|
||||
bool StoreDisk = true;
|
||||
|
||||
if (RWCacheDB->Full()) {
|
||||
StoreDisk = false;
|
||||
}
|
||||
|
||||
// hand the rest off to the writer thread
|
||||
Writer->QueueWork(fextl::make_unique<CacheStoreWorkItem>(this, RWCacheDB.get(), Key, LookupKey, std::move(Blob), std::move(IndexBlob)));
|
||||
Writer->QueueWork(fextl::make_unique<CacheStoreWorkItem>(this, RWCacheDB.get(), Key, LookupKey, std::span(BlobData, TotalSize),
|
||||
std::move(IndexBlob), StoreDisk));
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,258 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/WorkQueueThread.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_set.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <stdint.h>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <span>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace DiskCache {
|
||||
|
||||
namespace MesaFOZ {
|
||||
|
||||
#define FOSSILIZE_BLOB_HASH_LENGTH 40 /* SHA1 hexadecimal string length */
|
||||
|
||||
struct __attribute__((packed)) foz_payload_key {
|
||||
uint8_t bytes[FOSSILIZE_BLOB_HASH_LENGTH];
|
||||
};
|
||||
|
||||
struct __attribute__((packed)) foz_payload_header {
|
||||
uint32_t payload_size;
|
||||
uint32_t format;
|
||||
uint32_t crc;
|
||||
uint32_t uncompressed_size;
|
||||
};
|
||||
|
||||
struct mesa_index_db_file_entry;
|
||||
|
||||
} // namespace MesaFOZ
|
||||
|
||||
class IndexedDB;
|
||||
|
||||
struct MemoryLRUKey {
|
||||
uint64_t LookupKey;
|
||||
XXH128_hash_t GuestHash;
|
||||
uint64_t GuestFootprint;
|
||||
uint32_t Size;
|
||||
};
|
||||
|
||||
struct IndexEntry {
|
||||
IndexedDB* DB;
|
||||
uint64_t Offset;
|
||||
uint32_t Size;
|
||||
uint32_t GuestSize;
|
||||
XXH128_hash_t GuestHash;
|
||||
fextl::shared_ptr<fextl::vector<uint8_t>> MemoryBlob;
|
||||
fextl::vector<uint32_t> GuestExtents;
|
||||
std::optional<fextl::list<MemoryLRUKey>::iterator> LRUEntry;
|
||||
};
|
||||
|
||||
struct IndexCacheHead {
|
||||
struct IndexEntry MainEntry;
|
||||
uint64_t MainEntryFootprint;
|
||||
fextl::unique_ptr<fextl::multimap<uint64_t, IndexEntry>> MoreEntries; // sorted by guest footprint
|
||||
};
|
||||
|
||||
struct __attribute__((packed)) BlobFixedHeader {
|
||||
uint32_t GuestSize;
|
||||
uint32_t HostSize;
|
||||
uint32_t EntryPointCount;
|
||||
uint32_t SmallRelocCount;
|
||||
uint32_t ThunkRelocCount;
|
||||
XXH128_hash_t GuestHash;
|
||||
};
|
||||
|
||||
// packed struct for types 0, 2 and 3. type 1 is bigger and separate below
|
||||
struct __attribute__((packed)) BlobSmallRelocation {
|
||||
uint32_t Offset;
|
||||
uint8_t Type;
|
||||
union {
|
||||
struct __attribute__((packed)) {
|
||||
uint32_t Symbol;
|
||||
} Named;
|
||||
struct __attribute__((packed)) {
|
||||
uint64_t GuestRIP;
|
||||
} RIPLiteral;
|
||||
struct __attribute__((packed)) {
|
||||
uint8_t RegisterIndex;
|
||||
uint64_t GuestRIP;
|
||||
} RIPMove;
|
||||
struct __attribute__((packed)) {
|
||||
uint8_t RegisterIndex;
|
||||
uint8_t ValueSize;
|
||||
uint32_t SiteOffset;
|
||||
} PatchableData;
|
||||
};
|
||||
};
|
||||
|
||||
// type 1, implicit
|
||||
struct __attribute__((packed)) BlobThunkRelocation {
|
||||
uint32_t Offset;
|
||||
uint8_t RegisterIndex;
|
||||
uint8_t SymbolHash[32]; // sha256sum in the real RelocNamedThunkMove
|
||||
};
|
||||
|
||||
struct CodeHitData {
|
||||
fextl::vector<uint8_t> Blob;
|
||||
std::span<uint8_t> HostCode;
|
||||
std::span<const uint64_t> GuestPages;
|
||||
std::span<uint64_t> EntryPointRIPs;
|
||||
std::span<const uint32_t> EntryPointHostOffsets;
|
||||
|
||||
// the spans above point to memory owned by the Blob vec, so it's important this can't be copied
|
||||
CodeHitData() = default;
|
||||
CodeHitData(CodeHitData&&) = default;
|
||||
CodeHitData& operator=(CodeHitData&&) = default;
|
||||
CodeHitData(const CodeHitData&) = delete;
|
||||
CodeHitData& operator=(const CodeHitData&) = delete;
|
||||
};
|
||||
|
||||
using Index = fextl::robin_map<uint64_t, IndexCacheHead>;
|
||||
|
||||
class FOZFile {
|
||||
public:
|
||||
bool Open(const fextl::string& CacheFileName, bool ReadOnly);
|
||||
bool Lock(uint32_t TimeoutMS) {
|
||||
if (!FD) {
|
||||
return false;
|
||||
}
|
||||
return FD->Lock(TimeoutMS);
|
||||
}
|
||||
bool Unlock() {
|
||||
if (!FD) {
|
||||
return false;
|
||||
}
|
||||
return FD->Unlock();
|
||||
}
|
||||
File::File::FileHandleType GetHandle() {
|
||||
return FD ? FD->GetHandle() : (File::File::FileHandleType)-1;
|
||||
}
|
||||
ssize_t Size();
|
||||
bool ReadAll(fextl::vector<uint8_t>& Out); // from first blob
|
||||
bool ReadBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
|
||||
bool WriteBlob(const MesaFOZ::foz_payload_key& Key, std::span<const std::span<const uint8_t>> BlobChunks, uint64_t& OutBlobOffset);
|
||||
|
||||
private:
|
||||
static constexpr uint32_t OPEN_LOCK_TIMEOUT_MS = 100;
|
||||
|
||||
fextl::string FileName;
|
||||
fextl::unique_ptr<File::File> FD;
|
||||
bool ReadOnly = false;
|
||||
};
|
||||
|
||||
class IndexedDB {
|
||||
public:
|
||||
bool Open(const fextl::string& CacheDBName, bool ReadOnly);
|
||||
void PopulateIndex(Index& CacheIndex, bool& FoundMetadata);
|
||||
bool ReadCacheBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
|
||||
bool StoreCacheBlob(const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey, std::span<const uint8_t> Blob,
|
||||
MesaFOZ::mesa_index_db_file_entry& IndexEntry, std::span<const uint8_t> IndexBlob);
|
||||
bool Full() const {
|
||||
return MaxSizeReached;
|
||||
}
|
||||
|
||||
private:
|
||||
// stores run on the Writer, so returning quick isn't as important
|
||||
static constexpr uint32_t STORE_LOCK_TIMEOUT_MS = 1000;
|
||||
static constexpr uint64_t BIG_MAPPING_SIZE = 1ULL << 33;
|
||||
|
||||
FOZFile CacheFOZ;
|
||||
uint8_t* CacheFileMapping = nullptr;
|
||||
std::atomic<uint64_t> CacheFileSize;
|
||||
FOZFile IndexFOZ;
|
||||
bool ReadOnly = false;
|
||||
bool MaxSizeReached = false;
|
||||
|
||||
FEX_CONFIG_OPT(MaxFileSize, DISKCACHEMAXFILESIZE);
|
||||
};
|
||||
|
||||
class DiskCache {
|
||||
public:
|
||||
void Init(FEXCore::Context::ContextImpl* CTX);
|
||||
|
||||
std::optional<CodeHitData> Lookup(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP,
|
||||
std::optional<uint64_t>& GuestCodeKey);
|
||||
void Validate(uint64_t GuestCodeKey, const CodeHitData& Hit, const CPU::CPUBackend::CompiledCode& CompiledCode,
|
||||
std::optional<ExecutableFileSectionInfo> Region);
|
||||
bool Store(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP, uint64_t GuestCodeKey,
|
||||
std::span<const uint8_t> GuestCode, const CPU::CPUBackend::CompiledCode& CompiledCode,
|
||||
std::span<const FEXCore::CPU::Relocation> Relocations, const Frontend::Decoder::DecodedBlockInformation* DecodedBlockInfo);
|
||||
|
||||
bool IsWritingDiskCache() const {
|
||||
return WritingDiskCache;
|
||||
}
|
||||
bool IsReadingDiskCache() const {
|
||||
return ReadingDiskCache;
|
||||
}
|
||||
bool IsValidating() const {
|
||||
return Validation;
|
||||
}
|
||||
|
||||
private:
|
||||
bool OpenCacheDB(const fextl::string& CacheDBName, bool ReadOnly);
|
||||
uint64_t MakeLookupKey(Core::InternalThreadState* Thread, const uint64_t ModuleOffset, bool Writable, bool MonoBackpatcher);
|
||||
IndexEntry* LookupLocked(const uint64_t LookupKey, const XXH128_hash_t& GuestHash, const uint64_t GuestFootprint);
|
||||
|
||||
bool ReadingDiskCache {};
|
||||
bool WritingDiskCache {};
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
XXH128_hash_t BucketHash;
|
||||
fextl::vector<fextl::unique_ptr<IndexedDB>> ROCacheDBs;
|
||||
fextl::unique_ptr<IndexedDB> RWCacheDB;
|
||||
Index Index;
|
||||
std::mutex IndexLock;
|
||||
bool FoundMetadata = false;
|
||||
struct CacheStoreWorkItem;
|
||||
|
||||
struct PruneMemoryLRUWorkItem;
|
||||
std::atomic<uint64_t> MemoryLRUCurrentSize {};
|
||||
std::mutex MemoryLRULock;
|
||||
fextl::list<MemoryLRUKey> MemoryLRU;
|
||||
|
||||
// the Writer holds references to all this stuff above and needs to be last
|
||||
fextl::unique_ptr<WorkQueueThread> Writer;
|
||||
|
||||
FEX_CONFIG_OPT(EnableDiskCache, DISKCACHE);
|
||||
FEX_CONFIG_OPT(Validation, DISKCACHEVALIDATION);
|
||||
FEX_CONFIG_OPT(MapDiskCacheFiles, DISKCACHEFILEMAPPING);
|
||||
FEX_CONFIG_OPT(RelocationFilter, DISKCACHERELOCATIONFILTER);
|
||||
FEX_CONFIG_OPT(AnonCaching, DISKCACHEANONCACHING);
|
||||
FEX_CONFIG_OPT(BasePathOverride, DISKCACHEPATH);
|
||||
FEX_CONFIG_OPT(RODBNames, DISKCACHERODBNAMES);
|
||||
FEX_CONFIG_OPT(MemoryLRUMaxSize, DISKCACHEMEMORYSIZE);
|
||||
|
||||
uint64_t MemoryLRUEvictThreshold = MemoryLRUMaxSize / 25;
|
||||
};
|
||||
|
||||
static constexpr uint16_t AnonPrefixGuestBytes = 64;
|
||||
// The current version of the diskcache.
|
||||
// This must be changed any time codegen changes occur!
|
||||
// Be aware of the impact of changing this frequently!
|
||||
static constexpr uint16_t FormatVersion = 29;
|
||||
|
||||
static constexpr uint32_t LOOKUP_KEY_MAX_BUCKET_DEPTH = 500;
|
||||
|
||||
} // namespace DiskCache
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -276,8 +276,15 @@ void Dispatcher::EmitDispatcher() {
|
||||
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
constexpr size_t InterruptPageOffset =
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState);
|
||||
if constexpr (InterruptPageOffset <= 32760) {
|
||||
str(ARMEmitter::XReg::zr, STATE, InterruptPageOffset);
|
||||
} else {
|
||||
// Need to use vector 128-bit store for this range.
|
||||
// Doesn't matter which register we use to store.
|
||||
str(ARMEmitter::QReg::q0, STATE, InterruptPageOffset);
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
@@ -507,6 +514,64 @@ void Dispatcher::EmitDispatcher() {
|
||||
(void)b(&LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
// All dynamic and static registers are spilled coming in to this handler.
|
||||
// It's also the end of block and RIP might have changed, so we jump directly to the top of the loop.
|
||||
ThreadDispatchSyscallHandler = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Store in the state that we are in a syscall
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerFunc));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
// Syscall result is in any static register that the frontend desired.
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r1,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
});
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
(void)b(&LoopTop);
|
||||
}
|
||||
|
||||
{
|
||||
// All dynamic and static registers are spilled coming in to this handler.
|
||||
// It's also the end of block and RIP might have changed, so we jump directly to the top of the loop.
|
||||
ThreadDispatchRemoveCodeEntry = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Arguments are already in x0, x1. Just jump to the handler.
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadRemoveCodeEntryFromJIT));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r1,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
});
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
(void)b(&LoopTop);
|
||||
}
|
||||
|
||||
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
|
||||
auto Address = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -2620,6 +2685,8 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
|
||||
Ptrs.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Ptrs.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
|
||||
Ptrs.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
|
||||
Ptrs.ThreadDispatchSyscallHandler = ThreadDispatchSyscallHandler;
|
||||
Ptrs.ThreadDispatchRemoveCodeEntry = ThreadDispatchRemoveCodeEntry;
|
||||
Ptrs.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Ptrs.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Ptrs.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
|
||||
@@ -81,6 +81,8 @@ private:
|
||||
uint64_t AbsoluteLoopTopAddressEnterECFillSRA {};
|
||||
uint64_t ThreadPauseHandlerAddress {};
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA {};
|
||||
uint64_t ThreadDispatchSyscallHandler {};
|
||||
uint64_t ThreadDispatchRemoveCodeEntry {};
|
||||
uint64_t ExitFunctionLinkerAddress {};
|
||||
uint64_t SignalHandlerReturnAddress {};
|
||||
uint64_t SignalHandlerReturnAddressRT {};
|
||||
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
@@ -216,6 +217,7 @@ void Decoder::DecodeModRM_16(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
Operand->Type = DecodedOperand::OpType::SIB;
|
||||
Operand->Data.SIB.Scale = 1;
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
Operand->Data.SIB.PatchableDisp = false;
|
||||
|
||||
// Only called when ModRM.mod != 0b11
|
||||
struct Encodings {
|
||||
@@ -291,6 +293,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
// SIB
|
||||
Operand->Type = DecodedOperand::OpType::SIB;
|
||||
Operand->Data.SIB.Scale = 1 << SIB.scale;
|
||||
Operand->Data.SIB.PatchableDisp = false;
|
||||
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
{
|
||||
@@ -329,6 +332,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
auto [Literal, IsRelocation] = ReadData(4);
|
||||
Operand->Type = IsRelocation ? DecodedOperand::OpType::RIPRelativeRelocation : DecodedOperand::OpType::RIPRelative;
|
||||
Operand->Data.RIPLiteral.Value = Literal;
|
||||
Operand->Data.RIPLiteral.PatchableDisp = false;
|
||||
} else {
|
||||
// Register-direct addressing
|
||||
Operand->Type = DecodedOperand::OpType::GPRDirect;
|
||||
@@ -344,6 +348,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
Operand->Type = IsRelocation ? DecodedOperand::OpType::GPRIndirectRelocation : DecodedOperand::OpType::GPRIndirect;
|
||||
Operand->Data.GPRIndirect.GPR = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, ModRM.rm, false, false, false, false);
|
||||
Operand->Data.GPRIndirect.Displacement = Literal;
|
||||
Operand->Data.GPRIndirect.PatchableDisp = false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -496,15 +501,7 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
|
||||
auto* CurrentDest = &DecodeInst->Dest;
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ||
|
||||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RDX)) {
|
||||
// Some instructions hardcode their destination as RAX
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR =
|
||||
HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_DST_RAX) ? FEXCore::X86State::REG_RAX : FEXCore::X86State::REG_RDX;
|
||||
CurrentDest = &DecodeInst->Src[0];
|
||||
} else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
@@ -621,18 +618,6 @@ Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstI
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RAX)) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RAX;
|
||||
++CurrentSrc;
|
||||
} else if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_SRC_RCX)) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = FEXCore::X86State::REG_RCX;
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) {
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
@@ -1396,45 +1381,126 @@ bool Decoder::CheckIfCacheable(FEXCore::Core::InternalThreadState& Thread, const
|
||||
}
|
||||
|
||||
void Decoder::DetectDataMasks(uint64_t OpAddress, DecodedBlocks& Block) {
|
||||
FEXCore::X86Tables::DecodedOperand* LiteralToPatch = nullptr;
|
||||
|
||||
// cmp *, imm8 - seen varying in mono jitted code
|
||||
{
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
if ((DecodeInst->OPRaw == 0x80 || DecodeInst->OPRaw == 0x83) && ModRM.reg == 7 && LastFieldReadSize == 1) {
|
||||
for (auto& Src : DecodeInst->Src) {
|
||||
if (Src.IsLiteral()) {
|
||||
LiteralToPatch = &Src;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (LiteralToPatch && LiteralToPatch->Literal() != 0) {
|
||||
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, DataMaskType::MOV, LastFieldReadSize});
|
||||
|
||||
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
|
||||
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
|
||||
LiteralToPatch->Data.LiteralPatchable.Width = LastFieldReadSize;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (LastFieldReadSize < 4) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::X86Tables::DecodedOperand* LiteralToPatch = nullptr;
|
||||
DataMaskType Type;
|
||||
DataMaskType Type = DataMaskType::NONE;
|
||||
|
||||
// mov reg,imm
|
||||
if (DecodeInst->OP >= 0xB8 && DecodeInst->OP <= 0xBF) {
|
||||
// imm32 or imm64 at the end type instructions
|
||||
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_LITERAL_PATCHABLE) {
|
||||
for (auto& Src : DecodeInst->Src) {
|
||||
if (Src.IsLiteral()) {
|
||||
LiteralToPatch = &Src;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (DecodeInst->Dest.IsLiteral()) {
|
||||
LiteralToPatch = &DecodeInst->Dest;
|
||||
}
|
||||
|
||||
// we could filter to certain high values that are more likely to be pointers/etc?
|
||||
// const uint64_t Value = Lit->Data.Literal.Value;
|
||||
// if (LiteralToPatch && Value < 0x1000000ULL) {
|
||||
// LiteralToPatch = nullptr;
|
||||
// }
|
||||
Type = DataMaskType::MOV;
|
||||
// heuristic: if it's a small value, assume it's more likely to be part of the code around it
|
||||
bool IsMOV = (DecodeInst->OPRaw >= 0xB8 && DecodeInst->OPRaw <= 0xBF) || DecodeInst->OPRaw == 0xC7;
|
||||
if (LiteralToPatch && IsMOV && LiteralToPatch->Literal() < 0x10000ULL) {
|
||||
LiteralToPatch = nullptr;
|
||||
}
|
||||
if (LiteralToPatch && LiteralToPatch->Literal() != 0) {
|
||||
Type = DataMaskType::MOV;
|
||||
}
|
||||
}
|
||||
|
||||
// anything that has a patchable disp32
|
||||
bool TryDisp = DecodeInst->Flags & X86Tables::DecodeFlags::FLAG_DECODED_MODRM;
|
||||
if (TryDisp) {
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
{
|
||||
// todo we could handle both imm + disp with more load tracking
|
||||
bool FoundLiteral = false;
|
||||
FEXCore::X86Tables::DecodedOperand* OpToPatch = nullptr;
|
||||
for (auto& Src : DecodeInst->Src) {
|
||||
if (Src.IsLiteral()) {
|
||||
FoundLiteral = true;
|
||||
break;
|
||||
}
|
||||
if (Src.IsRIPRelative() || Src.IsGPRIndirect() || Src.IsSIB()) {
|
||||
OpToPatch = &Src;
|
||||
}
|
||||
}
|
||||
if (DecodeInst->Dest.IsLiteral()) {
|
||||
FoundLiteral = true;
|
||||
}
|
||||
if (DecodeInst->Dest.IsRIPRelative() || DecodeInst->Dest.IsGPRIndirect() || DecodeInst->Dest.IsSIB()) {
|
||||
OpToPatch = &DecodeInst->Dest;
|
||||
}
|
||||
if (!FoundLiteral && OpToPatch) {
|
||||
if (DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch == &IR::OpDispatchBuilder::NOPOp) {
|
||||
// if it's a nop disp, only mask data out, no other action required
|
||||
Type = DataMaskType::NOP;
|
||||
} else if (OpToPatch->IsRIPRelative()) {
|
||||
Type = DataMaskType::DISP;
|
||||
OpToPatch->Data.RIPLiteral.PatchableDisp = true;
|
||||
OpToPatch->Data.RIPLiteral.DispOffset = LastFieldReadOffset;
|
||||
} else if (OpToPatch->IsGPRIndirect()) {
|
||||
// filter out disp8
|
||||
if (!(ModRM.mod == 1)) {
|
||||
Type = DataMaskType::DISP;
|
||||
OpToPatch->Data.GPRIndirect.PatchableDisp = true;
|
||||
OpToPatch->Data.GPRIndirect.DispOffset = LastFieldReadOffset;
|
||||
}
|
||||
} else if (OpToPatch->IsSIB()) {
|
||||
FEXCore::X86Tables::SIBDecoded SIB;
|
||||
SIB.Hex = DecodeInst->SIB;
|
||||
// two disp32 cases here
|
||||
if (ModRM.mod == 0b10 || (ModRM.mod == 0 && SIB.base == 0b101)) {
|
||||
Type = DataMaskType::DISP;
|
||||
OpToPatch->Data.SIB.PatchableDisp = true;
|
||||
OpToPatch->Data.SIB.DispOffset = LastFieldReadOffset;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// jmp/call branches that use a literal rip-relative offset
|
||||
// some of those may be inlined by multiblock and will be cleaned up at decode end
|
||||
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_SETS_RIP && DecodeInst->Src[0].IsLiteral()) {
|
||||
if (DecodeInst->TableInfo->Flags & X86Tables::InstFlags::FLAGS_SETS_RIP && DecodeInst->Src[0].IsLiteral() && DecodeInst->Src[0].Literal() != 0) {
|
||||
LiteralToPatch = &DecodeInst->Src[0];
|
||||
Type = DataMaskType::BRANCH;
|
||||
}
|
||||
|
||||
// todo add a bunch more
|
||||
|
||||
if (LiteralToPatch) {
|
||||
if (Type != DataMaskType::NONE) {
|
||||
Block.DataMasks.push_back({OpAddress + LastFieldReadOffset, Type, LastFieldReadSize});
|
||||
|
||||
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
|
||||
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
|
||||
LiteralToPatch->Data.LiteralPatchable.Width = LastFieldReadSize;
|
||||
if (LiteralToPatch) {
|
||||
LiteralToPatch->Type = X86Tables::DecodedOperand::OpType::LiteralPatchable;
|
||||
LiteralToPatch->Data.LiteralPatchable.FieldOffset = LastFieldReadOffset;
|
||||
LiteralToPatch->Data.LiteralPatchable.Width = LastFieldReadSize;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1464,10 +1530,6 @@ void Decoder::PruneInlinedBranchDataMasks() {
|
||||
void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
// counter-intuitively, the masks are also needed for lookup on anon prefix decodes, not just stores
|
||||
bool WantsDataMasks = CTX->DiskCache.IsReadingDiskCache() || CTX->DiskCache.IsWritingDiskCache();
|
||||
// remove this if we ever fixup ValidateCode crc constant after relocations
|
||||
if (CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
WantsDataMasks = false;
|
||||
}
|
||||
|
||||
while (!FinalInstruction && (Paused || !BlocksToDecode.empty())) {
|
||||
bool Pausing = false;
|
||||
@@ -1621,9 +1683,6 @@ void Decoder::DecodeLoop(const uint8_t* _InstStream, uint64_t GuestSizePause) {
|
||||
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
|
||||
if (CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions)) {
|
||||
BlockIt->ForceFullSMCDetection = true;
|
||||
// todo abandon patching this for now, as the crc will fail and it will lock up redoing it over and over
|
||||
// we should fix the crc at relocation if this is important
|
||||
BlockIt->DataMasks.clear();
|
||||
}
|
||||
BranchTargetInMultiblockRange();
|
||||
}
|
||||
|
||||
@@ -32,7 +32,7 @@ public:
|
||||
UNIMPLEMENTED_INST,
|
||||
};
|
||||
|
||||
enum class DataMaskType : uint8_t { MOV, BRANCH };
|
||||
enum class DataMaskType : uint8_t { NONE, MOV, BRANCH, DISP, NOP };
|
||||
|
||||
struct DataMask final {
|
||||
uint64_t FieldAddress;
|
||||
|
||||
@@ -100,6 +100,7 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle)};
|
||||
|
||||
// SSE4.2 string instructions
|
||||
// NOTE: Currently unused. See VectorFallbacks.h.
|
||||
Info[Core::OPINDEX_VPCMPESTRX] = {ABIHandlers[FABI_I32_I64_I64_V128_V128_I16],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_VPCMPESTRX>::handle)};
|
||||
Info[Core::OPINDEX_VPCMPISTRX] = {ABIHandlers[FABI_I32_V128_V128_I16],
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include <cstring>
|
||||
|
||||
// NOTE: Currently unused. See VectorFallbacks.h.
|
||||
namespace FEXCore::CPU {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
|
||||
|
||||
@@ -13,29 +13,33 @@
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
// PCMPXSTRX control byte fields
|
||||
enum class AggregationOp {
|
||||
EqualAny = 0b00,
|
||||
Ranges = 0b01,
|
||||
EqualEach = 0b10,
|
||||
EqualOrdered = 0b11,
|
||||
};
|
||||
|
||||
enum class SourceData {
|
||||
U8,
|
||||
U16,
|
||||
S8,
|
||||
S16,
|
||||
};
|
||||
|
||||
enum class Polarity {
|
||||
Positive,
|
||||
Negative,
|
||||
PositiveMasked,
|
||||
NegativeMasked,
|
||||
};
|
||||
|
||||
// NB: The fallback handlers for the *PMCP*STR* instructions
|
||||
// are no longer used since we now emit inline ASM for them.
|
||||
// We preserve this as a reference implementation.
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
enum class AggregationOp {
|
||||
EqualAny = 0b00,
|
||||
Ranges = 0b01,
|
||||
EqualEach = 0b10,
|
||||
EqualOrdered = 0b11,
|
||||
};
|
||||
|
||||
enum class SourceData {
|
||||
U8,
|
||||
U16,
|
||||
S8,
|
||||
S16,
|
||||
};
|
||||
|
||||
enum class Polarity {
|
||||
Positive,
|
||||
Negative,
|
||||
PositiveMasked,
|
||||
NegativeMasked,
|
||||
};
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(uint64_t RAX, uint64_t RDX, VectorRegType lhs_v, VectorRegType rhs_v, uint16_t control) {
|
||||
__uint128_t lhs;
|
||||
memcpy(&lhs, &lhs_v, sizeof(lhs));
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -66,7 +67,17 @@ DEF_OP(EntrypointOffset) {
|
||||
|
||||
DEF_OP(PatchableGuestData) {
|
||||
auto Op = IROp->C<IR::IROp_PatchableGuestData>();
|
||||
InsertGuestPatchableDataMove(GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
|
||||
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE, GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
|
||||
}
|
||||
|
||||
DEF_OP(PatchableGuestRIP) {
|
||||
auto Op = IROp->C<IR::IROp_PatchableGuestRIP>();
|
||||
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE, GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
|
||||
}
|
||||
|
||||
DEF_OP(PatchableGuestCRC) {
|
||||
auto Op = IROp->C<IR::IROp_PatchableGuestRIP>();
|
||||
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_CRC_MOVE, GetReg(Node), Op->Value, Op->SiteAddress, (uint8_t)Op->SiteSize);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
|
||||
@@ -121,21 +121,10 @@ auto Arm64JITCore::InsertGuestPatchableRIPLiteral(uint64_t GuestRIP, uint64_t Si
|
||||
};
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestPatchableDataMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
void Arm64JITCore::InsertGuestPatchableMove(FEXCore::CPU::RelocationTypes Type, ARMEmitter::Register Reg, uint64_t Value,
|
||||
uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
Relocation MoveABI = Relocation::Default();
|
||||
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE};
|
||||
MoveABI.GuestPatchableData.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestPatchableData.ValueSize = ValueSize;
|
||||
MoveABI.GuestPatchableData.SiteAddress = SiteAddress;
|
||||
|
||||
// this might get patched on disk cache load
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Value, FEXCore::CPU::Arm64Emitter::PadType::DOPAD);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestPatchableRIPMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize) {
|
||||
Relocation MoveABI = Relocation::Default();
|
||||
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE};
|
||||
MoveABI.GuestPatchableData.Header = {.Offset = GetCursorOffset(), .Type = Type};
|
||||
MoveABI.GuestPatchableData.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestPatchableData.ValueSize = ValueSize;
|
||||
MoveABI.GuestPatchableData.SiteAddress = SiteAddress;
|
||||
|
||||
@@ -84,7 +84,8 @@ DEF_OP(ExitFunction) {
|
||||
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
if (Op->PatchSiteAddress) {
|
||||
InsertGuestPatchableRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP, Op->PatchSiteAddress, Op->PatchSiteSize);
|
||||
InsertGuestPatchableMove(RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE, EC_CALL_CHECKER_PC_REG, NewRIP, Op->PatchSiteAddress,
|
||||
Op->PatchSiteSize);
|
||||
} else {
|
||||
InsertGuestRIPMove(EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
}
|
||||
@@ -282,51 +283,11 @@ DEF_OP(CondJump) {
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
// Arguments are passed as follows:
|
||||
// X0: SyscallHandler
|
||||
// X1: ThreadState
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
|
||||
SpillStaticRegs(TMP1, {
|
||||
.GPRSpillMask = GPRSpillMask,
|
||||
.FPRSpillMask = FPRSpillMask,
|
||||
});
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerObj));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.SyscallHandlerFunc));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, STATE.R());
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
// Syscall result is in any static register that the frontend desired.
|
||||
FillStaticRegs({
|
||||
.OptionalReg = ARMEmitter::Reg::r1,
|
||||
.OptionalReg2 = ARMEmitter::Reg::r2,
|
||||
.GPRFillMask = GPRSpillMask,
|
||||
.FPRFillMask = FPRSpillMask,
|
||||
});
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegs();
|
||||
// Jump to the syscall dispatch handler. We won't return after this.
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadDispatchSyscallHandler));
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
@@ -413,27 +374,22 @@ DEF_OP(ValidateCode) {
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
auto Op = IROp->C<IR::IROp_ThreadRemoveCodeEntry>();
|
||||
|
||||
// Move the entry to ABI before saving state.
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->Entry));
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
// Store the new RIP to go to.
|
||||
str(GetReg(Op->NewRIP).X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
|
||||
// Move the entry to ABI before saving state.
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetReg(Op->EntryToInvalidate));
|
||||
|
||||
// Arguments are passed as follows:
|
||||
// X0: Thread
|
||||
// X1: RIP
|
||||
// X1: RIPToInvalidate
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadRemoveCodeEntryFromJIT));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegs();
|
||||
// Jump to the invalidate dispatch handler. We won't return after this.
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.ThreadDispatchRemoveCodeEntry));
|
||||
br(ARMEmitter::XReg::x2);
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
|
||||
@@ -36,7 +36,14 @@ $end_info$
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <optional>
|
||||
#include <type_traits>
|
||||
#include <unistd.h>
|
||||
#ifdef _WIN32
|
||||
#include <atomic>
|
||||
#else
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#endif
|
||||
|
||||
namespace {
|
||||
struct DivRem {
|
||||
@@ -63,8 +70,36 @@ LDIV(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) {
|
||||
};
|
||||
}
|
||||
|
||||
static void
|
||||
PrintValue(uint64_t Value) {
|
||||
#ifndef _WIN32
|
||||
|
||||
static std::optional<uint64_t>
|
||||
RDRANDFallback(uint64_t Reseed) {
|
||||
uint64_t Value {};
|
||||
FHU::Syscalls::getrandom(&Value, sizeof(Value), 0);
|
||||
return Value;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
// Windows does not have an equivalent to getrandom() without dynamically linking
|
||||
// bcrypt et al. Since this fallback is just for compat it does not need to be
|
||||
// cryptographic, so instead we vendor SplitMix64 as a naive fallback.
|
||||
|
||||
// Reference implementation by Sebastiano Vigna, public domain (CC0)
|
||||
// https://prng.di.unimi.it/splitmix64.c
|
||||
static std::atomic<uint64_t>
|
||||
RNGState {static_cast<uint64_t>(__builtin_readcyclecounter())};
|
||||
|
||||
static std::optional<uint64_t> RDRANDFallback(uint64_t Reseed) {
|
||||
const uint64_t State = RNGState.load(std::memory_order_relaxed) + 0x9E3779B97F4A7C15ULL;
|
||||
RNGState.store(State, std::memory_order_relaxed);
|
||||
uint64_t Value = (State ^ (State >> 30)) * 0xBF58476D1CE4E5B9ULL;
|
||||
Value = (Value ^ (Value >> 27)) * 0x94D049BB133111EBULL;
|
||||
return Value ^ (Value >> 31);
|
||||
}
|
||||
#endif
|
||||
|
||||
static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
@@ -664,6 +699,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
Ptrs.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore::ExitFunctionLink);
|
||||
Ptrs.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
Ptrs.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
Ptrs.RDRANDFallback = reinterpret_cast<uint64_t>(RDRANDFallback);
|
||||
}
|
||||
|
||||
CurrentCodeBuffer = SharedCodeBuffers.GetLatest();
|
||||
@@ -766,7 +802,7 @@ void Arm64JITCore::EmitTFCheck() {
|
||||
void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
// Used only for gdbserver at the moment
|
||||
constexpr size_t InterruptPageOffset =
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState);
|
||||
if constexpr (InterruptPageOffset <= 32760) {
|
||||
@@ -778,7 +814,7 @@ void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _WIN32
|
||||
static constexpr uint16_t SuspendMagic {0xCAFE};
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
|
||||
@@ -564,8 +564,8 @@ private:
|
||||
*/
|
||||
void InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant);
|
||||
|
||||
void InsertGuestPatchableDataMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize);
|
||||
void InsertGuestPatchableRIPMove(ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress, uint8_t ValueSize);
|
||||
void InsertGuestPatchableMove(FEXCore::CPU::RelocationTypes Type, ARMEmitter::Register Reg, uint64_t Value, uint64_t SiteAddress,
|
||||
uint8_t ValueSize);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
|
||||
@@ -309,7 +309,39 @@ DEF_OP(ProcessorID) {
|
||||
DEF_OP(RDRAND) {
|
||||
auto Op = IROp->C<IR::IROp_RDRAND>();
|
||||
|
||||
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
|
||||
if (CTX->HostFeatures.SupportsRAND) {
|
||||
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
|
||||
return;
|
||||
}
|
||||
|
||||
// Software fallback, call the host RNG generator.
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// x0 = Reseed
|
||||
// x1 = Generator
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Op->GetReseeded ? 1 : 0);
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.RDRANDFallback));
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, x1
|
||||
// std::optional<uint64_t>: value in x0, engaged flag in the low byte of x1. Match the hardware behaviour of setting Z when
|
||||
// no number was produced.
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1);
|
||||
tst(ARMEmitter::Size::i64Bit, TMP2, 0xFF);
|
||||
}
|
||||
|
||||
DEF_OP(Yield) {
|
||||
|
||||
@@ -37,6 +37,10 @@ enum class RelocationTypes : uint32_t {
|
||||
// Like PATCHABLE_RIP_LITERAL but puts it in a register
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_RIP_MOVE,
|
||||
|
||||
// Patchable guest CRC
|
||||
// Aligned to struct RelocGuestPatchableData
|
||||
RELOC_GUEST_PATCHABLE_CRC_MOVE,
|
||||
};
|
||||
|
||||
struct FEX_PACKED RelocationHeader final {
|
||||
|
||||
@@ -1112,8 +1112,10 @@ DEF_OP(VAddP) {
|
||||
// pairwise addition, the SVE version actually interleaves the
|
||||
// results of the pairwise addition (gross!), so we need to undo that.
|
||||
addp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
|
||||
// Extract the upper half first, since Dst may alias the LHS.
|
||||
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
|
||||
// Merge upper half with lower half.
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
|
||||
@@ -2167,6 +2169,46 @@ DEF_OP(VCMPGT) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUCMPGT) {
|
||||
const auto Op = IROp->C<IR::IROp_VUCMPGT>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSizePair16(IROp);
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
const auto ComparePred = ARMEmitter::PReg::p0;
|
||||
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
// General idea is to compare for unsigned greater-than, bitwise NOT
|
||||
// the valid values, then ORR the NOTed values with the original
|
||||
// values to form entries that are all 1s.
|
||||
cmphi(SubRegSize.Vector, ComparePred, Mask, Vector1.Z(), Vector2.Z());
|
||||
not_(SubRegSize.Vector, VTMP1.Z(), ComparePred.Merging(), Vector1.Z());
|
||||
movprfx(SubRegSize.Vector, Dst.Z(), ComparePred.Zeroing(), Vector1.Z());
|
||||
orr(SubRegSize.Vector, Dst.Z(), ComparePred.Merging(), Dst.Z(), VTMP1.Z());
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
cmhi(SubRegSize.Scalar, Dst, Vector1, Vector2);
|
||||
} else {
|
||||
cmhi(SubRegSize.Vector, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VCMPGTZ) {
|
||||
const auto Op = IROp->C<IR::IROp_VCMPGTZ>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -3833,6 +3875,171 @@ DEF_OP(VMul) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUSDot) {
|
||||
///< Dest = Acc + dot(Vector1 (unsigned 8-bit), Vector2 (signed 8-bit))
|
||||
// Matches:
|
||||
// - SVE - USDOT
|
||||
// - ASIMD - USDOT
|
||||
const auto Op = IROp->C<IR::IROp_VUSDot>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Acc = GetVReg(Op->Acc);
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// USDOT accumulates in to its destination register,
|
||||
// so we need to emit a move if Acc != Dst
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != Acc) {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
DestTmp = Dst;
|
||||
} else {
|
||||
DestTmp = VTMP1;
|
||||
}
|
||||
}
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Z(), Acc.Z());
|
||||
}
|
||||
|
||||
usdot(DestTmp.Z(), Vector1.Z(), Vector2.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Q(), Acc.Q());
|
||||
}
|
||||
|
||||
usdot(DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSDot) {
|
||||
///< Dest = Acc + dot(Vector1 (signed 8-bit), Vector2 (signed 8-bit))
|
||||
// Matches:
|
||||
// - SVE - SDOT
|
||||
// - ASIMD - SDOT
|
||||
const auto Op = IROp->C<IR::IROp_VSDot>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Acc = GetVReg(Op->Acc);
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// SDOT accumulates in to its destination register,
|
||||
// so we need to emit a move if Acc != Dst
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != Acc) {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
DestTmp = Dst;
|
||||
} else {
|
||||
DestTmp = VTMP1;
|
||||
}
|
||||
}
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Z(), Acc.Z());
|
||||
}
|
||||
|
||||
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Z(), Vector1.Z(), Vector2.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Q(), Acc.Q());
|
||||
}
|
||||
|
||||
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSAddLP) {
|
||||
const auto Op = IROp->C<IR::IROp_VSAddLP>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// SVE only has the accumulating form, so accumulate in to a zeroed register.
|
||||
// Zero a temporary instead if Dst aliases the source.
|
||||
const auto DestTmp = Dst == Vector ? VTMP1 : Dst;
|
||||
dup_imm(SubRegSize, DestTmp.Z(), 0);
|
||||
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
saddlp(SubRegSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSAdALP) {
|
||||
const auto Op = IROp->C<IR::IROp_VSAdALP>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Acc = GetVReg(Op->Acc);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
// SADALP accumulates in to its destination register,
|
||||
// so we need to emit a move if Acc != Dst
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != Acc) {
|
||||
DestTmp = Dst != Vector ? Dst : VTMP1;
|
||||
}
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Z(), Acc.Z());
|
||||
}
|
||||
|
||||
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (Dst == Vector) {
|
||||
// ASIMD has the non-accumulating form, which is cheaper than shuffling through a temporary.
|
||||
saddlp(SubRegSize, VTMP1.Q(), Vector.Q());
|
||||
add(SubRegSize, Dst.Q(), Acc.Q(), VTMP1.Q());
|
||||
return;
|
||||
}
|
||||
|
||||
if (Dst != Acc) {
|
||||
mov(Dst.Q(), Acc.Q());
|
||||
}
|
||||
|
||||
sadalp(SubRegSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUMull) {
|
||||
const auto Op = IROp->C<IR::IROp_VUMull>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -59,12 +59,6 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
|
||||
|
||||
FlushRegisterCache();
|
||||
_Syscall();
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_BLOCK_END) {
|
||||
// RIP could have been updated after coming back from the Syscall.
|
||||
NewRIP = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, rip));
|
||||
ExitFunction(NewRIP);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
@@ -227,10 +221,10 @@ void OpDispatchBuilder::SecondaryALUOp(OpcodeArgs) {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
ALUOp(Op, IROp, AtomicIROp, 1);
|
||||
ALUOp(Op, IROp, AtomicIROp, 1, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
@@ -245,6 +239,8 @@ void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
|
||||
Ref DestMem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Before = _AtomicFetchAdd(Size, ALUOp, DestMem);
|
||||
} else if (DestRAX) {
|
||||
Before = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
@@ -261,12 +257,14 @@ void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
Result = CalculateFlags_ADC(Size, Before, Src);
|
||||
}
|
||||
|
||||
if (!DestIsLockedMem(Op)) {
|
||||
if (DestRAX) {
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Result, OpSizeFromDst(Op));
|
||||
} else if (!DestIsLockedMem(Op)) {
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SBBOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
void OpDispatchBuilder::SBBOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
@@ -282,13 +280,17 @@ void OpDispatchBuilder::SBBOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
Ref DestMem = MakeSegmentAddress(Op, Op->Dest);
|
||||
auto SrcPlusCF = IncrementByCarry(OpSize, Src);
|
||||
Before = _AtomicFetchSub(Size, SrcPlusCF, DestMem);
|
||||
} else if (DestRAX) {
|
||||
Before = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
|
||||
Result = CalculateFlags_SBB(Size, Before, Src);
|
||||
|
||||
if (!DestIsLockedMem(Op)) {
|
||||
if (DestRAX) {
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Result, OpSizeFromDst(Op));
|
||||
} else if (!DestIsLockedMem(Op)) {
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
}
|
||||
@@ -298,7 +300,8 @@ void OpDispatchBuilder::SALCOp(OpcodeArgs) {
|
||||
|
||||
auto Result = NZCVSelect(OpSize::i32Bit, CondClass::UGE /* CF = 1 */, _InlineConstant(0xffffffff), _InlineConstant(0));
|
||||
|
||||
StoreResultGPR(Op, Result);
|
||||
// This inserts in to the low 8-bits.
|
||||
StoreGPRRegister(X86State::REG_RAX, Result, OpSizeFromDst(Op));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHOp(OpcodeArgs) {
|
||||
@@ -773,11 +776,11 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
OpSize = OpSize::i32Bit;
|
||||
}
|
||||
|
||||
uint64_t Target = Op->PC + Op->InstSize + Op->Src[1].Literal();
|
||||
uint64_t Target = Op->PC + Op->InstSize + Op->Src[0].Literal();
|
||||
|
||||
Ref CondReg = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
Ref CondReg = LoadGPRRegister(X86State::REG_RCX, SrcSize);
|
||||
CondReg = Sub(OpSize, CondReg, 1);
|
||||
StoreResultGPR(Op, Op->Src[0], CondReg);
|
||||
StoreGPRRegister(X86State::REG_RCX, CondReg, SrcSize);
|
||||
|
||||
// If LOOPE then jumps to target if RCX != 0 && ZF == 1
|
||||
// If LOOPNE then jumps to target if RCX != 0 && ZF == 0
|
||||
@@ -806,7 +809,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
StartNewBlock();
|
||||
|
||||
// Store the new RIP
|
||||
ExitRelocatedPC(Op, Op->Src[1].Literal());
|
||||
ExitRelocatedPC(Op, Op->Src[0].Literal());
|
||||
}
|
||||
|
||||
// Failure to take branch
|
||||
@@ -961,11 +964,16 @@ void OpDispatchBuilder::RETFARIndirectOp(OpcodeArgs) {
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::TESTOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
void OpDispatchBuilder::TESTOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX) {
|
||||
// TEST is an instruction that does an AND between the sources
|
||||
// Result isn't stored in result, only writes to flags
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Dest {};
|
||||
if (DestRAX) {
|
||||
Dest = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
@@ -1071,26 +1079,60 @@ void OpDispatchBuilder::MOVZXOp(OpcodeArgs) {
|
||||
StoreResultGPR(Op, Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CMPOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
void OpDispatchBuilder::CMPOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX) {
|
||||
// CMP is an instruction that does a SUB between the sources
|
||||
// Result isn't stored in result, only writes to flags
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Dest {};
|
||||
if (DestRAX) {
|
||||
Dest = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Dest, Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CQOOp(OpcodeArgs) {
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Size = OpSizeFromSrc(Op);
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX, Size, 0, true);
|
||||
Ref Upper = _Sbfe(std::max(OpSize::i32Bit, Size), 1, GetSrcBitSize(Op) - 1, Src);
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RDX, Upper, Size);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Upper);
|
||||
std::optional<Ref> OpDispatchBuilder::XCHGOpImpl(OpcodeArgs, Ref Src) {
|
||||
if (DestIsMem(Op)) {
|
||||
HandledLock = (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0;
|
||||
|
||||
Ref Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
if (IsMonoBackpatcherBlock) {
|
||||
_MonoBackpatcherWrite(OpSizeFromSrc(Op), Src, Dest);
|
||||
} else {
|
||||
return _AtomicSwap(OpSizeFromSrc(Op), Src, Dest);
|
||||
}
|
||||
return std::nullopt;
|
||||
} else {
|
||||
// AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult.
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Swap the contents
|
||||
// Order matters here since we don't want to swap context contents for one that effects the other
|
||||
StoreResultGPR(Op, Op->Dest, Src);
|
||||
return Dest;
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult.
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Res = XCHGOpImpl(Op, Src);
|
||||
if (Res) {
|
||||
StoreResultGPR(Op, Op->Src[0], *Res);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XCHGRAXOp(OpcodeArgs) {
|
||||
// Load both the source and the destination
|
||||
if (Op->OP == 0x90 && Op->Src[0].IsGPR() && Op->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX && Op->Dest.IsGPR() &&
|
||||
Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
if (Op->Dest.IsGPR() && Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
// This is one heck of a sucky special case
|
||||
// If we are the 0x90 XCHG opcode (Meaning source is GPR RAX)
|
||||
// and destination register is ALSO RAX
|
||||
@@ -1118,25 +1160,10 @@ void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
// AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult.
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
if (DestIsMem(Op)) {
|
||||
HandledLock = (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0;
|
||||
|
||||
Ref Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
if (IsMonoBackpatcherBlock) {
|
||||
_MonoBackpatcherWrite(OpSizeFromSrc(Op), Src, Dest);
|
||||
} else {
|
||||
auto Result = _AtomicSwap(OpSizeFromSrc(Op), Src, Dest);
|
||||
StoreResultGPR(Op, Op->Src[0], Result);
|
||||
}
|
||||
} else {
|
||||
// AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult.
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Swap the contents
|
||||
// Order matters here since we don't want to swap context contents for one that effects the other
|
||||
StoreResultGPR(Op, Op->Dest, Src);
|
||||
StoreResultGPR(Op, Op->Src[0], Dest);
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX, OpSize::iInvalid, 0, true);
|
||||
auto Res = XCHGOpImpl(Op, Src);
|
||||
if (Res) {
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, *Res, OpSizeFromDst(Op));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1147,7 +1174,8 @@ void OpDispatchBuilder::CDQOp(OpcodeArgs) {
|
||||
|
||||
Src = _Sbfe(DstSize <= OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, IR::OpSizeAsBits(SrcSize), 0, Src);
|
||||
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Src, DstSize);
|
||||
// This inserts in to the low 16-bits.
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Src, DstSize);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SAHFOp(OpcodeArgs) {
|
||||
@@ -1318,27 +1346,25 @@ void OpDispatchBuilder::MOVOffsetOp(OpcodeArgs) {
|
||||
// Source is memory(literal)
|
||||
// Dest is GPR
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.ForceLoad = true});
|
||||
StoreResultGPR(Op, Op->Dest, Src);
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Src, OpSizeFromDst(Op));
|
||||
break;
|
||||
}
|
||||
case 0xA2:
|
||||
case 0xA3: {
|
||||
// Source is GPR
|
||||
// Dest is memory(literal)
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX);
|
||||
|
||||
// This one is a bit special since the destination is a literal
|
||||
// So the destination gets stored in Src[1]
|
||||
StoreResultGPR(Op, Op->Src[1], Src);
|
||||
StoreResultGPR(Op, Op->Src[0], Src);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CPUIDOp(OpcodeArgs) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
Ref Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], GPRSize, Op->Flags);
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Leaf = LoadGPRRegister(X86State::REG_RCX);
|
||||
|
||||
Ref RAX = _AllocateGPR(false);
|
||||
@@ -1380,7 +1406,7 @@ void OpDispatchBuilder::XGetBVOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::SHLOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
Ref Result = _Lshl(Size == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Dest, Src);
|
||||
HandleShift(Op, Result, Dest, ShiftType::LSL, Src);
|
||||
@@ -1402,7 +1428,7 @@ void OpDispatchBuilder::SHLImmediateOp(OpcodeArgs, bool SHL1Bit) {
|
||||
void OpDispatchBuilder::SHROp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= OpSize::i32Bit});
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
auto ALUOp = _Lshr(std::max(OpSize::i32Bit, Size), Dest, Src);
|
||||
HandleShift(Op, ALUOp, Dest, ShiftType::LSR, Src);
|
||||
@@ -1431,7 +1457,7 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
|
||||
// Allow garbage on the shift, we're masking it anyway.
|
||||
Ref Shift = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Shift = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op.
|
||||
if (Size == 64) {
|
||||
@@ -1502,7 +1528,7 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) {
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
|
||||
Ref Shift = LoadGPRRegister(X86State::REG_RCX);
|
||||
Ref Shift = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
const auto Size = GetDstBitSize(Op);
|
||||
|
||||
@@ -1580,7 +1606,7 @@ void OpDispatchBuilder::ASHROp(OpcodeArgs, bool Immediate, bool SHR1Bit) {
|
||||
CalculateDeferredFlags();
|
||||
StoreResultGPR(Op, Result);
|
||||
} else {
|
||||
auto Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
Ref Result = _Ashr(OpSize, Dest, Src);
|
||||
|
||||
HandleShift(Op, Result, Dest, ShiftType::ASR, Src);
|
||||
@@ -1604,7 +1630,7 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
UnmaskedConst = GetConstantShift(Op, Is1Bit);
|
||||
UnmaskedSrc = ARef(UnmaskedConst);
|
||||
} else {
|
||||
UnmaskedSrc = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
UnmaskedSrc = ARef(LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true));
|
||||
}
|
||||
auto Src = UnmaskedSrc.And(Mask);
|
||||
|
||||
@@ -1994,11 +2020,11 @@ void OpDispatchBuilder::RCROp8x1Bit(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(_XorShift(OpSize::i32Bit, Res, Res, ShiftType::LSR, 1), SizeBit - 2, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::RCROp(OpcodeArgs, bool UseRCX) {
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
if (Size == 8 || Size == 16) {
|
||||
RCRSmallerOp(Op);
|
||||
RCRSmallerOp(Op, UseRCX);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -2008,49 +2034,54 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
CalculateDeferredFlags();
|
||||
const auto OpSize = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src), &Const)) {
|
||||
Const &= Mask;
|
||||
if (!Const) {
|
||||
ZeroShiftResult(Op);
|
||||
if (!UseRCX) {
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src), &Const)) {
|
||||
Const &= Mask;
|
||||
if (!Const) {
|
||||
ZeroShiftResult(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Res = Src >> Shift
|
||||
Ref Res = _Lshr(OpSize, Dest, Src);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// Constant folded version of the above, with fused shifts.
|
||||
if (Const > 1) {
|
||||
Res = _Orlshl(OpSize, Res, Dest, Size + 1 - Const);
|
||||
}
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source.
|
||||
SetCFDirect(Dest, Const - 1, true);
|
||||
|
||||
// Since shift != 0 we can inject the CF
|
||||
Res = _Orlshl(OpSize, Res, CF, Size - Const);
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (Const == 1) {
|
||||
auto Xor = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, Size - 2, true);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Res);
|
||||
return;
|
||||
}
|
||||
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Res = Src >> Shift
|
||||
Ref Res = _Lshr(OpSize, Dest, Src);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// Constant folded version of the above, with fused shifts.
|
||||
if (Const > 1) {
|
||||
Res = _Orlshl(OpSize, Res, Dest, Size + 1 - Const);
|
||||
}
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source.
|
||||
SetCFDirect(Dest, Const - 1, true);
|
||||
|
||||
// Since shift != 0 we can inject the CF
|
||||
Res = _Orlshl(OpSize, Res, CF, Size - Const);
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (Const == 1) {
|
||||
auto Xor = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Xor, Size - 2, true);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Res);
|
||||
return;
|
||||
}
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
Ref SrcMasked = _And(OpSize, Src, _InlineConstant(Mask));
|
||||
|
||||
Calculate_ShiftVariable(
|
||||
Op, SrcMasked,
|
||||
[this, Op, Size, OpSize]() {
|
||||
// Rematerialize loads to avoid crossblock liveness
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
// Res = Src >> Shift
|
||||
@@ -2086,21 +2117,29 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
OpSizeFromSrc(Op) == OpSize::i32Bit ? std::make_optional(&OpDispatchBuilder::ZeroShiftResult) : std::nullopt);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs, bool UseRCX) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
auto GetShift = [this, Op, UseRCX]() {
|
||||
if (UseRCX) {
|
||||
auto Src = ARef(LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true));
|
||||
return Src.And(0x1F);
|
||||
} else {
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
return Src.And(0x1F);
|
||||
}
|
||||
};
|
||||
|
||||
auto Src = GetShift();
|
||||
|
||||
// CF only changes if we actually shifted. OF undefined if we didn't shift.
|
||||
// The result is unchanged if we didn't shift. So branch over the whole thing.
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() {
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size, GetShift]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
auto Src = GetShift();
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
@@ -2210,11 +2249,11 @@ void OpDispatchBuilder::RCLOp1Bit(OpcodeArgs) {
|
||||
StoreResultGPR(Op, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::RCLOp(OpcodeArgs, bool UseRCX) {
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
if (Size == 8 || Size == 16) {
|
||||
RCLSmallerOp(Op);
|
||||
RCLSmallerOp(Op, UseRCX);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -2223,50 +2262,55 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
const auto OpSize = OpSizeFromSrc(Op);
|
||||
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src), &Const)) {
|
||||
Const &= Mask;
|
||||
if (!Const) {
|
||||
ZeroShiftResult(Op);
|
||||
if (!UseRCX) {
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src), &Const)) {
|
||||
Const &= Mask;
|
||||
if (!Const) {
|
||||
ZeroShiftResult(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
// Res = Src << Shift
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Res = _Lshl(OpSize, Dest, Src);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// Res |= (Src << (Size - Shift + 1));
|
||||
if (Const > 1) {
|
||||
Res = _Orlshr(OpSize, Res, Dest, Size + 1 - Const);
|
||||
}
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
SetCFDirect(Dest, Size - Const, true);
|
||||
|
||||
// Since Shift != 0 we can inject the CF
|
||||
Res = _Orlshl(OpSize, Res, CF, Const - 1);
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (Const == 1) {
|
||||
auto NewOF = _Xor(OpSize, Res, Dest);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, Size - 1, true);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Res);
|
||||
return;
|
||||
}
|
||||
|
||||
// Res = Src << Shift
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Res = _Lshl(OpSize, Dest, Src);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
// Res |= (Src << (Size - Shift + 1));
|
||||
if (Const > 1) {
|
||||
Res = _Orlshr(OpSize, Res, Dest, Size + 1 - Const);
|
||||
}
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
SetCFDirect(Dest, Size - Const, true);
|
||||
|
||||
// Since Shift != 0 we can inject the CF
|
||||
Res = _Orlshl(OpSize, Res, CF, Const - 1);
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (Const == 1) {
|
||||
auto NewOF = _Xor(OpSize, Res, Dest);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, Size - 1, true);
|
||||
}
|
||||
|
||||
StoreResultGPR(Op, Res);
|
||||
return;
|
||||
}
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
Ref SrcMasked = _And(OpSize, Src, _InlineConstant(Mask));
|
||||
|
||||
Calculate_ShiftVariable(
|
||||
Op, SrcMasked,
|
||||
[this, Op, Size, OpSize]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src = LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true);
|
||||
|
||||
// Res = Src << Shift
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
@@ -2301,21 +2345,29 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
OpSizeFromSrc(Op) == OpSize::i32Bit ? std::make_optional(&OpDispatchBuilder::ZeroShiftResult) : std::nullopt);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs, bool UseRCX) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
auto GetShift = [this, Op, UseRCX]() {
|
||||
if (UseRCX) {
|
||||
auto Src = ARef(LoadGPRRegister(X86State::REG_RCX, OpSize::iInvalid, 0, true));
|
||||
return Src.And(0x1F);
|
||||
} else {
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
return Src.And(0x1F);
|
||||
}
|
||||
};
|
||||
|
||||
auto Src = GetShift();
|
||||
|
||||
// CF only changes if we actually shifted. OF undefined if we didn't shift.
|
||||
// The result is unchanged if we didn't shift. So branch over the whole thing.
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() {
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size, GetShift]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
auto Src = GetShift();
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
@@ -3173,7 +3225,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
|
||||
if (!Repeat) {
|
||||
// Src is used only for a store of the same size so allow garbage
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX, Size, 0, true);
|
||||
|
||||
// Only ES prefix
|
||||
Ref Dest = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
@@ -3192,7 +3244,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
// FEX doesn't support partial faulting REP instructions.
|
||||
// Converting this to a `MemSet` IR op optimizes this quite significantly in our codegen.
|
||||
// If FEX is to gain support for faulting REP instructions, then this implementation needs to change significantly.
|
||||
Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Ref Dest = LoadGPRRegister(X86State::REG_RDI);
|
||||
|
||||
// Only ES prefix
|
||||
@@ -3429,7 +3481,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
|
||||
auto Src = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size);
|
||||
|
||||
StoreResultGPR(Op, Src);
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Src, Size);
|
||||
|
||||
// Offset the pointer
|
||||
Ref TailDest_RSI = OffsetByDir(Src_RSI, IR::OpSizeToSize(Size));
|
||||
@@ -3443,7 +3495,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size, AddrSize](int32_t PtrDir) {
|
||||
ForeachDirection([this, Size, AddrSize](int32_t PtrDir) {
|
||||
// XXX: Theoretically LODS could be optimized to
|
||||
// RSI += {-}(RCX * Size)
|
||||
// RAX = [RSI - Size]
|
||||
@@ -3476,7 +3528,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
|
||||
auto Src = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size);
|
||||
|
||||
StoreResultGPR(Op, Src);
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Src, Size);
|
||||
|
||||
Ref TailCounter = LoadGPRRegister(X86State::REG_RCX);
|
||||
Ref TailDest_RSI = LoadGPRRegister(X86State::REG_RSI);
|
||||
@@ -3523,7 +3575,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
Ref Src_RDI = LoadGPRRegister(X86State::REG_RDI, AddrSize);
|
||||
Ref Dest_RDI = AppendSegmentOffset(Src_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src1 = LoadGPRRegister(X86State::REG_RAX, Size, 0, true);
|
||||
auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size);
|
||||
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Src1, Src2);
|
||||
@@ -3566,7 +3618,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
Ref Src_RDI = LoadGPRRegister(X86State::REG_RDI, AddrSize);
|
||||
Ref Dest_RDI = AppendSegmentOffset(Src_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true);
|
||||
|
||||
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Src1 = LoadGPRRegister(X86State::REG_RAX, Size, 0, true);
|
||||
auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size);
|
||||
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Src1, Src2);
|
||||
@@ -4308,12 +4360,23 @@ AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, con
|
||||
if (Operand.IsGPRIndirectRelocation()) {
|
||||
A.Base = Add(GPRSize, _EntrypointOffset(GPRSize, Operand.Data.GPRIndirect.Displacement), A.Base);
|
||||
} else {
|
||||
A.Offset = static_cast<int32_t>(Operand.Data.GPRIndirect.Displacement);
|
||||
if (Operand.Data.GPRIndirect.PatchableDisp) {
|
||||
A.Base = Add(GPRSize, A.Base,
|
||||
_PatchableGuestData(OpSize::i64Bit, static_cast<int32_t>(Operand.Data.GPRIndirect.Displacement),
|
||||
Op->PC + Operand.Data.GPRIndirect.DispOffset, 4));
|
||||
} else {
|
||||
A.Offset = static_cast<int32_t>(Operand.Data.GPRIndirect.Displacement);
|
||||
}
|
||||
}
|
||||
A.NonTSO |= IsNonTSOReg(AccessType, Operand.Data.GPRIndirect.GPR);
|
||||
} else if (Operand.IsRIPRelative() || Operand.IsRIPRelativeRelocation()) {
|
||||
if (Is64BitMode) {
|
||||
A.Base = GetRelocatedPC(Op, static_cast<int32_t>(Operand.Data.RIPLiteral.Value));
|
||||
if (Operand.IsRIPRelative() && Operand.Data.RIPLiteral.PatchableDisp) {
|
||||
A.Base = _PatchableGuestRIP(OpSize::i64Bit, Op->PC + Op->InstSize + static_cast<int32_t>(Operand.Data.RIPLiteral.Value),
|
||||
Op->PC + Operand.Data.RIPLiteral.DispOffset, 4);
|
||||
} else {
|
||||
A.Base = GetRelocatedPC(Op, static_cast<int32_t>(Operand.Data.RIPLiteral.Value));
|
||||
}
|
||||
} else {
|
||||
// 32bit this isn't RIP relative but instead absolute
|
||||
if (Operand.IsRIPRelativeRelocation()) {
|
||||
@@ -4350,6 +4413,14 @@ AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, con
|
||||
} else {
|
||||
A.Base = EPOffset;
|
||||
}
|
||||
} else if (Operand.Data.SIB.PatchableDisp) {
|
||||
Ref PatchedDisp =
|
||||
_PatchableGuestData(OpSize::i64Bit, static_cast<int32_t>(Operand.Data.SIB.Offset), Op->PC + Operand.Data.SIB.DispOffset, 4);
|
||||
if (A.Base) {
|
||||
A.Base = Add(OpSize::i64Bit, A.Base, PatchedDisp);
|
||||
} else {
|
||||
A.Base = PatchedDisp;
|
||||
}
|
||||
} else {
|
||||
A.Offset = static_cast<int32_t>(Operand.Data.SIB.Offset);
|
||||
}
|
||||
@@ -4621,7 +4692,7 @@ void OpDispatchBuilder::MOVGPRNTOp(OpcodeArgs) {
|
||||
StoreResultGPR(Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx) {
|
||||
void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx, bool DestRAX) {
|
||||
// On x86, the canonical way to zero a register is XOR with itself. Detect and
|
||||
// emit optimal arm64 assembly.
|
||||
if (!DestIsLockedMem(Op) && ALUIROp == FEXCore::IR::IROps::OP_XOR && Op->Dest.IsGPR() && Op->Src[SrcIdx].IsGPR() &&
|
||||
@@ -4657,7 +4728,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
// promoting to a full size operation that preserves the upper bits.
|
||||
uint64_t Const;
|
||||
bool IsConst = IsValueConstant(WrapNode(Src), &Const);
|
||||
if (Size < OpSize::i32Bit && !DestIsLockedMem(Op) && Op->Dest.IsGPR() && !Op->Dest.Data.GPR.HighBits && IsConst &&
|
||||
if (Size < OpSize::i32Bit && !DestIsLockedMem(Op) && ((Op->Dest.IsGPR() && !Op->Dest.Data.GPR.HighBits) || DestRAX) && IsConst &&
|
||||
(ALUIROp == IR::IROps::OP_XOR || ALUIROp == IR::IROps::OP_OR || ALUIROp == IR::IROps::OP_ANDWITHFLAGS)) {
|
||||
|
||||
RoundedSize = ResultSize = GetGPROpSize();
|
||||
@@ -4686,6 +4757,8 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
Ref DestMem = MakeSegmentAddress(Op, Op->Dest);
|
||||
DeriveOp(FetchOp, AtomicFetchOp, _AtomicFetchAdd(Size, Src, DestMem));
|
||||
Dest = FetchOp;
|
||||
} else if (DestRAX) {
|
||||
Dest = LoadGPRRegister(X86State::REG_RAX, OpSizeFromSrc(Op), 0, true);
|
||||
} else {
|
||||
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
}
|
||||
@@ -4720,7 +4793,9 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
default: break;
|
||||
}
|
||||
|
||||
if (!DestIsLockedMem(Op)) {
|
||||
if (DestRAX) {
|
||||
StoreGPRResultWithZExtSemantics(X86State::REG_RAX, Result, ResultSize);
|
||||
} else if (!DestIsLockedMem(Op)) {
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, ResultSize, OpSize::iInvalid, MemoryAccessType::DEFAULT);
|
||||
}
|
||||
}
|
||||
@@ -4977,8 +5052,7 @@ void OpDispatchBuilder::CLZeroOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref DestMem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
_CacheLineZero(DestMem);
|
||||
_CacheLineZero(LoadGPRRegister(X86State::REG_RAX));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::Prefetch(OpcodeArgs, bool ForStore, bool Stream, uint8_t Level) {
|
||||
@@ -5013,6 +5087,11 @@ void OpDispatchBuilder::RDTSCPOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RDPIDOp(OpcodeArgs) {
|
||||
if (CTX->HostFeatures.HostType != FEXCore::HostFeatures::HostTypeEnum::Linux && !CTX->HostFeatures.SupportsCPUIndexInTPIDRRO) {
|
||||
// RDTSCP is unsupported on Win32 platforms if TPIDRRO isn't supported.
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
StoreResultGPR(Op, _ProcessorID());
|
||||
}
|
||||
|
||||
@@ -5040,7 +5119,7 @@ void OpDispatchBuilder::CRC32(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RDRANDOp(OpcodeArgs, bool Reseed) {
|
||||
if (!CTX->HostFeatures.SupportsRAND) {
|
||||
if (!CTX->HostFeatures.SupportsRAND && !CTX->SoftwareRNGEnabled()) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -363,7 +363,8 @@ public:
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedNoNopOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs, bool IsAVX);
|
||||
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
|
||||
void ALURAXOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp);
|
||||
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx, bool DestRAX);
|
||||
void LSLOp(OpcodeArgs);
|
||||
void INTOp(OpcodeArgs);
|
||||
void SyscallOp(OpcodeArgs, bool IsSyscallInst);
|
||||
@@ -374,8 +375,8 @@ public:
|
||||
void IRETOp(OpcodeArgs);
|
||||
void CallbackReturnOp(OpcodeArgs);
|
||||
void SecondaryALUOp(OpcodeArgs);
|
||||
void ADCOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void SBBOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void ADCOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
|
||||
void SBBOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
|
||||
void SALCOp(OpcodeArgs);
|
||||
void PUSHOp(OpcodeArgs);
|
||||
void PUSHREGOp(OpcodeArgs);
|
||||
@@ -395,16 +396,18 @@ public:
|
||||
void JUMPFARIndirectOp(OpcodeArgs);
|
||||
void CALLFARIndirectOp(OpcodeArgs);
|
||||
void RETFARIndirectOp(OpcodeArgs);
|
||||
void TESTOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void TESTOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
|
||||
void ARPLOp(OpcodeArgs);
|
||||
void MOVSXDOp(OpcodeArgs);
|
||||
void MOVSXOp(OpcodeArgs);
|
||||
void MOVZXOp(OpcodeArgs);
|
||||
void CMPOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void CMPOp(OpcodeArgs, uint32_t SrcIndex, bool DestRAX);
|
||||
void SETccOp(OpcodeArgs);
|
||||
void CQOOp(OpcodeArgs);
|
||||
void CDQOp(OpcodeArgs);
|
||||
std::optional<Ref> XCHGOpImpl(OpcodeArgs, Ref Src);
|
||||
void XCHGOp(OpcodeArgs);
|
||||
void XCHGRAXOp(OpcodeArgs);
|
||||
void SAHFOp(OpcodeArgs);
|
||||
void LAHFOp(OpcodeArgs);
|
||||
void MOVSegOp(OpcodeArgs, bool ToSeg);
|
||||
@@ -426,11 +429,11 @@ public:
|
||||
void RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool Is1Bit);
|
||||
void RCROp1Bit(OpcodeArgs);
|
||||
void RCROp8x1Bit(OpcodeArgs);
|
||||
void RCROp(OpcodeArgs);
|
||||
void RCRSmallerOp(OpcodeArgs);
|
||||
void RCROp(OpcodeArgs, bool UseRCX);
|
||||
void RCRSmallerOp(OpcodeArgs, bool UseRCX);
|
||||
void RCLOp1Bit(OpcodeArgs);
|
||||
void RCLOp(OpcodeArgs);
|
||||
void RCLSmallerOp(OpcodeArgs);
|
||||
void RCLOp(OpcodeArgs, bool UseRCX);
|
||||
void RCLSmallerOp(OpcodeArgs, bool UseRCX);
|
||||
|
||||
void BTOp(OpcodeArgs, uint32_t SrcIndex, enum BTAction Action);
|
||||
|
||||
@@ -665,6 +668,8 @@ public:
|
||||
|
||||
void VPMADDUBSWOp(OpcodeArgs);
|
||||
void VPMADDWDOp(OpcodeArgs);
|
||||
void VPDPBUSDOp(OpcodeArgs, bool Saturating);
|
||||
void VPDPWSSDOp(OpcodeArgs, bool Saturating);
|
||||
|
||||
void VPMASKMOVOp(OpcodeArgs, bool IsStore);
|
||||
|
||||
@@ -741,7 +746,7 @@ public:
|
||||
void X87FLDCW(OpcodeArgs);
|
||||
void X87FNSAVE(OpcodeArgs);
|
||||
void X87FNSTENV(OpcodeArgs);
|
||||
void X87FNSTSW(OpcodeArgs);
|
||||
void X87FNSTSW(OpcodeArgs, bool DestRAX);
|
||||
void X87FRSTOR(OpcodeArgs);
|
||||
void X87FSTCW(OpcodeArgs);
|
||||
void X87FXAM(OpcodeArgs);
|
||||
@@ -1036,6 +1041,9 @@ public:
|
||||
|
||||
void AVX128_VPMADDUBSW(OpcodeArgs);
|
||||
void AVX128_VPMADDWD(OpcodeArgs);
|
||||
void AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper);
|
||||
void AVX128_VPDPBUSD(OpcodeArgs, bool Saturating);
|
||||
void AVX128_VPDPWSSD(OpcodeArgs, bool Saturating);
|
||||
|
||||
void AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
@@ -1384,6 +1392,11 @@ private:
|
||||
const X86Tables::DecodedOperand& Imm, bool IsAVX);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask, bool IsAVX);
|
||||
Ref PCMPXSTRXSaturateExplicitLength(IR::OpSize ElementSize, IR::OpSize LengthSize, Ref RawLength);
|
||||
Ref PCMPXSTRXEqualAny(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements);
|
||||
Ref PCMPXSTRXRanges(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements, bool IsSigned);
|
||||
Ref PCMPXSTRXEqualEach(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements);
|
||||
Ref PCMPXSTRXEqualOrdered(IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1Length, Ref Src2Length, Ref Src1ValidElements, Ref Indices);
|
||||
|
||||
Ref PHADDSOpImpl(OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
@@ -1400,6 +1413,10 @@ private:
|
||||
|
||||
Ref PMADDUBSWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
|
||||
|
||||
Ref VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
|
||||
|
||||
Ref PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2);
|
||||
@@ -1501,6 +1518,20 @@ private:
|
||||
void StoreGPRRegister(uint32_t GPR, const Ref Src, IR::OpSize Size = OpSize::iInvalid, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, const Ref Src);
|
||||
|
||||
// Matches semantics around GPR storing that matches `StoreResult_WithOpSize` behaviour.
|
||||
void StoreGPRResultWithZExtSemantics(uint32_t GPR, Ref Src, IR::OpSize OpSize) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
if (GPRSize == OpSize::i64Bit && OpSize == OpSize::i32Bit) {
|
||||
// If the Source IR op is 64 bits, we need to zext the upper bits
|
||||
// For all other sizes, the upper bits are guaranteed to already be zero
|
||||
Src = GetOpSize(Src) == OpSize::i64Bit ? ARef(Src).Bfe(0, 32).Ref() : Src;
|
||||
StoreGPRRegister(GPR, Src, GPRSize);
|
||||
} else {
|
||||
StoreGPRRegister(GPR, Src, std::min(GPRSize, OpSize));
|
||||
}
|
||||
}
|
||||
|
||||
Ref _GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, bool Inline) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
const auto Offs = Op->PC + Op->InstSize + Offset - Entry;
|
||||
|
||||
@@ -1470,6 +1470,35 @@ void OpDispatchBuilder::AVX128_VPMADDWD(OpcodeArgs) {
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return PMADDWDOpImpl(OpSize::i128Bit, Src1, Src2); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper) {
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
const auto Is128Bit = Size == OpSize::i128Bit;
|
||||
|
||||
auto Acc = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit);
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
|
||||
|
||||
RefPair Result {};
|
||||
Result.Low = Helper(Acc.Low, Src1.Low, Src2.Low);
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
Result.High = Helper(Acc.High, Src1.High, Src2.High);
|
||||
}
|
||||
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPDPBUSD(OpcodeArgs, bool Saturating) {
|
||||
AVX128_VPDPImpl(Op,
|
||||
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPBUSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPDPWSSD(OpcodeArgs, bool Saturating) {
|
||||
AVX128_VPDPImpl(Op,
|
||||
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPWSSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Is128Bit = SrcSize == OpSize::i128Bit;
|
||||
|
||||
@@ -5,21 +5,30 @@
|
||||
namespace FEXCore::IR {
|
||||
constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
// Instructions
|
||||
{0x00, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0>},
|
||||
{0x00, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0, false>},
|
||||
{0x04, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ADD, FEXCore::IR::IROps::OP_ATOMICFETCHADD, 0, true>},
|
||||
|
||||
{0x08, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0>},
|
||||
{0x08, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0, false>},
|
||||
{0x0c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_OR, FEXCore::IR::IROps::OP_ATOMICFETCHOR, 0, true>},
|
||||
|
||||
{0x10, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0>},
|
||||
{0x10, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0, false>},
|
||||
{0x14, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 0, true>},
|
||||
|
||||
{0x18, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0>},
|
||||
{0x18, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0, false>},
|
||||
{0x1c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 0, true>},
|
||||
|
||||
{0x20, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0>},
|
||||
{0x20, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0, false>},
|
||||
{0x24, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_ANDWITHFLAGS, FEXCore::IR::IROps::OP_ATOMICFETCHAND, 0, true>},
|
||||
|
||||
{0x28, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0>},
|
||||
{0x28, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0, false>},
|
||||
{0x2c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_SUB, FEXCore::IR::IROps::OP_ATOMICFETCHSUB, 0, true>},
|
||||
|
||||
{0x30, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0>},
|
||||
{0x30, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0, false>},
|
||||
{0x34, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ALUOp, FEXCore::IR::IROps::OP_XOR, FEXCore::IR::IROps::OP_ATOMICFETCHXOR, 0, true>},
|
||||
|
||||
{0x38, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0, false>},
|
||||
{0x3c, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0, true>},
|
||||
|
||||
{0x38, 6, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 0>},
|
||||
{0x50, 8, &OpDispatchBuilder::PUSHREGOp},
|
||||
{0x58, 8, &OpDispatchBuilder::POPOp},
|
||||
{0x68, 1, &OpDispatchBuilder::PUSHOp},
|
||||
@@ -29,7 +38,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0x6C, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{0x70, 16, &OpDispatchBuilder::CondJUMPOp},
|
||||
{0x84, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
|
||||
{0x84, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0, false>},
|
||||
{0x86, 2, &OpDispatchBuilder::XCHGOp},
|
||||
{0x88, 4, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
|
||||
|
||||
@@ -37,7 +46,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0x8D, 1, &OpDispatchBuilder::LEAOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVSegOp, true>},
|
||||
{0x8F, 1, &OpDispatchBuilder::POPOp},
|
||||
{0x90, 8, &OpDispatchBuilder::XCHGOp},
|
||||
{0x90, 8, &OpDispatchBuilder::XCHGRAXOp},
|
||||
|
||||
{0x98, 1, &OpDispatchBuilder::CDQOp},
|
||||
{0x99, 1, &OpDispatchBuilder::CQOOp},
|
||||
@@ -49,7 +58,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0xA4, 2, &OpDispatchBuilder::MOVSOp},
|
||||
|
||||
{0xA6, 2, &OpDispatchBuilder::CMPSOp},
|
||||
{0xA8, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0>},
|
||||
{0xA8, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 0, true>},
|
||||
{0xAA, 2, &OpDispatchBuilder::STOSOp},
|
||||
{0xAC, 2, &OpDispatchBuilder::LODSOp},
|
||||
{0xAE, 2, &OpDispatchBuilder::SCASOp},
|
||||
|
||||
@@ -9,36 +9,36 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
// GROUP 1
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>}, // CMP
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x80), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1, false>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x81), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1, false>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 0), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 1), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADCOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SBBOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 4), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 5), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 6), 1, &OpDispatchBuilder::SecondaryALUOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_1, OpToIndex(0x83), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CMPOp, 1, false>},
|
||||
|
||||
// GROUP 2
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCROp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
|
||||
@@ -46,8 +46,8 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, true, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCROp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHRImmediateOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHLImmediateOp, false>}, // SAL
|
||||
@@ -73,8 +73,8 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, &OpDispatchBuilder::RCLSmallerOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, &OpDispatchBuilder::RCRSmallerOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLSmallerOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCRSmallerOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, &OpDispatchBuilder::SHLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
|
||||
@@ -82,16 +82,16 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, true, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RotateOp, false, false, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, &OpDispatchBuilder::RCLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, &OpDispatchBuilder::RCROp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCLOp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RCROp, true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, &OpDispatchBuilder::SHLOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, &OpDispatchBuilder::SHLOp}, // SAL
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ASHROp, false, false>}, // SAR
|
||||
|
||||
// GROUP 3
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 2), 1, &OpDispatchBuilder::NOTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 4), 1, &OpDispatchBuilder::MULOp},
|
||||
@@ -99,8 +99,8 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 6), 1, &OpDispatchBuilder::DIVOp}, // DIV
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF6), 7), 1, &OpDispatchBuilder::IDIVOp}, // IDIV
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::TESTOp, 1, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, &OpDispatchBuilder::NOTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, &OpDispatchBuilder::NEGOp}, // NEG
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
@@ -3249,9 +3250,10 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, MemBase, OpSize::i16Bit);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
Ref Top {};
|
||||
{
|
||||
auto NewFSW = _LoadMemGPR(OpSize::i16Bit, MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -3260,10 +3262,21 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
|
||||
_StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
|
||||
auto MMRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 32);
|
||||
_StoreContextFPR(OpSize::i128Bit, MMRegs.Low, MMBaseOffset() + i * 16);
|
||||
_StoreContextFPR(OpSize::i128Bit, MMRegs.High, MMBaseOffset() + (i + 1) * 16);
|
||||
auto SevenConst = Constant(7);
|
||||
auto low = Constant(~0ULL);
|
||||
auto high = Constant(0xFFFF);
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
Ref Reg = _LoadMemFPR(OpSize::i128Bit, MemBase, Constant(16 * i + 32), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(OpSize::i128Bit, Reg, Mask);
|
||||
if (ReducedPrecisionMode) {
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg);
|
||||
}
|
||||
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3760,6 +3773,84 @@ void OpDispatchBuilder::VPMADDUBSWOp(OpcodeArgs) {
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) {
|
||||
// Does four 8-bit unsigned * signed byte multiplies per 32-bit element, sums them and accumulates in to the destination
|
||||
|
||||
if (CTX->HostFeatures.SupportsI8MM) {
|
||||
// The I8MM extension maps onto VPDP* almost directly.
|
||||
if (!Saturating) {
|
||||
return _VUSDot(Size, Acc, Src1, Src2);
|
||||
}
|
||||
|
||||
auto DotProduct = _VUSDot(Size, LoadZeroVector(Size), Src1, Src2);
|
||||
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsDotProd) {
|
||||
// VSDOT assumes signed input, so we need to convert Src1 to signed, and then
|
||||
// perform correction afterwards using 0x40 to account for the unsigned input.
|
||||
auto Src1Signed = _VXor(Size, Src1, _VectorImm(Size, OpSize::i8Bit, 0x80));
|
||||
auto SixtyFour = _VectorImm(Size, OpSize::i8Bit, 0x40);
|
||||
|
||||
auto DotProduct = _VSDot(Size, Saturating ? LoadZeroVector(Size) : Acc, Src2, SixtyFour);
|
||||
DotProduct = _VSDot(Size, DotProduct, Src2, SixtyFour);
|
||||
DotProduct = _VSDot(Size, DotProduct, Src1Signed, Src2);
|
||||
if (Saturating) {
|
||||
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
return DotProduct;
|
||||
}
|
||||
|
||||
// Naive software implementation.
|
||||
auto Even = _VUnZip(Size, OpSize::i16Bit, Src1, Src2);
|
||||
auto Even1_16b = _VUXTL(Size, OpSize::i8Bit, Even);
|
||||
auto Even2_16b = _VSXTL2(Size, OpSize::i8Bit, Even);
|
||||
auto ResMul_Even = _VMul(Size, OpSize::i16Bit, Even1_16b, Even2_16b);
|
||||
|
||||
auto Odd = _VUnZip2(Size, OpSize::i16Bit, Src1, Src2);
|
||||
auto Odd1_16b = _VUXTL(Size, OpSize::i8Bit, Odd);
|
||||
auto Odd2_16b = _VSXTL2(Size, OpSize::i8Bit, Odd);
|
||||
auto ResMul_Odd = _VMul(Size, OpSize::i16Bit, Odd1_16b, Odd2_16b);
|
||||
|
||||
auto DotProduct = _VSAdALP(Size, OpSize::i16Bit, _VSAddLP(Size, OpSize::i16Bit, ResMul_Even), ResMul_Odd);
|
||||
if (Saturating) {
|
||||
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) {
|
||||
auto DotProduct = PMADDWDOpImpl(Size, Src1, Src2);
|
||||
if (!Saturating) {
|
||||
return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
|
||||
auto NegDotProduct = _VNeg(Size, OpSize::i32Bit, DotProduct);
|
||||
return _VSQSub(Size, OpSize::i32Bit, Acc, NegDotProduct);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPDPBUSDOp(OpcodeArgs, bool Saturating) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
|
||||
Ref Result = VPDPBUSDOpImpl(Size, Acc, Src1, Src2, Saturating);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPDPWSSDOp(OpcodeArgs, bool Saturating) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
|
||||
Ref Result = VPDPWSSDOpImpl(Size, Acc, Src1, Src2, Saturating);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
if (Signed) {
|
||||
@@ -5360,6 +5451,132 @@ void OpDispatchBuilder::VPERMILRegOp(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
// Computes the number of valid elements in a string for the explicit case,
|
||||
// where "valid" means that the character is not after the Nth character in the string.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXSaturateExplicitLength(OpSize ElementSize, OpSize LengthSize, Ref RawLength) {
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
SaveNZCV();
|
||||
_SubNZCV(LengthSize, RawLength, Constant(0));
|
||||
Ref AbsLength = _Neg(LengthSize, RawLength, CondClass::MI);
|
||||
return _Select(OpSize::i32Bit, LengthSize, CondClass::ULT, AbsLength, Constant(NumElements), AbsLength, Constant(NumElements));
|
||||
}
|
||||
|
||||
// Equal Any is a "is character in set" operation.
|
||||
// The set of characters is passed in Src1, and we then check if every
|
||||
// character in Src2 is in that set.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXEqualAny(OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements) {
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
// First we sanitize Src1, and zero out elements after the end of the string,
|
||||
// removing them from the matching set.
|
||||
Ref Src1Element0 = _VDupElement(OpSize::i128Bit, ElementSize, Src1, 0);
|
||||
Ref SanitizedSrc1 = _VBSL(OpSize::i128Bit, Src1ValidElements, Src1, Src1Element0);
|
||||
|
||||
// Now walk through every element in Src1, broadcast it into a temporary vector,
|
||||
// and compare it against every element in Src2, then OR the results
|
||||
// together to get the final match vector.
|
||||
Ref Matches = _VCMPEQ(OpSize::i128Bit, ElementSize, Src2, Src1Element0);
|
||||
for (uint32_t i = 1; i < NumElements; i++) {
|
||||
Ref Src1Element = _VDupElement(OpSize::i128Bit, ElementSize, SanitizedSrc1, i);
|
||||
Ref ElementMatches = _VCMPEQ(OpSize::i128Bit, ElementSize, Src2, Src1Element);
|
||||
Matches = _VOr(OpSize::i128Bit, Matches, ElementMatches);
|
||||
}
|
||||
|
||||
// Handle the case where Src1 was entirely empty, and also sanitize the results
|
||||
// for any NULLs in Src2, since those don't count towards matching.
|
||||
Ref Src1NotEmpty = _VDupElement(OpSize::i128Bit, ElementSize, Src1ValidElements, 0);
|
||||
Matches = _VAnd(OpSize::i128Bit, Matches, Src1NotEmpty);
|
||||
return _VAnd(OpSize::i128Bit, Matches, Src2ValidElements);
|
||||
}
|
||||
|
||||
|
||||
// The ranges aggregation is a weird one. It checks if every character in Src2 is
|
||||
// within a character range (ie. a-z or A-Z) similar to regex.
|
||||
// Ranges are passed in Src1 as pairs of characters in adjacent lanes.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXRanges(OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements, bool IsSigned) {
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
// Signed or unsigned greater than comparison, based on the Imm8 value
|
||||
const auto GreaterThan = [&](Ref Lhs, Ref Rhs) {
|
||||
return IsSigned ? _VCMPGT(OpSize::i128Bit, ElementSize, Lhs, Rhs) : _VUCMPGT(OpSize::i128Bit, ElementSize, Lhs, Rhs);
|
||||
};
|
||||
|
||||
// First we walk through the ranges from Src1, and construct
|
||||
// temporary vectors to represent the lower and upper bounds
|
||||
// of the comparison. Then check if both comparisons are true,
|
||||
// zero out invalid lanes, and OR the result into the final result vector.
|
||||
Ref Result {};
|
||||
for (uint32_t i = 0; i < NumElements; i += 2) {
|
||||
Ref LowerBound = _VDupElement(OpSize::i128Bit, ElementSize, Src1, i);
|
||||
Ref UpperBound = _VDupElement(OpSize::i128Bit, ElementSize, Src1, i + 1);
|
||||
Ref BelowLower = GreaterThan(LowerBound, Src2);
|
||||
Ref AboveUpper = GreaterThan(Src2, UpperBound);
|
||||
|
||||
// A range is only valid if its upper bound is a valid Src1 element.
|
||||
Ref RangeValid = _VDupElement(OpSize::i128Bit, ElementSize, Src1ValidElements, i + 1);
|
||||
|
||||
// RangeValid & ~BelowLower & ~AboveUpper
|
||||
Ref InRange = _VAndn(OpSize::i128Bit, RangeValid, BelowLower);
|
||||
InRange = _VAndn(OpSize::i128Bit, InRange, AboveUpper);
|
||||
Result = Result ? _VOr(OpSize::i128Bit, Result, InRange) : InRange;
|
||||
}
|
||||
|
||||
// Invalid (ie. NULL) Src2 characters don't count towards matching.
|
||||
return _VAnd(OpSize::i128Bit, Result, Src2ValidElements);
|
||||
}
|
||||
|
||||
// This aggregation is the simplest, and is the most similar
|
||||
// to strcmp(). It Just compares lanewise which characters are
|
||||
// equal in each string.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXEqualEach(OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1ValidElements, Ref Src2ValidElements) {
|
||||
// Elements match when both are valid and equal, or when both are invalid.
|
||||
Ref Equal = _VCMPEQ(OpSize::i128Bit, ElementSize, Src1, Src2);
|
||||
Ref BothValid = _VAnd(OpSize::i128Bit, Src1ValidElements, Src2ValidElements);
|
||||
Ref EitherValid = _VOr(OpSize::i128Bit, Src1ValidElements, Src2ValidElements);
|
||||
Ref ValidMatches = _VAnd(OpSize::i128Bit, Equal, BothValid);
|
||||
return _VOrn(OpSize::i128Bit, ValidMatches, EitherValid);
|
||||
}
|
||||
|
||||
// EqualOrdered implements a strstr()-like semantic finding all instances
|
||||
// of the string Src1 in Src2.
|
||||
Ref OpDispatchBuilder::PCMPXSTRXEqualOrdered(OpSize ElementSize, Ref Src1, Ref Src2, Ref Src1Length, Ref Src2Length, Ref Src1ValidElements,
|
||||
Ref Indices) {
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
// Broadcast needle[i] into every lane of a temp vector, then compare it to
|
||||
// the haystack vector. On succesive iterations, we shift the haystack over
|
||||
// by one element, and compare it against the next needle element.
|
||||
// AND together the results of all comparisons, and what is left should
|
||||
// be a vector where the only 'true' elements are lanes where the full
|
||||
// needle was found in the haystack, or those positions after the end of the string.
|
||||
Ref Result {};
|
||||
for (uint32_t i = 0; i < NumElements; i++) {
|
||||
Ref Needle = _VDupElement(OpSize::i128Bit, ElementSize, Src1, i);
|
||||
Ref Haystack = i == 0 ? Src2 : _VExtr(OpSize::i128Bit, ElementSize, Needle, Src2, i);
|
||||
Ref ElementsEqual = _VCMPEQ(OpSize::i128Bit, ElementSize, Haystack, Needle);
|
||||
|
||||
// TODO: I think a more optimal version of this is possible, perhaps using
|
||||
// MATCH from SVE2?
|
||||
Ref ElementsValid = _VDupElement(OpSize::i128Bit, ElementSize, Src1ValidElements, i);
|
||||
Ref ElementsEqualValid = _VOrn(OpSize::i128Bit, ElementsEqual, ElementsValid);
|
||||
|
||||
// Ternary to handle the first row case
|
||||
Result = Result ? _VAnd(OpSize::i128Bit, Result, ElementsEqualValid) : ElementsEqualValid;
|
||||
}
|
||||
|
||||
// Clear positions in the result after the end of the string.
|
||||
Ref FirstClearedPosition = Add(OpSize::i32Bit, _Sub(OpSize::i32Bit, Src2Length, Src1Length), 1);
|
||||
FirstClearedPosition =
|
||||
_Select(OpSize::i32Bit, OpSize::i32Bit, CondClass::ULT, Src2Length, Constant(NumElements), FirstClearedPosition, Constant(NumElements));
|
||||
FirstClearedPosition =
|
||||
_Select(OpSize::i32Bit, OpSize::i32Bit, CondClass::EQ, Src1Length, Constant(0), Constant(NumElements), FirstClearedPosition);
|
||||
|
||||
Ref FirstClearedPositionVector = _VDupFromGPR(OpSize::i128Bit, ElementSize, FirstClearedPosition);
|
||||
Ref KeptPositions = _VCMPGT(OpSize::i128Bit, ElementSize, FirstClearedPositionVector, Indices);
|
||||
return _VAnd(OpSize::i128Bit, Result, KeptPositions);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask, bool IsAVX) {
|
||||
const uint16_t Control = Op->Src[1].Literal();
|
||||
|
||||
@@ -5372,84 +5589,166 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Dest, OpSize::i128Bit, Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i128Bit, Op->Flags, {.Align = OpSize::i8Bit});
|
||||
|
||||
Ref IntermediateResult {};
|
||||
// Parse the immediate control bits.
|
||||
// See section 4.1 in Intel SDM for details.
|
||||
const auto Format = static_cast<CPU::SourceData>(Control & 0b11);
|
||||
const auto Aggregation = static_cast<CPU::AggregationOp>((Control >> 2) & 0b11);
|
||||
const auto Polarity = static_cast<CPU::Polarity>((Control >> 4) & 0b11);
|
||||
const bool OutputSelection = (Control & (1 << 6)) != 0;
|
||||
|
||||
// Helper constants
|
||||
const bool IsWords = Format == CPU::SourceData::U16 || Format == CPU::SourceData::S16;
|
||||
const bool IsSigned = Format == CPU::SourceData::S8 || Format == CPU::SourceData::S16;
|
||||
const auto ElementSize = IsWords ? OpSize::i16Bit : OpSize::i8Bit;
|
||||
const bool IsEqualOrdered = Aggregation == CPU::AggregationOp::EqualOrdered;
|
||||
const bool NeedsSrc1ValidElements = true;
|
||||
const bool NeedsSrc2ValidElements = !IsEqualOrdered || Polarity == CPU::Polarity::NegativeMasked;
|
||||
const uint32_t NumElements = IR::NumElements(OpSize::i128Bit, ElementSize);
|
||||
|
||||
// Load a vector where each element contains the value of its own index
|
||||
Ref Indices = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, IsWords ? NAMED_VECTOR_INCREMENTAL_U16_INDEX : NAMED_VECTOR_INCREMENTAL_U8_INDEX);
|
||||
|
||||
// Helper lambda that computes the lowest index element set "all high" in mask.
|
||||
Ref VecNumElements {};
|
||||
const auto LowestIndex = [&](Ref Mask) {
|
||||
if (!VecNumElements) {
|
||||
VecNumElements = _VectorImm(OpSize::i128Bit, ElementSize, NumElements);
|
||||
}
|
||||
Ref Candidates = _VBSL(OpSize::i128Bit, Mask, Indices, VecNumElements);
|
||||
return _VUMinV(OpSize::i128Bit, ElementSize, Candidates);
|
||||
};
|
||||
|
||||
Ref Src1Length {};
|
||||
Ref Src2Length {};
|
||||
Ref Src2LengthVector {};
|
||||
Ref Src1ValidElementsMask {};
|
||||
|
||||
// Compute the valid element masks for the two source strings.
|
||||
if (IsExplicit) {
|
||||
// Will be 4 in the absence of a REX.W bit and 8 in the presence of a REX.W bit.
|
||||
//
|
||||
// While the control bit immediate for the instruction itself is only ever 8 bits
|
||||
// in size, we use it as a 16-bit value so that we can use the 8th bit to signify
|
||||
// whether or not RAX and RDX should be interpreted as a 64-bit value.
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Is64Bit = SrcSize == OpSize::i64Bit;
|
||||
const auto NewControl = uint16_t(Control | (uint16_t(Is64Bit) << 8));
|
||||
const auto LengthSize = OpSizeFromSrc(Op);
|
||||
|
||||
Ref SrcRAX = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref SrcRDX = LoadGPRRegister(X86State::REG_RDX);
|
||||
Src1Length = PCMPXSTRXSaturateExplicitLength(ElementSize, LengthSize, LoadGPRRegister(X86State::REG_RAX));
|
||||
Src2Length = PCMPXSTRXSaturateExplicitLength(ElementSize, LengthSize, LoadGPRRegister(X86State::REG_RDX));
|
||||
|
||||
IntermediateResult = _VPCMPESTRX(Src1, Src2, SrcRAX, SrcRDX, NewControl);
|
||||
} else {
|
||||
IntermediateResult = _VPCMPISTRX(Src1, Src2, Control);
|
||||
}
|
||||
|
||||
if (IsMask) {
|
||||
// For the masked variant of the instructions, if control[6] is set, then we
|
||||
// need to expand the intermediate result into a byte or word mask (depending
|
||||
// on data size specified in control[1]) along the entire length of XMM0,
|
||||
// where set bits in the intermediate result set the corresponding entry
|
||||
// in XMM0 to all 1s and unset bits set the corresponding entry to all 0s.
|
||||
//
|
||||
// If control[6] is not set, then we just store the intermediate result as-is
|
||||
// into the least significant bits of XMM0 and zero extend it.
|
||||
const auto IsExpandedMask = (Control & 0b0100'0000) != 0;
|
||||
|
||||
if (IsExpandedMask) {
|
||||
// We need to iterate over the intermediate result and
|
||||
// expand the mask into XMM0 elements.
|
||||
const auto ElementSize = 1U << (Control & 1);
|
||||
const auto NumElements = 16U >> (Control & 1);
|
||||
|
||||
Ref Result = LoadZeroVector(OpSize::i128Bit);
|
||||
for (uint32_t i = 0; i < NumElements; i++) {
|
||||
Ref SignBit = _Sbfe(OpSize::i64Bit, 1, i, IntermediateResult);
|
||||
Result = _VInsGPR(OpSize::i128Bit, IR::SizeToOpSize(ElementSize), i, Result, SignBit);
|
||||
}
|
||||
|
||||
if (IsAVX) {
|
||||
StoreXMMRegister(0, Result);
|
||||
} else {
|
||||
StoreXMMRegister_WithAVXInsert(VectorOpType::SSE, 0, Result);
|
||||
}
|
||||
} else {
|
||||
// We insert the intermediate result as-is.
|
||||
Ref Result = _VCastFromGPR(OpSize::i128Bit, OpSize::i16Bit, IntermediateResult);
|
||||
|
||||
if (IsAVX) {
|
||||
StoreXMMRegister(0, Result);
|
||||
} else {
|
||||
StoreXMMRegister_WithAVXInsert(VectorOpType::SSE, 0, Result);
|
||||
}
|
||||
if (NeedsSrc1ValidElements) {
|
||||
Src1ValidElementsMask = _VCMPGT(OpSize::i128Bit, ElementSize, _VDupFromGPR(OpSize::i128Bit, ElementSize, Src1Length), Indices);
|
||||
}
|
||||
if (NeedsSrc2ValidElements) {
|
||||
Src2LengthVector = _VDupFromGPR(OpSize::i128Bit, ElementSize, Src2Length);
|
||||
}
|
||||
} else {
|
||||
Ref ZeroConst = Constant(0);
|
||||
// Get index of the first NUL element, to determine length of input strings.
|
||||
Ref Src1FirstNull = LowestIndex(_VCMPEQZ(OpSize::i128Bit, ElementSize, Src1));
|
||||
Ref Src2FirstNull = LowestIndex(_VCMPEQZ(OpSize::i128Bit, ElementSize, Src2));
|
||||
Src1Length = _VExtractToGPR(OpSize::i128Bit, ElementSize, Src1FirstNull, 0);
|
||||
Src2Length = _VExtractToGPR(OpSize::i128Bit, ElementSize, Src2FirstNull, 0);
|
||||
|
||||
// For the indexed variant of the instructions, if control[6] is set, then we
|
||||
// store the index of the most significant bit into ECX. If it's not set,
|
||||
// then we store the least significant bit.
|
||||
const auto UseMSBIndex = (Control & 0b0100'0000) != 0;
|
||||
if (NeedsSrc1ValidElements) {
|
||||
// If the index in the lane is less than the index of the first NUL, then it's a valid element.
|
||||
Src1ValidElementsMask = _VCMPGT(OpSize::i128Bit, ElementSize, _VDupElement(OpSize::i128Bit, ElementSize, Src1FirstNull, 0), Indices);
|
||||
}
|
||||
if (NeedsSrc2ValidElements) {
|
||||
Src2LengthVector = _VDupElement(OpSize::i128Bit, ElementSize, Src2FirstNull, 0);
|
||||
}
|
||||
}
|
||||
|
||||
Ref ResultNoFlags = _Bfe(OpSize::i32Bit, 16, 0, IntermediateResult);
|
||||
Ref Src1HasInvalidElements = Select01(OpSize::i32Bit, CondClass::ULT, Src1Length, Constant(NumElements));
|
||||
Ref Src2HasInvalidElements = Select01(OpSize::i32Bit, CondClass::ULT, Src2Length, Constant(NumElements));
|
||||
Ref Src2ValidElementsMask = NeedsSrc2ValidElements ? _VCMPGT(OpSize::i128Bit, ElementSize, Src2LengthVector, Indices) : nullptr;
|
||||
|
||||
Ref IfZero = Constant(16 >> (Control & 1));
|
||||
Ref IfNotZero = UseMSBIndex ? _FindMSB(IR::OpSize::i32Bit, ResultNoFlags) : _FindLSB(IR::OpSize::i32Bit, ResultNoFlags);
|
||||
Ref Result = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, ResultNoFlags, ZeroConst, IfZero, IfNotZero);
|
||||
// Perform the aggregation operations, see section 4.1.3 in the Intel SDM.
|
||||
Ref Matches {};
|
||||
switch (Aggregation) {
|
||||
case CPU::AggregationOp::EqualAny:
|
||||
Matches = PCMPXSTRXEqualAny(ElementSize, Src1, Src2, Src1ValidElementsMask, Src2ValidElementsMask);
|
||||
break;
|
||||
case CPU::AggregationOp::Ranges:
|
||||
Matches = PCMPXSTRXRanges(ElementSize, Src1, Src2, Src1ValidElementsMask, Src2ValidElementsMask, IsSigned);
|
||||
break;
|
||||
case CPU::AggregationOp::EqualEach:
|
||||
Matches = PCMPXSTRXEqualEach(ElementSize, Src1, Src2, Src1ValidElementsMask, Src2ValidElementsMask);
|
||||
break;
|
||||
case CPU::AggregationOp::EqualOrdered:
|
||||
Matches = PCMPXSTRXEqualOrdered(ElementSize, Src1, Src2, Src1Length, Src2Length, Src1ValidElementsMask, Indices);
|
||||
break;
|
||||
}
|
||||
|
||||
// Invert the matches if the polarity is negative or negative masked,
|
||||
// see section 4.1.4 in the Intel SDM.
|
||||
if (Polarity == CPU::Polarity::Negative) {
|
||||
Matches = _VNot(OpSize::i128Bit, Matches);
|
||||
} else if (Polarity == CPU::Polarity::NegativeMasked) {
|
||||
Matches = _VXor(OpSize::i128Bit, Matches, Src2ValidElementsMask);
|
||||
}
|
||||
|
||||
// Compute the flags and result
|
||||
Ref CF {};
|
||||
Ref OF {};
|
||||
|
||||
// PCMPESTRM and PCMPISTRM
|
||||
// These instructions return a mask of the matching elements.
|
||||
if (IsMask) {
|
||||
Ref Result = Matches;
|
||||
// Byte/word mask rather than a bit mask.
|
||||
if (OutputSelection) {
|
||||
// Result is already a byte/word mask, so just compute flags.
|
||||
// CF set when no elements matched
|
||||
// OF is set when the first element matched.
|
||||
Ref AnyMatch = _VExtractToGPR(OpSize::i128Bit, ElementSize, _VUMaxV(OpSize::i128Bit, ElementSize, Matches), 0);
|
||||
CF = Select01(OpSize::i32Bit, CondClass::EQ, AnyMatch, Constant(0));
|
||||
OF = _And(OpSize::i64Bit, _VExtractToGPR(OpSize::i128Bit, ElementSize, Matches, 0), Constant(1));
|
||||
|
||||
} else {
|
||||
// Convert per element vector to a bit mask with one bit per lane.
|
||||
Ref BitPositions = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_MOVMASKB);
|
||||
Ref Zero = LoadZeroVector(OpSize::i128Bit);
|
||||
|
||||
// If the elements are words, truncate to bytes first.
|
||||
Result = IsWords ? _VUnZip(OpSize::i128Bit, OpSize::i8Bit, Matches, Zero) : Matches;
|
||||
// Now repeat this pattern until we collapse each byte into a single bit.
|
||||
Result = _VAnd(OpSize::i128Bit, Result, BitPositions);
|
||||
Result = _VAddP(OpSize::i128Bit, OpSize::i8Bit, Result, Zero);
|
||||
Result = _VAddP(OpSize::i128Bit, OpSize::i8Bit, Result, Zero);
|
||||
Result = _VAddP(OpSize::i128Bit, OpSize::i8Bit, Result, Zero);
|
||||
|
||||
// Compute the flags from the bit mask in Result
|
||||
Ref BitMask = _VExtractToGPR(OpSize::i128Bit, IsWords ? OpSize::i8Bit : OpSize::i16Bit, Result, 0);
|
||||
CF = Select01(OpSize::i32Bit, CondClass::EQ, BitMask, Constant(0));
|
||||
OF = _And(OpSize::i64Bit, BitMask, Constant(1));
|
||||
}
|
||||
|
||||
StoreXMMRegister_WithAVXInsert(IsAVX ? VectorOpType::AVX : VectorOpType::SSE, 0, Result);
|
||||
|
||||
// PCMPESTRI and PCMPISTRI
|
||||
// These instructions return either the index of the first match
|
||||
// or the number of elements in the string if no match.
|
||||
} else {
|
||||
Ref LowestMatch = _VExtractToGPR(OpSize::i128Bit, ElementSize, LowestIndex(Matches), 0);
|
||||
CF = Select01(OpSize::i32Bit, CondClass::EQ, LowestMatch, Constant(NumElements));
|
||||
OF = Select01(OpSize::i32Bit, CondClass::EQ, LowestMatch, Constant(0));
|
||||
|
||||
Ref Result = LowestMatch;
|
||||
// Most significant rather than least significant index.
|
||||
if (OutputSelection) {
|
||||
// The max is also 0 when nothing matched, LowestMatch tells those apart.
|
||||
Ref Candidates = _VAnd(OpSize::i128Bit, Indices, Matches);
|
||||
Ref HighestMatchVector = _VUMaxV(OpSize::i128Bit, ElementSize, Candidates);
|
||||
Ref HighestMatch = _VExtractToGPR(OpSize::i128Bit, ElementSize, HighestMatchVector, 0);
|
||||
Result = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClass::EQ, LowestMatch, Constant(NumElements), Constant(NumElements), HighestMatch);
|
||||
}
|
||||
|
||||
// Store the result, it is already zero-extended to 64-bit implicitly.
|
||||
StoreGPRRegister(X86State::REG_RCX, Result);
|
||||
}
|
||||
|
||||
// Set all of the necessary flags. NZCV stored in bits 28...31 like the hw op.
|
||||
SetNZCV(IntermediateResult);
|
||||
CFInverted = false;
|
||||
// SF/ZF are set when Src1/Src2 respectively have invalid elements.
|
||||
Ref SF = Src1HasInvalidElements;
|
||||
Ref ZF = Src2HasInvalidElements;
|
||||
Ref NZCV = _Orlshl(OpSize::i64Bit, _Lshl(OpSize::i64Bit, OF, Constant(28)), CF, 29);
|
||||
NZCV = _Orlshl(OpSize::i64Bit, NZCV, ZF, 30);
|
||||
NZCV = _Orlshl(OpSize::i64Bit, NZCV, SF, 31);
|
||||
SetNZCV(NZCV);
|
||||
CFInverted = true;
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
|
||||
@@ -768,10 +768,14 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
// Store Status Word
|
||||
// There's no load Status Word instruction but you can load it through frstor
|
||||
// or fldenv.
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs, bool DestRAX) {
|
||||
Ref TopValue = _SyncStackToSlow();
|
||||
Ref StatusWord = ReconstructFSW_Helper(TopValue);
|
||||
StoreResultGPR(Op, StatusWord);
|
||||
if (DestRAX) {
|
||||
StoreGPRRegister(X86State::REG_RAX, StatusWord, OpSize::i16Bit);
|
||||
} else {
|
||||
StoreResultGPR(Op, StatusWord);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
|
||||
@@ -856,22 +860,103 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
auto a = _ReadStackValue(0);
|
||||
Ref Result =
|
||||
ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
|
||||
Ref Value = ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Result) : _Bfe(OpSize::i64Bit, 1, 15, Result);
|
||||
// Extract the sign bit, which goes in C1
|
||||
Ref Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Value) : _Bfe(OpSize::i64Bit, 1, 15, Value);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
|
||||
|
||||
// Claim this is a normal number
|
||||
// We don't support anything else
|
||||
auto TopValid = _StackValidTag(0);
|
||||
auto NotEmpty = _StackValidTag(0);
|
||||
Ref IsEmpty = _Xor(OpSize::i64Bit, NotEmpty, Constant(1));
|
||||
Ref IsNaN {};
|
||||
Ref IsDenormal {};
|
||||
Ref IsInf {};
|
||||
Ref IsZero {};
|
||||
Ref IsUnsupported {};
|
||||
Ref NoSignBit {};
|
||||
|
||||
// In the case of top being invalid then C3:C2:C0 is 0b101
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
|
||||
// TODO: The codegen for this is not optimal, and can probably be improved
|
||||
// if FXAM ends up on the hot path for some workload.
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
constexpr uint64_t ExponentMask = 0x7FF0'0000'0000'0000ULL;
|
||||
NoSignBit = _Bfe(OpSize::i64Bit, 63, 0, Value);
|
||||
IsInf = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(ExponentMask));
|
||||
IsNaN = Select01(OpSize::i64Bit, CondClass::UGT, NoSignBit, Constant(ExponentMask));
|
||||
|
||||
IsZero = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(0));
|
||||
// 64 bit floats can't represent an x87 denormal, nor any of the
|
||||
// unsupported encodings.
|
||||
IsDenormal = Constant(0);
|
||||
IsUnsupported = Constant(0);
|
||||
} else {
|
||||
Ref Mantissa = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 0);
|
||||
|
||||
// "J" is the name given to the msb of the mantissa in the SDM.
|
||||
Ref JBit = _Bfe(OpSize::i64Bit, 1, 63, Mantissa);
|
||||
Ref Exponent = _Bfe(OpSize::i64Bit, 15, 0, Value);
|
||||
Ref IsExponentZero = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0));
|
||||
Ref IsExponentMax = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0x7FFF));
|
||||
|
||||
// Inf is when mantissa only has the J bit set, exponent is all 1's.
|
||||
Ref IsOnlyJBit = Select01(OpSize::i64Bit, CondClass::EQ, Mantissa, Constant(1ULL << 63));
|
||||
IsInf = _And(OpSize::i64Bit, IsExponentMax, IsOnlyJBit);
|
||||
|
||||
// NaN is when the low 63 bits of the mantissa are non-zero
|
||||
// and exponent is max, and the J bit is set.
|
||||
Ref Fraction = _Bfe(OpSize::i64Bit, 63, 0, Mantissa);
|
||||
Ref FractionNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Fraction, Constant(0));
|
||||
Ref IsExponentMaxWithJBit = _And(OpSize::i64Bit, IsExponentMax, JBit);
|
||||
IsNaN = _And(OpSize::i64Bit, IsExponentMaxWithJBit, FractionNonZero);
|
||||
|
||||
// Zero and Denormal are basically the same as the 64-bit case.
|
||||
Ref MantissaNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Mantissa, Constant(0));
|
||||
Ref MantissaZero = _Xor(OpSize::i64Bit, MantissaNonZero, Constant(1));
|
||||
IsZero = _And(OpSize::i64Bit, IsExponentZero, MantissaZero);
|
||||
IsDenormal = _And(OpSize::i64Bit, IsExponentZero, MantissaNonZero);
|
||||
|
||||
// This is where things are weird. If the J bit is not set
|
||||
// and the exponent is non-zero, then this is an "unsupported"
|
||||
// encoding, which I believe is left in for legacy reasons.
|
||||
Ref IsSupported = _Or(OpSize::i64Bit, IsExponentZero, JBit);
|
||||
IsUnsupported = _Xor(OpSize::i64Bit, IsSupported, Constant(1));
|
||||
}
|
||||
|
||||
// NormalFiniteNumber = !Zero && !Denormal && !Inf && !NaN && !Empty && !Unsupported
|
||||
Ref temp1 = _Or(OpSize::i64Bit, IsZero, IsDenormal);
|
||||
Ref temp2 = _Or(OpSize::i64Bit, IsInf, IsNaN);
|
||||
Ref temp3 = _Or(OpSize::i64Bit, IsUnsupported, IsEmpty);
|
||||
temp1 = _Or(OpSize::i64Bit, temp1, temp2);
|
||||
temp2 = _Or(OpSize::i64Bit, temp1, temp3);
|
||||
Ref NormalFiniteNumber = _Xor(OpSize::i64Bit, temp2, Constant(1));
|
||||
|
||||
// Set C3, C2, C0 based on the class of the FP value
|
||||
// Table is from "FXAM" page in the SDM.
|
||||
// +----------------------+----+----+----+
|
||||
// | Class | C3 | C2 | C0 |
|
||||
// +----------------------+----+----+----+
|
||||
// | Unsupported | 0 | 0 | 0 |
|
||||
// | NaN | 0 | 0 | 1 |
|
||||
// | Normal finite number | 0 | 1 | 0 |
|
||||
// | Infinity | 0 | 1 | 1 |
|
||||
// | Zero | 1 | 0 | 0 |
|
||||
// | Empty | 1 | 0 | 1 |
|
||||
// | Denormal number | 1 | 1 | 0 |
|
||||
// +----------------------+----+----+----+
|
||||
|
||||
// C0 = IsNaN || IsInf || IsEmpty
|
||||
Ref C0 = _Or(OpSize::i64Bit, IsNaN, IsInf);
|
||||
C0 = _Or(OpSize::i64Bit, C0, IsEmpty);
|
||||
|
||||
// C2 = (IsInf || Denormal || NormalFiniteNumber) && !IsEmpty
|
||||
Ref C2 = _Or(OpSize::i64Bit, IsInf, IsDenormal);
|
||||
C2 = _Or(OpSize::i64Bit, C2, NormalFiniteNumber);
|
||||
C2 = _And(OpSize::i64Bit, C2, NotEmpty);
|
||||
|
||||
// C3 = Zero || IsEmpty || Denormal
|
||||
Ref C3 = _Or(OpSize::i64Bit, IsZero, IsEmpty);
|
||||
C3 = _Or(OpSize::i64Bit, C3, IsDenormal);
|
||||
|
||||
auto C2 = TopValid;
|
||||
auto C0 = C3; // Mirror C3 until something other than zero is supported
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(C0);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(C2);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(C3);
|
||||
|
||||
@@ -16,7 +16,7 @@ static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
CodeBuffer::CodeBuffer(size_t Size)
|
||||
CodeBuffer::CodeBuffer(size_t Size, bool ShouldBeNamed)
|
||||
: AllocatedSize(Size) {
|
||||
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
|
||||
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
|
||||
@@ -28,7 +28,9 @@ CodeBuffer::CodeBuffer(size_t Size)
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMemJIT", Ptr, Size);
|
||||
if (ShouldBeNamed) {
|
||||
FEXCore::Allocator::VirtualName("FEXMemJIT", Ptr, Size);
|
||||
}
|
||||
|
||||
// Huge-pages reduce the amount of iTLB misses dramatically when it works.
|
||||
FEXCore::Allocator::VirtualTHPControl(Ptr, Size, FEXCore::Allocator::THPControl::Enable);
|
||||
@@ -43,6 +45,17 @@ CodeBuffer::~CodeBuffer() {
|
||||
FEXCore::Allocator::VirtualFree(Ptr, AllocatedSize);
|
||||
}
|
||||
|
||||
SharedCodeBufferManager::SharedCodeBufferManager() {
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
|
||||
// Only name the JIT buffers if perf JIT naming is disabled.
|
||||
// `perf top` prefers VMA names over the JIT symbols file for some reason.
|
||||
// Breaks memory tracking when naming is enabled, but it's a debug feature so it isn't expected to be enabled by default.
|
||||
NameJITBuffers = !(GlobalJITNaming || LibraryJITNaming || BlockJITNaming);
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::AllocateNew(size_t Size) {
|
||||
#ifndef _WIN32
|
||||
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
|
||||
@@ -66,7 +79,7 @@ fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::AllocateNew(size_t Size)
|
||||
}
|
||||
#endif
|
||||
|
||||
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
|
||||
auto Buffer = fextl::make_shared<CodeBuffer>(Size, NameJITBuffers);
|
||||
|
||||
Latest = Buffer;
|
||||
|
||||
|
||||
@@ -23,7 +23,7 @@ namespace FEXCore::CPU {
|
||||
struct CodeBuffer {
|
||||
fextl::unique_ptr<GuestToHostMap> LookupCache;
|
||||
|
||||
CodeBuffer(size_t Size);
|
||||
CodeBuffer(size_t Size, bool ShouldBeNamed);
|
||||
CodeBuffer(const CodeBuffer&) = delete;
|
||||
CodeBuffer& operator=(const CodeBuffer&) = delete;
|
||||
CodeBuffer(CodeBuffer&& oth) = delete;
|
||||
@@ -111,6 +111,7 @@ private:
|
||||
*/
|
||||
class SharedCodeBufferManager {
|
||||
public:
|
||||
SharedCodeBufferManager();
|
||||
virtual ~SharedCodeBufferManager() = default;
|
||||
|
||||
// Get the CodeBuffer that was most recently allocated.
|
||||
@@ -131,5 +132,7 @@ private:
|
||||
fextl::shared_ptr<CodeBuffer> Latest;
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
|
||||
|
||||
bool NameJITBuffers {true};
|
||||
};
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -83,22 +83,22 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
|
||||
},
|
||||
// ENTRY_27
|
||||
{
|
||||
{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::DAAOp } },
|
||||
{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::DAAOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_2F
|
||||
{
|
||||
{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::DASOp } },
|
||||
{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::DASOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_37
|
||||
{
|
||||
{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::AAAOp } },
|
||||
{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::AAAOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_3F
|
||||
{
|
||||
{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::AASOp } },
|
||||
{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::AASOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_40
|
||||
@@ -134,23 +134,23 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
|
||||
},
|
||||
// ENTRY_A0
|
||||
{
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_A1
|
||||
{
|
||||
{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_A2
|
||||
{
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_A3
|
||||
{
|
||||
{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_MEM_OFFSET | FLAGS_LITERAL_PATCHABLE, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_CE
|
||||
{
|
||||
@@ -159,17 +159,17 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
|
||||
},
|
||||
// ENTRY_D4
|
||||
{
|
||||
{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, { .OpDispatch = &IR::OpDispatchBuilder::AAMOp } },
|
||||
{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1, { .OpDispatch = &IR::OpDispatchBuilder::AAMOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_D5
|
||||
{
|
||||
{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, { .OpDispatch = &IR::OpDispatchBuilder::AADOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1, { .OpDispatch = &IR::OpDispatchBuilder::AADOp } },
|
||||
{"REX2", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_D6
|
||||
{
|
||||
{"SALC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::SALCOp } },
|
||||
{"SALC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 0, { .OpDispatch = &IR::OpDispatchBuilder::SALCOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_EA
|
||||
@@ -204,8 +204,8 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x01, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x02, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x03, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x04, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x05, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x04, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x05, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
{0x06, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_06] }}},
|
||||
{0x07, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_07] }}},
|
||||
@@ -214,16 +214,16 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x09, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x0A, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x0B, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x0C, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x0D, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x0C, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x0D, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x0E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_0E] }}},
|
||||
|
||||
{0x10, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x11, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x12, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x13, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x14, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x15, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x14, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x15, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x16, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_16] }}},
|
||||
{0x17, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_17] }}},
|
||||
|
||||
@@ -231,8 +231,8 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x19, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x1A, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x1B, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x1C, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x1D, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x1C, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x1D, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x1E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1E] }}},
|
||||
{0x1F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1F] }}},
|
||||
|
||||
@@ -240,32 +240,32 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x21, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x22, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x23, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x24, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x25, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x24, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x25, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
{0x27, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_27] }}},
|
||||
{0x28, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x29, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x2A, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x2B, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x2C, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x2D, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x2C, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x2D, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x2F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_2F] }}},
|
||||
|
||||
{0x30, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x31, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x32, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x33, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x34, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x35, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x34, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x35, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
{0x37, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_37] }}},
|
||||
{0x38, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x39, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x3A, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x3B, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x3C, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x3D, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x3C, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0x3D, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0x3F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_3F] }}},
|
||||
|
||||
{0x40, 8, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_40] }}},
|
||||
@@ -321,9 +321,9 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x8D, 1, X86InstInfo{"LEA", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{0x8E, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_16BIT) | FLAGS_MODRM, 0}},
|
||||
{0x8F, 1, X86InstInfo{"POP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_ZERO_REG | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0x90, 8, X86InstInfo{"XCHG", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0x98, 1, X86InstInfo{"CDQE", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0x99, 1, X86InstInfo{"CQO", TYPE_INST, FLAGS_SF_DST_RDX | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0x90, 8, X86InstInfo{"XCHG", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0}},
|
||||
{0x98, 1, X86InstInfo{"CDQE", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0x99, 1, X86InstInfo{"CQO", TYPE_INST, FLAGS_NONE, 0}},
|
||||
|
||||
// These three are all X87 instructions
|
||||
{0x9B, 1, X86InstInfo{"FWAIT", TYPE_INST, FLAGS_NONE, 0}},
|
||||
@@ -343,17 +343,17 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT), 1}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
|
||||
{0xB0, 8, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_REX_IN_BYTE , 1}},
|
||||
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2, 4}},
|
||||
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
{0xC2, 1, X86InstInfo{"RET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2}},
|
||||
{0xC3, 1, X86InstInfo{"RET", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END , 0}},
|
||||
@@ -362,7 +362,7 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0xCA, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2}},
|
||||
{0xCB, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 1}},
|
||||
{0xCE, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_CE] }}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
|
||||
|
||||
@@ -371,9 +371,9 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0xD6, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_D6] }}},
|
||||
{0xD7, 1, X86InstInfo{"XLAT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
|
||||
{0xE0, 1, X86InstInfo{"LOOPNE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
|
||||
{0xE1, 1, X86InstInfo{"LOOPE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
|
||||
{0xE2, 1, X86InstInfo{"LOOP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
|
||||
{0xE0, 1, X86InstInfo{"LOOPNE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT, 1}},
|
||||
{0xE1, 1, X86InstInfo{"LOOPE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT, 1}},
|
||||
{0xE2, 1, X86InstInfo{"LOOP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT, 1}},
|
||||
{0xE3, 1, X86InstInfo{"JrCXZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
|
||||
// Should just throw GP
|
||||
|
||||
@@ -36,11 +36,11 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> PrimaryGroup_ArchSelect_LUT = {{
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::ADCOp, 1> }},
|
||||
{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::ADCOp, 1, false> }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SBBOp, 1> }},
|
||||
{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SBBOp, 1, false> }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
@@ -56,7 +56,7 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> PrimaryGroup_ArchSelect_LUT = {{
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::CMPOp, 1> }},
|
||||
{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::CMPOp, 1, false> }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
}};
|
||||
@@ -75,14 +75,14 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 6), 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 7), 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
|
||||
// Duplicates the 0x80 opcode group
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 0), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = PrimaryGroup_ArchSelect_LUT[ENTRY_1_82_0] }}},
|
||||
@@ -140,23 +140,23 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD1), 6), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD1), 7), 1, X86InstInfo{"SAR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, X86InstInfo{"ROL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, X86InstInfo{"ROR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, X86InstInfo{"RCL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, X86InstInfo{"RCR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, X86InstInfo{"SHR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, X86InstInfo{"SAR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, X86InstInfo{"ROL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, X86InstInfo{"ROR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 2), 1, X86InstInfo{"RCL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 3), 1, X86InstInfo{"RCR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, X86InstInfo{"SHR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 6), 1, X86InstInfo{"SHL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, X86InstInfo{"SAR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, X86InstInfo{"ROL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, X86InstInfo{"ROR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, X86InstInfo{"RCL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, X86InstInfo{"RCR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, X86InstInfo{"SHR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, X86InstInfo{"SAR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, X86InstInfo{"ROL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, X86InstInfo{"ROR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 2), 1, X86InstInfo{"RCL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 3), 1, X86InstInfo{"RCR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, X86InstInfo{"SHR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 6), 1, X86InstInfo{"SHL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, X86InstInfo{"SAR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
// GROUP 3
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
@@ -168,7 +168,7 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 1, X86InstInfo{"DIV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 7), 1, X86InstInfo{"IDIV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, X86InstInfo{"NOT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, X86InstInfo{"NEG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
@@ -196,7 +196,7 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 0), 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT, 1}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC6), 7), 1, X86InstInfo{"XABORT", TYPE_INST, FLAGS_MODRM, 1}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_LITERAL_PATCHABLE, 4}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 1), 5, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_11, OpToIndex(0xC7), 7), 1, X86InstInfo{"XBEGIN", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT | FLAGS_SETS_RIP | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
};
|
||||
|
||||
@@ -50,7 +50,7 @@ constexpr std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableO
|
||||
{((3 << 3) | 1), 1, X86InstInfo{"RDTSCP", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 2), 1, X86InstInfo{"MONITORX", TYPE_PRIV, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 3), 1, X86InstInfo{"MWAITX", TYPE_PRIV, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_SRC_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{((3 << 3) | 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{((3 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
|
||||
@@ -26,8 +26,8 @@ enum Secondary_LUT {
|
||||
|
||||
constexpr std::array<X86InstInfo[2], ENTRY_MAX> Secondary_ArchSelect_LUT = {{
|
||||
{
|
||||
{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, { .OpDispatch = &IR::OpDispatchBuilder::NOPOp } },
|
||||
{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SyscallOp, true> } },
|
||||
{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 0, { .OpDispatch = &IR::OpDispatchBuilder::NOPOp } },
|
||||
{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY | FLAGS_BLOCK_END, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SyscallOp, true> } },
|
||||
},
|
||||
{
|
||||
{"PUSH FS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX> } },
|
||||
@@ -204,17 +204,17 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
|
||||
{0xA0, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A0] }}},
|
||||
{0xA1, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A1] }}},
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xA4, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1}},
|
||||
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA6, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA8, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A8] }}},
|
||||
{0xA9, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A9] }}},
|
||||
{0xAA, 1, X86InstInfo{"RSM", TYPE_PRIV, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAB, 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xAC, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1}},
|
||||
{0xAD, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAD, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAE, 1, X86InstInfo{"", TYPE_GROUP_15, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAF, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
|
||||
|
||||
@@ -312,6 +312,11 @@ namespace AVX128 {
|
||||
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VSSHR>}, // VPSRAVD
|
||||
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VUSHL>}, // VPSLLV
|
||||
|
||||
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, false>},
|
||||
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, true>},
|
||||
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, false>},
|
||||
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i128Bit>},
|
||||
@@ -757,6 +762,11 @@ namespace AVX256 {
|
||||
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::VPSRAVDOp},
|
||||
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::VPSLLVOp},
|
||||
|
||||
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, false>},
|
||||
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, true>},
|
||||
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, false>},
|
||||
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i128Bit>},
|
||||
@@ -1207,6 +1217,11 @@ auto BaseTableLambda = [](const auto RuntimeTable) consteval {
|
||||
{OPD(2, 0b01, 0x46), 1, X86InstInfo{"VPSRAVD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x47), 1, X86InstInfo{"VPSLLV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(2, 0b01, 0x50), 1, X86InstInfo{"VPDPBUSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x51), 1, X86InstInfo{"VPDPBUSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x52), 1, X86InstInfo{"VPDPWSSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x53), 1, X86InstInfo{"VPDPWSSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, X86InstInfo{"VPBROADCASTD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x59), 1, X86InstInfo{"VPBROADCASTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x5A), 1, X86InstInfo{"VBROADCASTI128", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_L_1 | FLAGS_SF_MOD_MEM_ONLY | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
@@ -185,10 +185,14 @@ struct DecodedOperand {
|
||||
struct {
|
||||
int64_t Displacement;
|
||||
uint8_t GPR;
|
||||
bool PatchableDisp;
|
||||
uint8_t DispOffset;
|
||||
} GPRIndirect; // Shared with GPRIndirectRelocation
|
||||
|
||||
struct {
|
||||
int64_t Value;
|
||||
bool PatchableDisp;
|
||||
uint8_t DispOffset;
|
||||
} RIPLiteral; // Shared with RIPLiteralRelocation
|
||||
|
||||
struct LiteralType {
|
||||
@@ -211,7 +215,9 @@ struct DecodedOperand {
|
||||
uint8_t Scale;
|
||||
uint8_t Index; // ~0 invalid
|
||||
uint8_t Base; // ~0 invalid
|
||||
} SIB; // Shared with SIBRelocation
|
||||
bool PatchableDisp;
|
||||
uint8_t DispOffset;
|
||||
} SIB; // Shared with SIBRelocation
|
||||
};
|
||||
|
||||
TypeUnion Data;
|
||||
@@ -349,11 +355,8 @@ namespace InstFlags {
|
||||
constexpr InstFlagType FLAGS_X87_FLAGS = (1ULL << 10);
|
||||
|
||||
// Non-XMM subflags
|
||||
constexpr InstFlagType FLAGS_SF_DST_RAX = (1ULL << 11);
|
||||
constexpr InstFlagType FLAGS_SF_DST_RDX = (1ULL << 12);
|
||||
constexpr InstFlagType FLAGS_SF_SRC_RAX = (1ULL << 13);
|
||||
constexpr InstFlagType FLAGS_SF_SRC_RCX = (1ULL << 14);
|
||||
constexpr InstFlagType FLAGS_SF_REX_IN_BYTE = (1ULL << 15);
|
||||
constexpr InstFlagType FLAGS_SF_REX_IN_BYTE = (1ULL << 11);
|
||||
// subflag [15:12] unused
|
||||
|
||||
// XMM subflags
|
||||
constexpr InstFlagType FLAGS_SF_UNUSED = (1ULL << 11); // No assigned behavior yet
|
||||
@@ -403,7 +406,8 @@ namespace InstFlags {
|
||||
|
||||
constexpr InstFlagType FLAGS_CALL = (1ULL << 30);
|
||||
constexpr InstFlagType FLAGS_SUPPORTS_LOCK = (1ULL << 31);
|
||||
// Flags [57..32]: Undefined
|
||||
constexpr InstFlagType FLAGS_LITERAL_PATCHABLE = (1ULL << 32);
|
||||
// Flags [57..33]: Undefined
|
||||
// Flags [60..58]: Dst size
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
// Flags [63..61]: Src size
|
||||
@@ -419,13 +423,6 @@ namespace InstFlags {
|
||||
constexpr InstFlagType SIZE_256BIT = 0b110;
|
||||
constexpr InstFlagType SIZE_64BITDEF = 0b111; // Default mode is 64bit instead of typical 32bit
|
||||
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY;
|
||||
#else
|
||||
// Syscall ends a block on WIN32 because the instruction can update the CPU's RIP.
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY | FLAGS_BLOCK_END;
|
||||
#endif
|
||||
|
||||
constexpr InstFlagType GetSizeDstFlags(InstFlagType Flags) {
|
||||
return (Flags >> FLAGS_SIZE_DST_OFF) & SIZE_MASK;
|
||||
}
|
||||
|
||||
@@ -216,7 +216,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F64OpTable = {{
|
||||
// 5 = Invalid
|
||||
{OPDReg(0xDD, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSAVE},
|
||||
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, false>},
|
||||
|
||||
{OPD(0xDD, 0xC0), 8, &OpDispatchBuilder::X87FFREE},
|
||||
{OPD(0xDD, 0xC8), 8, &OpDispatchBuilder::FXCH},
|
||||
@@ -284,7 +284,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F64OpTable = {{
|
||||
{OPD(0xDF, 0xD0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
|
||||
{OPD(0xDF, 0xD8), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
|
||||
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, true>},
|
||||
{OPD(0xDF, 0xE8), 8,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::FCOMIF64, OpSize::f80Bit, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
{OPD(0xDF, 0xF0), 8,
|
||||
@@ -483,7 +483,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F80OpTable = {{
|
||||
// 5 = Invalid
|
||||
{OPDReg(0xDD, 6) | 0x00, 8, &OpDispatchBuilder::X87FNSAVE},
|
||||
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPDReg(0xDD, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, false>},
|
||||
|
||||
{OPD(0xDD, 0xC0), 8, &OpDispatchBuilder::X87FFREE},
|
||||
{OPD(0xDD, 0xC8), 8, &OpDispatchBuilder::FXCH},
|
||||
@@ -545,7 +545,7 @@ constexpr std::array<DispatchTableEntry, 140> X87F80OpTable = {{
|
||||
{OPD(0xDF, 0xD0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
|
||||
{OPD(0xDF, 0xD8), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTToStack>},
|
||||
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::X87FNSTSW},
|
||||
{OPD(0xDF, 0xE0), 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::X87FNSTSW, true>},
|
||||
{OPD(0xDF, 0xE8), 8,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::FCOMI, OpSize::f80Bit, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>},
|
||||
{OPD(0xDF, 0xF0), 8,
|
||||
@@ -794,7 +794,7 @@ auto GenerateX87TableLambda = [](const auto DispatchTable) consteval {
|
||||
// / 3
|
||||
{OPD(0xDF, 0xD8), 8, X86InstInfo{"FSTP", TYPE_X87, FLAGS_SF_MOD_DST | FLAGS_POP, 0}},
|
||||
// / 4
|
||||
{OPD(0xDF, 0xE0), 1, X86InstInfo{"FNSTSW", TYPE_INST, GenFlagsSameSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0}},
|
||||
{OPD(0xDF, 0xE0), 1, X86InstInfo{"FNSTSW", TYPE_INST, GenFlagsSameSize(SIZE_16BIT), 0}},
|
||||
{OPD(0xDF, 0xE1), 7, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
// / 5
|
||||
{OPD(0xDF, 0xE8), 8, X86InstInfo{"FUCOMIP", TYPE_INST, FLAGS_POP, 0}},
|
||||
|
||||
@@ -201,7 +201,7 @@
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"ThreadRemoveCodeEntry GPR:$Entry": {
|
||||
"ThreadRemoveCodeEntry GPR:$EntryToInvalidate, GPR:$NewRIP": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
@@ -275,7 +275,8 @@
|
||||
"The boolean argument asks if we should be reading the reseeded number or not",
|
||||
"Reseeded RNG calculation is more expensive and will be heavier to use",
|
||||
"Returns the 64-bit number",
|
||||
"Sets the Z flag if the number is valid.",
|
||||
"Falls back to a host RNG call when the hardware doesn't support it",
|
||||
"Sets the Z flag if the number is invalid.",
|
||||
"RNG hardware is allowed to fail early and return. Software must always check this"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
@@ -960,6 +961,20 @@
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"GPR = PatchableGuestRIP OpSize:#Size, i64:$Value, i64:$SiteAddress, i64:$SiteSize": {
|
||||
"Desc": ["Loads GuestRIP-relative Value in a patchable way",
|
||||
"On disk cache load the value is patched from live guest PC and live displacement at SiteAddress"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"GPR = PatchableGuestCRC OpSize:#Size, i64:$Value, i64:$GuestAddress, i64:$GuestSize": {
|
||||
"Desc": ["Loads Guest CRC in a patchable way",
|
||||
"On disk cache load the value is recomputed from live guest bytes and patched"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"GPR = Constant i64:$Constant, ConstPad:$Pad{IR::ConstPad::NoPad}, i32:$MaxBytes{0}": {
|
||||
"Desc": ["Generates a 64bit constant inside of a GPR",
|
||||
"Unsupported to create a constant in FPR"
|
||||
@@ -2363,6 +2378,43 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Unsigned by signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
|
||||
"Requires FEAT_I8MM."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "OpSize::i32Bit",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
]
|
||||
},
|
||||
"FPR = VSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
|
||||
"Requires FEAT_DotProd."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "OpSize::i32Bit",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
]
|
||||
},
|
||||
"FPR = VSAddLP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Signed add long pairwise. Adds adjacent pairs of elements in to elements of twice the size.",
|
||||
"ElementSize is the source size"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSAdALP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Acc, FPR:$Vector": {
|
||||
"Desc": ["Signed add and accumulate long pairwise. Adds adjacent pairs of elements in to the elements of Acc, which are twice the size.",
|
||||
"ElementSize is the source size"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VUMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Wide unsigned multiply returning the high results"],
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -2466,6 +2518,14 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VUCMPGT OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Vector compare unsigned greater than",
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero"
|
||||
],
|
||||
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFCMPEQ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
@@ -2533,7 +2593,8 @@
|
||||
},
|
||||
|
||||
"GPR = VPCMPESTRX FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u16:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
|
||||
"Desc": ["NOTE: Currently unused. The OpcodeDispatcher implements the SSE4.2 string instructions inline.",
|
||||
"Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
|
||||
"This will return the intermediate result of a PCMPESTR-type operation, but NOT the final",
|
||||
"result. This must be derived from the intermediate result",
|
||||
|
||||
@@ -2545,7 +2606,8 @@
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = VPCMPISTRX FPR:$LHS, FPR:$RHS, u8:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPISTRI/PCMPISTRM instruction",
|
||||
"Desc": ["NOTE: Currently unused. The OpcodeDispatcher implements the SSE4.2 string instructions inline.",
|
||||
"Performs intermediate behavior analogous to the x86 PCMPISTRI/PCMPISTRM instruction",
|
||||
"This will return the intermediate result of a PCMPISTR-type operation, but NOT the final",
|
||||
"result. This must be derived from the intermediate result",
|
||||
|
||||
|
||||
@@ -172,6 +172,8 @@ static void PrintArg(fextl::ostringstream* out, const IRListView*, NamedVectorCo
|
||||
return "u16_incremental_index";
|
||||
case NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER:
|
||||
return "u16_incremental_index_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_INCREMENTAL_U8_INDEX:
|
||||
return "u8_incremental_index";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT:
|
||||
return "addsubps_invert";
|
||||
case NamedVectorConstant::NAMED_VECTOR_PADDSUBPS_INVERT_UPPER:
|
||||
|
||||
@@ -19,6 +19,8 @@ namespace FEXCore::IR {
|
||||
|
||||
static bool IsFragmentExit(FEXCore::IR::IROps Op) {
|
||||
switch (Op) {
|
||||
case OP_THREADREMOVECODEENTRY:
|
||||
case OP_SYSCALL:
|
||||
case OP_EXITFUNCTION:
|
||||
case OP_BREAK: return true;
|
||||
default: return false;
|
||||
|
||||
@@ -218,7 +218,12 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
const uintptr_t StackLocation_u64 = reinterpret_cast<uintptr_t>(alloca(0));
|
||||
|
||||
const int MapsFD = open("/proc/self/maps", O_RDONLY);
|
||||
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
|
||||
if (MapsFD == -1) {
|
||||
// No procfs, as in BuildKit's emulator probe (an empty chroot). The assert
|
||||
// above it compiles out in release, and CollectMemoryGaps then spins on
|
||||
// read(-1) forever; reserve nothing instead.
|
||||
return {};
|
||||
}
|
||||
|
||||
auto Regions = CollectMemoryGaps(Begin, End, MapsFD);
|
||||
close(MapsFD);
|
||||
@@ -282,6 +287,9 @@ fextl::vector<MemoryRegion> Setup48BitAllocatorIfExists(size_t PageSize) {
|
||||
uintptr_t Begin48BitVA = 0x0'8000'0000'0000ULL;
|
||||
uintptr_t End48BitVA = 0x1'0000'0000'0000ULL;
|
||||
auto Regions = StealMemoryRegion(Begin48BitVA, End48BitVA);
|
||||
if (Regions.empty()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
Alloc64 = Alloc::OSAllocator::Create64BitAllocatorWithRegions(Regions);
|
||||
AssignHookOverrides(PageSize);
|
||||
|
||||
@@ -0,0 +1,366 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <dirent.h>
|
||||
#include <fcntl.h>
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <windows.h>
|
||||
#include <winnt.h>
|
||||
#include <winternl.h>
|
||||
#include <ntstatus.h>
|
||||
|
||||
// FEX today doesn't have Windows include in FEXCore.
|
||||
extern "C" NTSTATUS WINAPI NtQueryDirectoryFile(HANDLE, HANDLE, PIO_APC_ROUTINE, PVOID, PIO_STATUS_BLOCK, PVOID, ULONG,
|
||||
FILE_INFORMATION_CLASS, BOOLEAN, PUNICODE_STRING, BOOLEAN);
|
||||
|
||||
extern "C" NTSTATUS RtlUnicodeToUTF8N(OUT PCHAR UTF8StringDestination, IN ULONG UTF8StringMaxByteCount,
|
||||
OUT PULONG UTF8StringActualByteCount, IN PCWCH UnicodeStringSource, IN ULONG UnicodeStringByteCount);
|
||||
#endif
|
||||
|
||||
namespace FEXCore::FileUtils {
|
||||
#ifndef _WIN32
|
||||
static inline bool unlinkat(int fd, const char* path, bool dir) {
|
||||
if (::unlinkat(fd, path, dir ? AT_REMOVEDIR : 0) == -1) {
|
||||
return errno == ENOENT;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool RecursiveRemoveDirectory(int parent_fd, const char* Directory) {
|
||||
// Don't follow symlinks and ensure it closes on exec.
|
||||
constexpr int DIR_FLAGS = O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC;
|
||||
int dir_fd = ::openat(parent_fd, Directory, DIR_FLAGS);
|
||||
|
||||
if (dir_fd == -1) {
|
||||
if (errno == ENOENT) {
|
||||
// Probably raced something. Non-error.
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Walk the directory listing.
|
||||
bool Result = true;
|
||||
// Four pages arbitrary chosen to be a balance between NFS wanting to return data in page-size granules and
|
||||
// local filesystems returning arbitrary sizes.
|
||||
size_t dirent_size = 4096 * 4;
|
||||
uint8_t* dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
ssize_t read = getdents64(dir_fd, dirent_buffer, dirent_size);
|
||||
|
||||
if (read == -1) {
|
||||
if (errno == EINVAL) {
|
||||
// Buffer too small? Scale and try again.
|
||||
dirent_size *= 2;
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
// Anything else just exit.
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
|
||||
if (read == 0) {
|
||||
// Done.
|
||||
break;
|
||||
}
|
||||
|
||||
for (size_t dirent_offset = 0; dirent_offset < read;) {
|
||||
auto path_dirent = reinterpret_cast<const struct dirent*>(dirent_buffer + dirent_offset);
|
||||
std::string_view path_name_view = path_dirent->d_name;
|
||||
|
||||
if (path_name_view == "." || path_name_view == "..") {
|
||||
// Skip these two special files.
|
||||
dirent_offset += path_dirent->d_reclen;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (path_dirent->d_type == DT_DIR) {
|
||||
// Recurse directories as we find them and remove them.
|
||||
if (!RecursiveRemoveDirectory(dir_fd, path_dirent->d_name)) {
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
}
|
||||
|
||||
// Remove anything possible.
|
||||
if (!unlinkat(dir_fd, path_dirent->d_name, path_dirent->d_type == DT_DIR)) {
|
||||
// Couldn't unlink the file for some reason.
|
||||
LogMan::Msg::IFmt("Failed to remove file: {}", path_name_view);
|
||||
Result = false;
|
||||
goto end;
|
||||
}
|
||||
|
||||
// dirent is a VLA so we need to increment by reported size.
|
||||
dirent_offset += path_dirent->d_reclen;
|
||||
}
|
||||
}
|
||||
|
||||
end:
|
||||
if (dirent_buffer) {
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
}
|
||||
|
||||
close(dir_fd);
|
||||
return Result;
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool RecursiveRemoveDirectory(const fextl::string& Directory) {
|
||||
return RecursiveRemoveDirectory(AT_FDCWD, Directory.c_str()) && unlinkat(AT_FDCWD, Directory.c_str(), true);
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void WalkDirectory(std::string_view Directory,
|
||||
fextl::move_only_function<void(std::string_view name, bool is_dir, const void* user_data)> Callback,
|
||||
const void* user_data) {
|
||||
constexpr int DIR_FLAGS = O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC;
|
||||
int dir_fd = ::openat(AT_FDCWD, fextl::string(Directory).c_str(), DIR_FLAGS);
|
||||
if (dir_fd == -1) {
|
||||
if (errno == ENOENT) {
|
||||
// Probably raced something. Non-error.
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Four pages arbitrary chosen to be a balance between NFS wanting to return data in page-size granules and
|
||||
// local filesystems returning arbitrary sizes.
|
||||
size_t dirent_size = 4096 * 4;
|
||||
uint8_t* dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
goto end;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
ssize_t read = getdents64(dir_fd, dirent_buffer, dirent_size);
|
||||
|
||||
if (read == -1) {
|
||||
if (errno == EINVAL) {
|
||||
// Buffer too small? Scale and try again.
|
||||
dirent_size *= 2;
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
goto end;
|
||||
}
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
// Anything else just exit.
|
||||
goto end;
|
||||
}
|
||||
|
||||
if (read == 0) {
|
||||
// Done.
|
||||
break;
|
||||
}
|
||||
|
||||
for (size_t dirent_offset = 0; dirent_offset < read;) {
|
||||
auto path_dirent = reinterpret_cast<const struct dirent*>(dirent_buffer + dirent_offset);
|
||||
std::string_view path_name_view = path_dirent->d_name;
|
||||
|
||||
if (path_name_view == "." || path_name_view == "..") {
|
||||
// Skip these two special files.
|
||||
dirent_offset += path_dirent->d_reclen;
|
||||
continue;
|
||||
}
|
||||
|
||||
Callback(path_name_view, path_dirent->d_type == DT_DIR, user_data);
|
||||
|
||||
// dirent is a VLA so we need to increment by reported size.
|
||||
dirent_offset += path_dirent->d_reclen;
|
||||
}
|
||||
}
|
||||
|
||||
end:
|
||||
if (dirent_buffer) {
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
}
|
||||
|
||||
close(dir_fd);
|
||||
}
|
||||
|
||||
#else
|
||||
static inline std::optional<UNICODE_STRING> PathToNTPath(std::string_view Path) {
|
||||
UNICODE_STRING PathW;
|
||||
if (!RtlCreateUnicodeStringFromAsciiz(&PathW, fextl::string(Path).c_str())) {
|
||||
return std::nullopt;
|
||||
}
|
||||
UNICODE_STRING NTPath;
|
||||
bool Success = RtlDosPathNameToNtPathName_U(PathW.Buffer, &NTPath, nullptr, nullptr);
|
||||
RtlFreeUnicodeString(&PathW);
|
||||
if (!Success) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
return NTPath;
|
||||
}
|
||||
|
||||
static inline void FreeNTPath(UNICODE_STRING Path) {
|
||||
RtlFreeUnicodeString(&Path);
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void WalkDirectory(std::string_view Directory,
|
||||
fextl::move_only_function<void(std::string_view name, bool is_dir, const void* user_data)> Callback,
|
||||
const void* user_data) {
|
||||
auto NTPath = PathToNTPath(Directory);
|
||||
if (!NTPath) {
|
||||
return;
|
||||
}
|
||||
|
||||
NTSTATUS Status {};
|
||||
|
||||
HANDLE dir_fd {};
|
||||
OBJECT_ATTRIBUTES attr {};
|
||||
IO_STATUS_BLOCK io {};
|
||||
|
||||
InitializeObjectAttributes(&attr, &*NTPath, OBJ_CASE_INSENSITIVE, nullptr, nullptr);
|
||||
Status = NtOpenFile(&dir_fd, FILE_LIST_DIRECTORY | SYNCHRONIZE, &attr, &io, FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE,
|
||||
FILE_DIRECTORY_FILE | FILE_SYNCHRONOUS_IO_NONALERT);
|
||||
|
||||
FreeNTPath(*NTPath);
|
||||
|
||||
if (!NT_SUCCESS(Status)) {
|
||||
return;
|
||||
}
|
||||
|
||||
BOOLEAN FirstQuery = TRUE;
|
||||
size_t dirent_size = 4096 * 4;
|
||||
uint8_t* dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
goto end;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
Status = NtQueryDirectoryFile(dir_fd,
|
||||
nullptr, // Event
|
||||
nullptr, // ApcRoutine
|
||||
nullptr, // ApcContext
|
||||
&io, dirent_buffer, dirent_size, FileDirectoryInformation,
|
||||
FALSE, // ReturnSingleEntry
|
||||
nullptr, // FileName
|
||||
FirstQuery);
|
||||
FirstQuery = FALSE;
|
||||
|
||||
if (Status == STATUS_NO_MORE_FILES) {
|
||||
// No more files
|
||||
break;
|
||||
}
|
||||
|
||||
if (!NT_SUCCESS(Status)) {
|
||||
if (Status == STATUS_BUFFER_TOO_SMALL || Status == STATUS_INFO_LENGTH_MISMATCH) {
|
||||
// Buffer too small? Scale and try again.
|
||||
dirent_size *= 2;
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
uint8_t* dirent_buffer = reinterpret_cast<uint8_t*>(FEXCore::Allocator::malloc(dirent_size));
|
||||
|
||||
if (!dirent_buffer) {
|
||||
// Ran out of memory?
|
||||
goto end;
|
||||
}
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
// Any other failure, exit loop.
|
||||
break;
|
||||
}
|
||||
|
||||
// Iterate the entries returned.
|
||||
for (size_t dirent_offset = 0;;) {
|
||||
auto Info = reinterpret_cast<FILE_DIRECTORY_INFORMATION*>(dirent_buffer + dirent_offset);
|
||||
|
||||
std::wstring_view EntryName(Info->FileName, Info->FileNameLength / sizeof(wchar_t));
|
||||
|
||||
if (EntryName != L"." && EntryName != L"..") {
|
||||
ULONG utf8_bytes_needed = 0;
|
||||
ULONG actual_utf8_bytes = 0;
|
||||
RtlUnicodeToUTF8N(nullptr, 0, &utf8_bytes_needed, Info->FileName, Info->FileNameLength);
|
||||
char* dynamic_buf = reinterpret_cast<char*>(FEXCore::Allocator::malloc(utf8_bytes_needed + 1));
|
||||
if (dynamic_buf) {
|
||||
RtlUnicodeToUTF8N(dynamic_buf, utf8_bytes_needed + 1, &actual_utf8_bytes, Info->FileName, Info->FileNameLength);
|
||||
std::string_view name_view(dynamic_buf, actual_utf8_bytes);
|
||||
bool is_dir = (Info->FileAttributes & FILE_ATTRIBUTE_DIRECTORY) != 0;
|
||||
Callback(name_view, is_dir, user_data);
|
||||
FEXCore::Allocator::free(dynamic_buf);
|
||||
}
|
||||
}
|
||||
|
||||
if (Info->NextEntryOffset == 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
dirent_offset += Info->NextEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
end:
|
||||
if (dirent_buffer) {
|
||||
FEXCore::Allocator::free(dirent_buffer);
|
||||
}
|
||||
|
||||
NtClose(dir_fd);
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool RecursiveRemoveDirectory(const fextl::string& Directory) {
|
||||
using CallbackType = void (*)(std::string_view name, bool is_dir, const void* user_data);
|
||||
struct UserData {
|
||||
CallbackType RemoveFile {};
|
||||
std::string_view base_path;
|
||||
};
|
||||
|
||||
CallbackType RemoveFile = [](std::string_view name, bool is_dir, const void* user_data) {
|
||||
auto Data = reinterpret_cast<const UserData*>(user_data);
|
||||
|
||||
auto full_path = std::format("{}/{}", Data->base_path, name);
|
||||
if (is_dir) {
|
||||
UserData NewData {
|
||||
.RemoveFile = Data->RemoveFile,
|
||||
.base_path = full_path,
|
||||
};
|
||||
|
||||
WalkDirectory(full_path, Data->RemoveFile, &NewData);
|
||||
}
|
||||
|
||||
if (is_dir) {
|
||||
RemoveDirectoryA(full_path.c_str());
|
||||
} else {
|
||||
DeleteFileA(full_path.c_str());
|
||||
}
|
||||
};
|
||||
|
||||
UserData Data {
|
||||
.RemoveFile = RemoveFile,
|
||||
.base_path = Directory,
|
||||
};
|
||||
|
||||
WalkDirectory(Directory, RemoveFile, &Data);
|
||||
|
||||
return RemoveDirectoryA(Directory.c_str()) != 0;
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::FileUtils
|
||||
@@ -7,7 +7,8 @@
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Threads {
|
||||
static fextl::unique_ptr<FEXCore::Threads::Thread> CreateThread_Default(ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags) {
|
||||
static fextl::unique_ptr<FEXCore::Threads::Thread>
|
||||
CreateThread_Default(ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags, const char* ThreadName) {
|
||||
ERROR_AND_DIE_FMT("Frontend didn't setup thread creation!");
|
||||
}
|
||||
|
||||
@@ -20,8 +21,9 @@ static FEXCore::Threads::Pointers Ptrs = {
|
||||
.CleanupAfterFork = CleanupAfterFork_Default,
|
||||
};
|
||||
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> FEXCore::Threads::Thread::Create(ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags) {
|
||||
return Ptrs.CreateThread(Func, Arg, Flags);
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread>
|
||||
FEXCore::Threads::Thread::Create(ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags, const char* ThreadName) {
|
||||
return Ptrs.CreateThread(Func, Arg, Flags, ThreadName);
|
||||
}
|
||||
|
||||
void FEXCore::Threads::Thread::CleanupAfterFork() {
|
||||
|
||||
@@ -4,8 +4,8 @@
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
WorkQueueThread::WorkQueueThread(FEXCore::Threads::Flags ThreadFlags) {
|
||||
Thread = FEXCore::Threads::Thread::Create(ThreadEntry, this, ThreadFlags);
|
||||
WorkQueueThread::WorkQueueThread(FEXCore::Threads::Flags ThreadFlags, const char* ThreadName) {
|
||||
Thread = FEXCore::Threads::Thread::Create(ThreadEntry, this, ThreadFlags, ThreadName);
|
||||
}
|
||||
|
||||
WorkQueueThread::~WorkQueueThread() {
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
#include <arm_acle.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
template<typename T>
|
||||
static inline uint32_t crc32(const T* Ptr, size_t Size) {
|
||||
uint32_t Result {};
|
||||
#if defined(ARCHITECTURE_arm64)
|
||||
#define do_crc(type, suffix) \
|
||||
while (Size >= sizeof(type)) { \
|
||||
Result = __crc32##suffix(Result, *reinterpret_cast<const type*>(Ptr)); \
|
||||
Ptr += sizeof(type); \
|
||||
Size -= sizeof(type); \
|
||||
}
|
||||
do_crc(uint64_t, d);
|
||||
do_crc(uint32_t, w);
|
||||
do_crc(uint16_t, h);
|
||||
do_crc(uint8_t, b);
|
||||
#undef do_crc
|
||||
#else
|
||||
// Emulate arm64 crc32.
|
||||
// This is basically just the pseudo-code for crc32b.
|
||||
// Doesn't need to be fast, just needs to match.
|
||||
auto reverse_bits = [](auto bits) {
|
||||
decltype(bits) Result {};
|
||||
for (size_t i = 0; i < (sizeof(decltype(bits)) * 8); ++i) {
|
||||
Result = (Result << 1) | ((bits >> i) & 1);
|
||||
}
|
||||
return Result;
|
||||
};
|
||||
|
||||
auto Poly32Mod2 = [](uint64_t data) -> uint32_t {
|
||||
constexpr static size_t bits = 40;
|
||||
constexpr static uint64_t poly = 0x04C11DB7U;
|
||||
for (size_t i = (bits - 1); i >= 32; --i) {
|
||||
if (((data >> i) & 1) != 0) {
|
||||
const uint64_t poly_shift = poly << (i - 32);
|
||||
const uint64_t data_mask = (1ULL << i) - 1;
|
||||
data = (data & data_mask) ^ poly_shift;
|
||||
}
|
||||
}
|
||||
|
||||
return data;
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < Size; ++i) {
|
||||
uint64_t TempAcc = static_cast<uint64_t>(reverse_bits(Result)) << 8;
|
||||
uint64_t TempVal = static_cast<uint64_t>(reverse_bits(reinterpret_cast<const uint8_t*>(Ptr)[i])) << 32;
|
||||
Result = reverse_bits(Poly32Mod2(TempAcc ^ TempVal));
|
||||
}
|
||||
#endif
|
||||
return Result;
|
||||
}
|
||||
} // namespace FEXCore::Utils
|
||||
@@ -297,6 +297,16 @@ struct FEX_DEFAULT_VISIBILITY Getter : public Value<typename detail::ConfigOptio
|
||||
using OptionInfo = detail::ConfigOptionInfo<Option>;
|
||||
Getter()
|
||||
: Value<typename OptionInfo::Type> {Option, OptionInfo::Default()} {}
|
||||
|
||||
// Check if the config option is actually the default.
|
||||
bool IsDefault() const {
|
||||
auto Value = FEXCore::Config::GetConv<typename detail::ConfigOptionInfo<Option>::Type>(Option);
|
||||
if (!Value) {
|
||||
return true;
|
||||
}
|
||||
|
||||
return *Value == OptionInfo::Default();
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -51,6 +51,7 @@ struct ExecutableFileInfo {
|
||||
uint64_t FileId = 0;
|
||||
fextl::string Filename;
|
||||
fextl::robin_map<uint32_t, GuestRelocationType> Relocations;
|
||||
uint64_t MappedSize = 0;
|
||||
};
|
||||
|
||||
// Information associated with a specific section of an executable file
|
||||
@@ -234,14 +235,6 @@ class AbstractCodeCache {
|
||||
public:
|
||||
virtual ~AbstractCodeCache() = default;
|
||||
|
||||
/**
|
||||
* Computes a unique identifier for the referenced binary file to be used for
|
||||
* generating the code map.
|
||||
* This identifier is independent of FEX build/runtime configuration and
|
||||
* stable across FEX updates.
|
||||
*/
|
||||
virtual uint64_t ComputeCodeMapId(std::string_view Filename, int FD) = 0;
|
||||
|
||||
/**
|
||||
* Bundles the current Core state (CodeBuffer, GuestToHostMapping, ...) to a code cache and writes it to the given file descriptor.
|
||||
* Returns true on success.
|
||||
|
||||
@@ -350,6 +350,7 @@ struct JITPointers {
|
||||
uint64_t MonoBackpatcherWrite {};
|
||||
uint64_t LUDIV {};
|
||||
uint64_t LDIV {};
|
||||
uint64_t RDRANDFallback {};
|
||||
uint64_t ThunkCallbackRet {};
|
||||
|
||||
// Handles returning/calling ARM64EC code from the JIT, expects the target PC in TMP3
|
||||
@@ -370,6 +371,8 @@ struct JITPointers {
|
||||
uint64_t ExitFunctionLinker {};
|
||||
uint64_t ThreadStopHandlerSpillSRA {};
|
||||
uint64_t ThreadPauseHandlerSpillSRA {};
|
||||
uint64_t ThreadDispatchSyscallHandler {};
|
||||
uint64_t ThreadDispatchRemoveCodeEntry {};
|
||||
uint64_t GuestSignal_SIGILL {};
|
||||
uint64_t GuestSignal_SIGTRAP {};
|
||||
uint64_t GuestSignal_SIGSEGV {};
|
||||
@@ -427,7 +430,7 @@ struct CpuStateFrame {
|
||||
|
||||
InternalThreadState* Thread;
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
#ifdef _WIN32
|
||||
// Set by the kernel on ARM64EC whenever the JIT should cooperatively suspend running guest code.
|
||||
uint32_t SuspendDoorbell {};
|
||||
#endif
|
||||
|
||||
@@ -1,231 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "FEXCore/Core/CodeCache.h"
|
||||
#include "FEXCore/Core/Context.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "FEXCore/Config/Config.h"
|
||||
#include "FEXCore/Utils/File.h"
|
||||
#include "FEXCore/Utils/WorkQueueThread.h"
|
||||
#include "FEXCore/fextl/memory.h"
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_set.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <stdint.h>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <span>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore {
|
||||
#include "FEXCore/Utils/CompilerDefs.h"
|
||||
|
||||
namespace Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
#include <cstdint>
|
||||
|
||||
namespace DiskCache {
|
||||
|
||||
namespace MesaFOZ {
|
||||
|
||||
#define FOSSILIZE_BLOB_HASH_LENGTH 40 /* SHA1 hexadecimal string length */
|
||||
|
||||
struct __attribute__((packed)) foz_payload_key {
|
||||
uint8_t bytes[FOSSILIZE_BLOB_HASH_LENGTH];
|
||||
};
|
||||
|
||||
struct __attribute__((packed)) foz_payload_header {
|
||||
uint32_t payload_size;
|
||||
uint32_t format;
|
||||
uint32_t crc;
|
||||
uint32_t uncompressed_size;
|
||||
};
|
||||
|
||||
} // namespace MesaFOZ
|
||||
|
||||
class IndexedDB;
|
||||
|
||||
struct IndexEntry {
|
||||
IndexedDB* DB;
|
||||
uint64_t Offset;
|
||||
uint32_t Size;
|
||||
uint32_t GuestSize;
|
||||
XXH128_hash_t GuestHash;
|
||||
fextl::vector<uint32_t> GuestExtents;
|
||||
};
|
||||
|
||||
struct IndexCacheHead {
|
||||
struct IndexEntry MainEntry;
|
||||
uint64_t MainEntryFootprint;
|
||||
fextl::unique_ptr<fextl::multimap<uint64_t, IndexEntry>> MoreEntries; // sorted by guest footprint
|
||||
};
|
||||
|
||||
struct __attribute__((packed)) BlobFixedHeader {
|
||||
uint32_t GuestSize;
|
||||
uint32_t HostSize;
|
||||
uint32_t EntryPointCount;
|
||||
uint32_t SmallRelocCount;
|
||||
uint32_t ThunkRelocCount;
|
||||
XXH128_hash_t GuestHash;
|
||||
};
|
||||
|
||||
// packed struct for types 0, 2 and 3. type 1 is bigger and separate below
|
||||
struct __attribute__((packed)) BlobSmallRelocation {
|
||||
uint32_t Offset;
|
||||
uint8_t Type;
|
||||
union {
|
||||
struct __attribute__((packed)) {
|
||||
uint32_t Symbol;
|
||||
} Named;
|
||||
struct __attribute__((packed)) {
|
||||
uint64_t GuestRIP;
|
||||
} RIPLiteral;
|
||||
struct __attribute__((packed)) {
|
||||
uint8_t RegisterIndex;
|
||||
uint64_t GuestRIP;
|
||||
} RIPMove;
|
||||
struct __attribute__((packed)) {
|
||||
uint8_t RegisterIndex;
|
||||
uint8_t ValueSize;
|
||||
uint32_t SiteOffset;
|
||||
} PatchableData;
|
||||
};
|
||||
};
|
||||
|
||||
// type 1, implicit
|
||||
struct __attribute__((packed)) BlobThunkRelocation {
|
||||
uint32_t Offset;
|
||||
uint8_t RegisterIndex;
|
||||
uint8_t SymbolHash[32]; // sha256sum in the real RelocNamedThunkMove
|
||||
};
|
||||
|
||||
struct CodeHitData {
|
||||
fextl::vector<uint8_t> Blob;
|
||||
std::span<uint8_t> HostCode;
|
||||
std::span<const uint64_t> GuestPages;
|
||||
std::span<uint64_t> EntryPointRIPs;
|
||||
std::span<const uint32_t> EntryPointHostOffsets;
|
||||
|
||||
// the spans above point to memory owned by the Blob vec, so it's important this can't be copied
|
||||
CodeHitData() = default;
|
||||
CodeHitData(CodeHitData&&) = default;
|
||||
CodeHitData& operator=(CodeHitData&&) = default;
|
||||
CodeHitData(const CodeHitData&) = delete;
|
||||
CodeHitData& operator=(const CodeHitData&) = delete;
|
||||
};
|
||||
|
||||
using Index = fextl::robin_map<uint64_t, IndexCacheHead>;
|
||||
|
||||
class FOZFile {
|
||||
public:
|
||||
bool Open(const fextl::string& CacheFileName, bool ReadOnly);
|
||||
bool Lock(uint32_t TimeoutMS) {
|
||||
if (!FD) {
|
||||
return false;
|
||||
}
|
||||
return FD->Lock(TimeoutMS);
|
||||
}
|
||||
bool Unlock() {
|
||||
if (!FD) {
|
||||
return false;
|
||||
}
|
||||
return FD->Unlock();
|
||||
}
|
||||
File::File::FileHandleType GetHandle() {
|
||||
return FD ? FD->GetHandle() : (File::File::FileHandleType)-1;
|
||||
}
|
||||
ssize_t Size();
|
||||
bool ReadAll(fextl::vector<uint8_t>& Out); // from first blob
|
||||
bool ReadBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
|
||||
bool WriteBlob(const MesaFOZ::foz_payload_key& Key, std::span<const std::span<const uint8_t>> BlobChunks, uint64_t& OutBlobOffset);
|
||||
|
||||
private:
|
||||
static constexpr uint32_t OPEN_LOCK_TIMEOUT_MS = 100;
|
||||
|
||||
fextl::string FileName;
|
||||
fextl::unique_ptr<File::File> FD;
|
||||
bool ReadOnly = false;
|
||||
};
|
||||
|
||||
class IndexedDB {
|
||||
public:
|
||||
bool Open(const fextl::string& CacheDBName, bool ReadOnly);
|
||||
void PopulateIndex(Index& CacheIndex, bool& FoundMetadata);
|
||||
bool ReadCacheBlob(uint64_t Offset, std::span<uint8_t> OutBlob);
|
||||
bool StoreCacheBlob(const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey, std::span<const uint8_t> Blob, Index& CacheIndex,
|
||||
std::mutex& IndexMutex, std::span<const uint8_t> IndexBlob);
|
||||
|
||||
private:
|
||||
// stores run on the Writer, so returning quick isn't as important
|
||||
static constexpr uint32_t STORE_LOCK_TIMEOUT_MS = 1000;
|
||||
static constexpr uint64_t BIG_MAPPING_SIZE = 1ULL << 33;
|
||||
static constexpr uint32_t LOOKUP_KEY_MAX_BUCKET_DEPTH = 20;
|
||||
|
||||
FOZFile CacheFOZ;
|
||||
uint8_t* CacheFileMapping = nullptr;
|
||||
std::atomic<uint64_t> CacheFileSize;
|
||||
FOZFile IndexFOZ;
|
||||
bool ReadOnly = false;
|
||||
};
|
||||
|
||||
class DiskCache {
|
||||
public:
|
||||
void Init(FEXCore::Context::ContextImpl* CTX);
|
||||
|
||||
std::optional<CodeHitData> Lookup(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP,
|
||||
std::optional<uint64_t>& GuestCodeKey);
|
||||
void Validate(uint64_t GuestCodeKey, const CodeHitData& Hit, const CPU::CPUBackend::CompiledCode& CompiledCode,
|
||||
std::optional<ExecutableFileSectionInfo> Region);
|
||||
bool Store(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP, uint64_t GuestCodeKey,
|
||||
std::span<const uint8_t> GuestCode, const CPU::CPUBackend::CompiledCode& CompiledCode,
|
||||
std::span<const FEXCore::CPU::Relocation> Relocations, const Frontend::Decoder::DecodedBlockInformation* DecodedBlockInfo);
|
||||
|
||||
bool IsWritingDiskCache() const {
|
||||
return WritingDiskCache;
|
||||
}
|
||||
bool IsReadingDiskCache() const {
|
||||
return ReadingDiskCache;
|
||||
}
|
||||
bool IsValidating() const {
|
||||
return Validation;
|
||||
}
|
||||
|
||||
private:
|
||||
bool OpenCacheDB(const fextl::string& CacheDBName, bool ReadOnly);
|
||||
uint64_t MakeLookupKey(Core::InternalThreadState* Thread, const uint64_t ModuleOffset, bool Writable, bool MonoBackpatcher);
|
||||
|
||||
bool ReadingDiskCache {};
|
||||
bool WritingDiskCache {};
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
XXH128_hash_t BucketHash;
|
||||
fextl::vector<fextl::unique_ptr<IndexedDB>> ROCacheDBs;
|
||||
fextl::unique_ptr<IndexedDB> RWCacheDB;
|
||||
Index Index;
|
||||
std::mutex IndexLock;
|
||||
bool FoundMetadata = false;
|
||||
struct CacheStoreWorkItem;
|
||||
|
||||
// the Writer holds references to all this stuff above and needs to be last
|
||||
fextl::unique_ptr<WorkQueueThread> Writer;
|
||||
|
||||
FEX_CONFIG_OPT(EnableDiskCache, DISKCACHE);
|
||||
FEX_CONFIG_OPT(Validation, DISKCACHEVALIDATION);
|
||||
FEX_CONFIG_OPT(MapDiskCacheFiles, DISKCACHEFILEMAPPING);
|
||||
FEX_CONFIG_OPT(RelocationFilter, DISKCACHERELOCATIONFILTER);
|
||||
FEX_CONFIG_OPT(AnonCaching, DISKCACHEANONCACHING);
|
||||
FEX_CONFIG_OPT(BasePathOverride, DISKCACHEPATH);
|
||||
FEX_CONFIG_OPT(RODBNames, DISKCACHERODBNAMES);
|
||||
};
|
||||
|
||||
static constexpr uint16_t AnonPrefixGuestBytes = 64;
|
||||
|
||||
// TODO: This header is in global installed header path, but uses internal headers.
|
||||
// Migrate this once that is fixed.
|
||||
static constexpr uint16_t FormatVersion = 19;
|
||||
FEX_DEFAULT_VISIBILITY uint16_t GetFormatVersion();
|
||||
|
||||
} // namespace DiskCache
|
||||
|
||||
} // namespace FEXCore
|
||||
namespace FEXCore::DiskCache {
|
||||
FEX_DEFAULT_VISIBILITY uint16_t GetFormatVersion();
|
||||
} // namespace FEXCore::DiskCache
|
||||
@@ -32,16 +32,23 @@ struct HostFeatures {
|
||||
return 4 << DCacheLineLog2;
|
||||
}
|
||||
|
||||
struct CacheHash {
|
||||
uint64_t HostFeaturesHash;
|
||||
HostTypeEnum HostType;
|
||||
};
|
||||
|
||||
[[nodiscard]]
|
||||
uint64_t HashForCaching() const {
|
||||
// As long as the number of options is 64-bit or below, we can just return it.
|
||||
// Skip CPUMIDRs as it doesn't affect codegen.
|
||||
static_assert(offsetof(HostFeatures, CPUMIDRs) == 8);
|
||||
uint64_t Result {};
|
||||
memcpy(&Result, this, sizeof(Result));
|
||||
CacheHash HashForCaching() const {
|
||||
static_assert(offsetof(HostFeatures, HostType) == 8);
|
||||
|
||||
CacheHash Result {};
|
||||
memcpy(&Result.HostFeaturesHash, this, sizeof(uint64_t));
|
||||
Result.HostType = HostType;
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
// Affects codegen and is basically machine description.
|
||||
uint32_t DCacheLineLog2 : 4 {};
|
||||
uint32_t SupportsCacheMaintenanceOps : 1 {};
|
||||
uint32_t SupportsAES : 1 {};
|
||||
@@ -71,16 +78,24 @@ struct HostFeatures {
|
||||
uint32_t Supports3DNow : 1 {};
|
||||
uint32_t SupportsSSE4a : 1 {};
|
||||
uint32_t SupportsMOPS : 1 {};
|
||||
uint32_t SupportsI8MM : 1 {};
|
||||
uint32_t SupportsDotProd : 1 {};
|
||||
uint32_t PreferZVAForVZero : 1 {};
|
||||
uint32_t SupportsAFP : 1 {};
|
||||
uint32_t SupportsFloatExceptions : 1 {};
|
||||
// Flag if this is InstCountCI
|
||||
uint32_t IsInstCountCI : 1 {};
|
||||
HostTypeEnum HostType : 2 {};
|
||||
uint32_t pad : 26 {};
|
||||
|
||||
// This affects codegen, but it isn't machine state
|
||||
HostTypeEnum HostType {};
|
||||
|
||||
// MIDR information
|
||||
// Also used for determining number of CPU cores for CPUID
|
||||
fextl::vector<uint32_t> CPUMIDRs;
|
||||
|
||||
// The Linux PID of this process. Useful for punching through perf-top information.
|
||||
uint32_t ProcessPID {};
|
||||
uint32_t pad2 {};
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -15,6 +15,7 @@ namespace FEXCore::IR {
|
||||
enum NamedVectorConstant : uint8_t {
|
||||
NAMED_VECTOR_INCREMENTAL_U16_INDEX = 0,
|
||||
NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER,
|
||||
NAMED_VECTOR_INCREMENTAL_U8_INDEX,
|
||||
NAMED_VECTOR_PADDSUBPS_INVERT,
|
||||
NAMED_VECTOR_PADDSUBPS_INVERT_UPPER,
|
||||
NAMED_VECTOR_PADDSUBPD_INVERT,
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::FileUtils {
|
||||
/**
|
||||
* @brief Removes an entirely directory passed in. Including the path itself.
|
||||
*
|
||||
* @return True if the directory didn't exist or was deleted.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY bool RecursiveRemoveDirectory(const fextl::string& Directory);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void WalkDirectory(std::string_view Directory,
|
||||
fextl::move_only_function<void(std::string_view name, bool is_dir, const void* user_data)> Callback,
|
||||
const void* user_data);
|
||||
} // namespace FEXCore::FileUtils
|
||||
@@ -13,7 +13,7 @@ struct Flags {
|
||||
using ThreadFunc = void* (*)(void* user_ptr);
|
||||
|
||||
class Thread;
|
||||
using CreateThreadFunc = fextl::unique_ptr<Thread> (*)(ThreadFunc Func, void* Arg, Flags Flags);
|
||||
using CreateThreadFunc = fextl::unique_ptr<Thread> (*)(ThreadFunc Func, void* Arg, Flags Flags, const char* ThreadName);
|
||||
using CleanupAfterForkFunc = void (*)();
|
||||
|
||||
struct Pointers {
|
||||
@@ -34,7 +34,7 @@ public:
|
||||
* @name Calls provided API functions
|
||||
* @{ */
|
||||
|
||||
static fextl::unique_ptr<Thread> Create(ThreadFunc Func, void* Arg, Flags Flags = {});
|
||||
static fextl::unique_ptr<Thread> Create(ThreadFunc Func, void* Arg, Flags Flags = {}, const char* ThreadName = nullptr);
|
||||
|
||||
static void CleanupAfterFork();
|
||||
|
||||
|
||||
@@ -18,7 +18,7 @@ public:
|
||||
virtual void Run() = 0;
|
||||
};
|
||||
|
||||
WorkQueueThread(FEXCore::Threads::Flags Flags = {});
|
||||
WorkQueueThread(FEXCore::Threads::Flags Flags = {}, const char* ThreadName = nullptr);
|
||||
~WorkQueueThread();
|
||||
|
||||
void QueueWork(fextl::unique_ptr<WorkItem> Work);
|
||||
|
||||
@@ -47,7 +47,7 @@ public:
|
||||
|
||||
// Second, wrap the relocated argument in a single-capture lambda
|
||||
auto wrapped_lambda = [moved_lambda](Args... args) {
|
||||
return (*moved_lambda)(std::forward<Args>(args)...);
|
||||
return std::invoke(*moved_lambda, std::forward<Args>(args)...);
|
||||
};
|
||||
|
||||
// Third, assign the result to std::function, ensuring it's indeed
|
||||
|
||||
@@ -6,6 +6,7 @@ foreach(TEST ${TESTS})
|
||||
add_executable(FEXCore_Tests_${TEST_NAME} ${TEST})
|
||||
target_link_libraries(FEXCore_Tests_${TEST_NAME} PRIVATE ${LIBS})
|
||||
target_include_directories(FEXCore_Tests_${TEST_NAME} PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/../../Source/")
|
||||
target_compile_options(FEXCore_Tests_${TEST_NAME} PRIVATE ${FEX_TUNE_COMPILE_FLAGS})
|
||||
set_target_properties(FEXCore_Tests_${TEST_NAME} PROPERTIES RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/FEXCore_Tests")
|
||||
catch_discover_tests(FEXCore_Tests_${TEST_NAME} TEST_SUFFIX ".${TEST_NAME}.FEXCore_Tests")
|
||||
endforeach()
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <catch2/catch_all.hpp>
|
||||
|
||||
#include "Utils/crc32.h"
|
||||
|
||||
TEST_CASE("Simple") {
|
||||
uint32_t data = 0x41424344U;
|
||||
CHECK(FEXCore::Utils::crc32(&data, sizeof(data)) == 0xa53ea072);
|
||||
}
|
||||
@@ -3937,12 +3937,9 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE load and broadcast element
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i8Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rb {z30.b}, p6/z, [x29, #31]");
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i8Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rb {z30.b}, p6/z, [x29, #63]");
|
||||
|
||||
// TODO: Several instances are commented out due to a reported bug in the vixl dissassembler.
|
||||
// Uncomment these when it's fixed.
|
||||
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rb {z30.h}, p6/z, [x29]");
|
||||
// TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rb {z30.h}, p6/z, [x29, #31]");
|
||||
// TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rb {z30.h}, p6/z, [x29, #63]");
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rb {z30.h}, p6/z, [x29, #31]");
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rb {z30.h}, p6/z, [x29, #63]");
|
||||
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rb {z30.s}, p6/z, [x29]");
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rb {z30.s}, p6/z, [x29, #31]");
|
||||
@@ -3953,8 +3950,8 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE load and broadcast element
|
||||
TEST_SINGLE(ld1rb(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rb {z30.d}, p6/z, [x29, #63]");
|
||||
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rsb {z30.h}, p6/z, [x29]");
|
||||
// TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rsb {z30.h}, p6/z, [x29, #31]");
|
||||
// TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rsb {z30.h}, p6/z, [x29, #63]");
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rsb {z30.h}, p6/z, [x29, #31]");
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rsb {z30.h}, p6/z, [x29, #63]");
|
||||
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rsb {z30.s}, p6/z, [x29]");
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 31), "ld1rsb {z30.s}, p6/z, [x29, #31]");
|
||||
@@ -3965,8 +3962,8 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE load and broadcast element
|
||||
TEST_SINGLE(ld1rsb(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 63), "ld1rsb {z30.d}, p6/z, [x29, #63]");
|
||||
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rh {z30.h}, p6/z, [x29]");
|
||||
// TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 64), "ld1rh {z30.h}, p6/z, [x29, #64]");
|
||||
// TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 126), "ld1rh {z30.h}, p6/z, [x29, #126]");
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 64), "ld1rh {z30.h}, p6/z, [x29, #64]");
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 126), "ld1rh {z30.h}, p6/z, [x29, #126]");
|
||||
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 0), "ld1rh {z30.s}, p6/z, [x29]");
|
||||
TEST_SINGLE(ld1rh(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Zeroing(), Reg::r29, 64), "ld1rh {z30.s}, p6/z, [x29, #64]");
|
||||
|
||||
@@ -660,25 +660,22 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Advanced SIMD scalar x inde
|
||||
TEST_SINGLE(sqrdmulh(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "sqrdmulh s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(sqrdmulh(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "sqrdmulh s30, s29, v28.s[3]");
|
||||
|
||||
// TODO: Commented out due to a bug in vixl's decoder (which has been reported).
|
||||
// Uncomment these when fixed.
|
||||
|
||||
// TEST_SINGLE(fmla(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmla h30, h29, v15.h[4]");
|
||||
// TEST_SINGLE(fmla(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmla h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmla h30, h29, v15.h[4]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmla h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmla s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "fmla s30, s29, v28.s[3]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmla d30, d29, v28.d[0]");
|
||||
TEST_SINGLE(fmla(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 1), "fmla d30, d29, v28.d[1]");
|
||||
|
||||
// TEST_SINGLE(fmls(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmls h30, h29, v15.h[4]");
|
||||
// TEST_SINGLE(fmls(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmls h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmls h30, h29, v15.h[4]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmls h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmls s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "fmls s30, s29, v28.s[3]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmls d30, d29, v28.d[0]");
|
||||
TEST_SINGLE(fmls(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 1), "fmls d30, d29, v28.d[1]");
|
||||
|
||||
// TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmul h30, h29, v15.h[4]");
|
||||
// TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmul h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmul h30, h29, v15.h[4]");
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmul h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmul s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "fmul s30, s29, v28.s[3]");
|
||||
TEST_SINGLE(fmul(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmul d30, d29, v28.d[0]");
|
||||
@@ -694,8 +691,8 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Scalar: Advanced SIMD scalar x inde
|
||||
TEST_SINGLE(sqrdmlsh(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "sqrdmlsh s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(sqrdmlsh(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "sqrdmlsh s30, s29, v28.s[3]");
|
||||
|
||||
// TEST_SINGLE(fmulx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmulx h30, h29, v15.h[4]");
|
||||
// TEST_SINGLE(fmulx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmulx h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 4), "fmulx h30, h29, v15.h[4]");
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i16Bit, VReg::v30, VReg::v29, VReg::v15, 7), "fmulx h30, h29, v15.h[7]");
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmulx s30, s29, v28.s[0]");
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i32Bit, VReg::v30, VReg::v29, VReg::v28, 3), "fmulx s30, s29, v28.s[3]");
|
||||
TEST_SINGLE(fmulx(ScalarRegSize::i64Bit, VReg::v30, VReg::v29, VReg::v28, 0), "fmulx d30, d29, v28.d[0]");
|
||||
|
||||
@@ -96,6 +96,10 @@ inline int32_t renameat2(int olddirfd, const char* oldpath, int newdirfd, const
|
||||
inline int32_t pidfd_open(pid_t pid, unsigned int flags) {
|
||||
return ::syscall(SYS_pidfd_open, pid, flags);
|
||||
}
|
||||
|
||||
inline ssize_t getrandom(void* buf, size_t buflen, unsigned int flags) {
|
||||
return ::syscall(SYS_getrandom, buf, buflen, flags);
|
||||
}
|
||||
#else
|
||||
|
||||
inline int32_t getcpu(uint32_t* cpu, uint32_t* node) {
|
||||
|
||||
@@ -9,7 +9,9 @@ Furthermore, a per-app configuration system allows tweaking performance per game
|
||||
We also provide a user-friendly FEXConfig GUI to explore and change these settings.
|
||||
|
||||
## Prerequisites
|
||||
FEX requires ARMv8.0+ hardware. It has been tested with the following Linux distributions, though others are likely to work as well:
|
||||
FEX requires ARMv8.0-a or newer hardware, with minimum extensions FEAT_FP and FEAT_CRC32.
|
||||
|
||||
It has been tested with the following Linux distributions, though others are likely to work as well:
|
||||
|
||||
- Arch Linux
|
||||
- Fedora Linux
|
||||
|
||||
@@ -60,6 +60,8 @@ class HostFeatures(Flag) :
|
||||
FEATURE_LRCPC2 = (1 << 15)
|
||||
FEATURE_FRINTTS = (1 << 16)
|
||||
FEATURE_MOPS = (1 << 17)
|
||||
FEATURE_I8MM = (1 << 18)
|
||||
FEATURE_DOTPROD = (1 << 19)
|
||||
|
||||
HostFeaturesLookup = {
|
||||
"SVE128" : HostFeatures.FEATURE_SVE128,
|
||||
@@ -80,6 +82,8 @@ HostFeaturesLookup = {
|
||||
"LRCPC2" : HostFeatures.FEATURE_LRCPC2,
|
||||
"FRINTTS" : HostFeatures.FEATURE_FRINTTS,
|
||||
"MOPS" : HostFeatures.FEATURE_MOPS,
|
||||
"I8MM" : HostFeatures.FEATURE_I8MM,
|
||||
"DOTPROD" : HostFeatures.FEATURE_DOTPROD,
|
||||
}
|
||||
|
||||
def GetHostFeatures(data):
|
||||
|
||||
@@ -87,6 +87,7 @@ class HostFeatures(Flag) :
|
||||
FEATURE_CLFLOPT = (1 << 21)
|
||||
FEATURE_FSGSBASE = (1 << 22)
|
||||
FEATURE_EMMI = (1 << 23)
|
||||
FEATURE_AVX_VNNI = (1 << 24)
|
||||
|
||||
RegStringLookup = {
|
||||
"NONE": Regs.REG_NONE,
|
||||
@@ -173,6 +174,7 @@ HostFeaturesLookup = {
|
||||
"CLFLOPT" : HostFeatures.FEATURE_CLFLOPT,
|
||||
"FSGSBASE" : HostFeatures.FEATURE_FSGSBASE,
|
||||
"EMMI" : HostFeatures.FEATURE_EMMI,
|
||||
"AVX_VNNI" : HostFeatures.FEATURE_AVX_VNNI,
|
||||
}
|
||||
|
||||
def parse_hexstring(s):
|
||||
|
||||
@@ -374,6 +374,8 @@ static void OverrideFeatures(FEXCore::HostFeatures* Features, uint64_t ForceSVEW
|
||||
ENABLE_DISABLE_OPTION(Supports3DNow, 3DNOW, 3DNOW);
|
||||
ENABLE_DISABLE_OPTION(SupportsSSE4a, SSE4A, SSE4A);
|
||||
ENABLE_DISABLE_OPTION(SupportsMOPS, MOPS, MOPS);
|
||||
ENABLE_DISABLE_OPTION(SupportsI8MM, I8MM, I8MM);
|
||||
ENABLE_DISABLE_OPTION(SupportsDotProd, DOTPROD, DOTPROD);
|
||||
GET_SINGLE_OPTION(Crypto, CRYPTO);
|
||||
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
@@ -504,6 +506,8 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
HostFeatures.SupportsSVEBitPerm = Features.Supports(CPUFeatures::Feature::SVE_BitPerm);
|
||||
HostFeatures.SupportsECV = Features.Supports(CPUFeatures::Feature::ECV);
|
||||
HostFeatures.SupportsWFXT = Features.Supports(CPUFeatures::Feature::WFxt);
|
||||
HostFeatures.SupportsI8MM = Features.Supports(CPUFeatures::Feature::I8MM);
|
||||
HostFeatures.SupportsDotProd = Features.Supports(CPUFeatures::Feature::DotProd);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode enable SVE with 256-bit wide registers.
|
||||
@@ -537,6 +541,7 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
}
|
||||
#endif
|
||||
|
||||
HostFeatures.Supports3DNow = true;
|
||||
HostFeatures.SupportsAVX = true;
|
||||
HostFeatures.SupportsAES256 = HostFeatures.SupportsAVX && HostFeatures.SupportsAES;
|
||||
HostFeatures.SupportsPreserveAllABI = FEX_HAS_PRESERVE_ALL_ATTR;
|
||||
@@ -553,15 +558,6 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
#ifdef _WIN32
|
||||
// Disable 3DNow! by default to better match the set of extensions exposed on modern CPUs.
|
||||
// This works around a bug that manifests in some games using native d3dx9 DLLs (most easily reproduced in WoW64 builds).
|
||||
// For example, Fallout: New Vegas and some old EA games will run with a blackscreen.
|
||||
HostFeatures.Supports3DNow = false;
|
||||
#else
|
||||
HostFeatures.Supports3DNow = true;
|
||||
#endif
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
@@ -637,6 +633,7 @@ FEXCore::HostFeatures FetchHostFeatures() {
|
||||
|
||||
HostFeatures.SupportsCPUIndexInTPIDRRO = false;
|
||||
HostFeatures.HostType = FEXCore::HostFeatures::HostTypeEnum::Linux;
|
||||
HostFeatures.ProcessPID = ::getpid();
|
||||
return HostFeatures;
|
||||
}
|
||||
} // namespace FEX
|
||||
@@ -40,6 +40,11 @@ public:
|
||||
Feat_vaes = data_7.ecx & (1U << 9);
|
||||
Feat_pclmulqdq = Feat_pclmulqdq && (data_7.ecx & (1U << 10));
|
||||
Feat_rdpid = data_7.ecx & (1U << 22);
|
||||
|
||||
if (data_7.eax >= 1) {
|
||||
auto data_7_1 = cpuid(0x7, 0x1);
|
||||
Feat_avx_vnni = Feat_avx && (data_7_1.eax & (1U << 4));
|
||||
}
|
||||
}
|
||||
|
||||
data = cpuid(0x8000'0000U);
|
||||
@@ -79,6 +84,7 @@ public:
|
||||
bool Feat_rdpid {};
|
||||
bool Feat_clflopt {};
|
||||
bool Feat_fsgsbase {};
|
||||
bool Feat_avx_vnni {};
|
||||
|
||||
private:
|
||||
struct cpuid_data {
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
{
|
||||
"Config": {
|
||||
"DiskCache": "1",
|
||||
"X87ReducedPrecision": "1",
|
||||
"RootFS": "@FEX_ROOTFS_PATH@/",
|
||||
"ThunkHostLibs": "@FEX_COMPAT_TOOL@/usr/lib/aarch64-linux-gnu/fex-emu/HostThunks",
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include "Common/HostFeatures.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/DiskCache.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
@@ -13,10 +14,6 @@
|
||||
|
||||
#include <sys/stat.h>
|
||||
|
||||
namespace FEXCore::DiskCache {
|
||||
uint16_t GetFormatVersion();
|
||||
}
|
||||
|
||||
namespace CodeSize {
|
||||
class CodeSizeValidation final {
|
||||
public:
|
||||
@@ -557,6 +554,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
FEATURE_LRCPC2 = (1U << 15),
|
||||
FEATURE_FRINTTS = (1U << 16),
|
||||
FEATURE_MOPS = (1U << 17),
|
||||
FEATURE_I8MM = (1U << 18),
|
||||
FEATURE_DOTPROD = (1U << 19),
|
||||
};
|
||||
|
||||
uint64_t SVEWidth = 0;
|
||||
@@ -610,6 +609,12 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_MOPS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEMOPS);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_I8MM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEI8MM);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_DOTPROD) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEDOTPROD);
|
||||
}
|
||||
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_TSO) {
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "1");
|
||||
@@ -668,6 +673,12 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_MOPS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEMOPS);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_I8MM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEI8MM);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_DOTPROD) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEDOTPROD);
|
||||
}
|
||||
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_TSO) {
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0");
|
||||
|
||||
@@ -321,6 +321,7 @@ public:
|
||||
FEATURE_CLFLOPT = (1 << 21),
|
||||
FEATURE_FSGSBASE = (1 << 22),
|
||||
FEATURE_EMMI = (1 << 23),
|
||||
FEATURE_AVX_VNNI = (1 << 24),
|
||||
};
|
||||
|
||||
bool Requires3DNow() const {
|
||||
@@ -395,6 +396,9 @@ public:
|
||||
bool RequiresEMMI() const {
|
||||
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_EMMI;
|
||||
}
|
||||
bool RequiresAVXVNNI() const {
|
||||
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_AVX_VNNI;
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ConfigDumpGPRs, DUMPGPRS);
|
||||
@@ -602,6 +606,9 @@ public:
|
||||
bool RequiresEMMI() const {
|
||||
return Config.RequiresEMMI();
|
||||
}
|
||||
bool RequiresAVXVNNI() const {
|
||||
return Config.RequiresAVXVNNI();
|
||||
}
|
||||
|
||||
private:
|
||||
constexpr static uint64_t STACK_OFFSET = 0xc000'0000;
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <elf.h>
|
||||
#include <fcntl.h>
|
||||
@@ -288,6 +290,100 @@ struct ELFParser {
|
||||
return false;
|
||||
}
|
||||
|
||||
struct MappedSection {
|
||||
const void* base {};
|
||||
const void* ptr {};
|
||||
size_t size {};
|
||||
};
|
||||
|
||||
MappedSection MapSection(int fd, uint64_t offset, size_t Size) {
|
||||
// Need to map from [offset, offset+Size).
|
||||
const uint64_t PageAlignedBase = FEXCore::AlignDown(offset, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
const uint64_t OffsetInPage = (offset - PageAlignedBase);
|
||||
const uint64_t TotalSize = OffsetInPage + Size;
|
||||
auto ptr = ::mmap(nullptr, TotalSize, PROT_READ, MAP_PRIVATE, fd, PageAlignedBase);
|
||||
if (ptr == MAP_FAILED) {
|
||||
return {};
|
||||
}
|
||||
|
||||
return MappedSection {
|
||||
.base = ptr,
|
||||
.ptr = reinterpret_cast<const void*>(reinterpret_cast<uintptr_t>(ptr) + OffsetInPage),
|
||||
.size = TotalSize,
|
||||
};
|
||||
}
|
||||
|
||||
void FreeSection(MappedSection& section) {
|
||||
::munmap(const_cast<void*>(section.base), section.size);
|
||||
}
|
||||
|
||||
// Returns an ELF file's build-id if it exists.
|
||||
// Not all ELF files have a build id so it needs to be optional.
|
||||
fextl::vector<uint8_t> GetBuildID() {
|
||||
if (fd == -1 || !EnsureSectionHeadersLoaded()) {
|
||||
return {};
|
||||
}
|
||||
|
||||
const Elf64_Shdr* StrHeader = &shdrs->at(ehdr.e_shstrndx);
|
||||
auto SHStringSection = MapSection(fd, StrHeader->sh_offset, StrHeader->sh_size);
|
||||
if (SHStringSection.base == nullptr) {
|
||||
return {};
|
||||
}
|
||||
|
||||
auto find_name = [&SHStringSection](int offset) -> std::string_view {
|
||||
if (offset >= SHStringSection.size) {
|
||||
return {};
|
||||
}
|
||||
|
||||
return reinterpret_cast<const char*>(SHStringSection.ptr) + offset;
|
||||
};
|
||||
|
||||
fextl::vector<uint8_t> BuildID {};
|
||||
|
||||
for (const auto& shdr : *shdrs) {
|
||||
if (shdr.sh_type != SHT_NOTE || shdr.sh_size == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
auto SectionName = find_name(shdr.sh_name);
|
||||
if (SectionName != ".note.gnu.build-id") {
|
||||
continue;
|
||||
}
|
||||
|
||||
auto BuildIDSection = MapSection(fd, shdr.sh_offset, shdr.sh_size);
|
||||
if (BuildIDSection.base == nullptr) {
|
||||
// Couldn't map
|
||||
break;
|
||||
}
|
||||
|
||||
struct ELFNote {
|
||||
uint32_t NameSize;
|
||||
uint32_t DescSize;
|
||||
uint32_t Type;
|
||||
char Name[];
|
||||
};
|
||||
|
||||
auto Note = reinterpret_cast<const ELFNote*>(BuildIDSection.ptr);
|
||||
const auto DataOffset = (Note->NameSize + 3) & ~3;
|
||||
|
||||
if (Note->Type == NT_GNU_BUILD_ID && Note->NameSize == 4 && std::string_view(Note->Name, Note->NameSize - 1) == "GNU" &&
|
||||
shdr.sh_size <= (12 + DataOffset + Note->DescSize)) {
|
||||
auto Desc = reinterpret_cast<const uint8_t*>(&Note->Name[0] + DataOffset);
|
||||
BuildID.insert(BuildID.end(), Desc, Desc + Note->DescSize);
|
||||
}
|
||||
|
||||
FreeSection(BuildIDSection);
|
||||
|
||||
if (!BuildID.empty()) {
|
||||
// Found the build-id.
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
FreeSection(SHStringSection);
|
||||
return BuildID;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parses relocation sections (SHT_REL/SHT_RELA) and returns a map of
|
||||
* offsets to relocations that FEX's JIT must know about.
|
||||
|
||||
@@ -393,6 +393,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
|
||||
int FEXFD {StealFEXFDFromEnv("FEX_EXECVEFD")};
|
||||
int FEXSeccompFD {StealFEXFDFromEnv("FEX_SECCOMPFD")};
|
||||
const bool FEXFDPathBacked = getenv("FEX_EXECVEFD_PATH") != nullptr;
|
||||
unsetenv("FEX_EXECVEFD_PATH");
|
||||
|
||||
// Early init trivial handlers.
|
||||
LogMan::Throw::InstallHandler(FEX::Logging::AssertHandler);
|
||||
@@ -511,9 +513,14 @@ int main(int argc, char** argv, char** const envp) {
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_FILENAME, Program.ProgramPath);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_CONFIG_NAME, Program.ProgramName);
|
||||
} else if (FEXFD != -1) {
|
||||
// Anonymous program.
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_FILENAME, "<Anonymous>");
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_CONFIG_NAME, "<Anonymous>");
|
||||
if (FEXFDPathBacked) {
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_FILENAME, Program.ProgramPath);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_CONFIG_NAME, Program.ProgramName);
|
||||
} else {
|
||||
// Anonymous program.
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_FILENAME, "<Anonymous>");
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_CONFIG_NAME, "<Anonymous>");
|
||||
}
|
||||
} else {
|
||||
{
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
|
||||
@@ -415,7 +415,7 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
|
||||
const auto NtDll = GetModuleHandleW(L"ntdll.dll");
|
||||
const bool IsWine = !!GetProcAddress(NtDll, "wine_get_version");
|
||||
auto HostFeatures = FEX::Windows::CPUFeatures::FetchHostFeatures(
|
||||
IsWine, Is64Bit ? FEXCore::HostFeatures::HostTypeEnum::Arm64ec : FEXCore::HostFeatures::HostTypeEnum::Wow64);
|
||||
IsWine, Is64Bit ? FEXCore::HostFeatures::HostTypeEnum::Arm64ec : FEXCore::HostFeatures::HostTypeEnum::Wow64, 0);
|
||||
#endif
|
||||
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext(HostFeatures);
|
||||
|
||||
@@ -359,7 +359,12 @@ static fextl::string GenerateCPUInfo(FEXCore::Context::Context* ctx, uint32_t CP
|
||||
add_flag_if(res_d_1.eax & (1 << 3), "xsaves");
|
||||
add_flag_if(res_d_1.eax & (1 << 4), "xfd");
|
||||
|
||||
add_flag_if(res_7_1.eax & (1 << 4), "avx_vnni");
|
||||
add_flag_if(res_7_1.eax & (1 << 5), "avx512_bf16");
|
||||
add_flag_if(res_7_1.eax & (1 << 23), "avx_ifma");
|
||||
add_flag_if(res_7_1.edx & (1 << 4), "avx_vnni_int8");
|
||||
add_flag_if(res_7_1.edx & (1 << 5), "avx_ne_convert");
|
||||
add_flag_if(res_7_1.edx & (1 << 10), "avx_vnni_int16");
|
||||
add_flag_if(res_8000_0008.ebx & (1 << 0), "clzero");
|
||||
add_flag_if(res_8000_0008.ebx & (1 << 1), "irperf");
|
||||
add_flag_if(res_8000_0008.ebx & (1 << 2), "xsaveerptr");
|
||||
|
||||
@@ -316,6 +316,10 @@ uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname
|
||||
// This will stay inside of our emulated environment since binfmt_misc will capture it
|
||||
const bool IsBinfmtCompatible = SyscallHandler->IsInterpreterInstalled() && !NeedsFDCopy &&
|
||||
(Type == ELFLoader::ELFContainer::ELFType::TYPE_X86_32 || Type == ELFLoader::ELFContainer::ELFType::TYPE_X86_64);
|
||||
// Without a visible binfmt interpreter, the next FEX process resolves guest paths through RootFS.
|
||||
// Keep the already resolved host ELF open so a binary outside that RootFS can still be executed.
|
||||
const bool NeedsLoaderFD = !IsFDExec && !SyscallHandler->IsInterpreterInstalled() &&
|
||||
(Type == ELFLoader::ELFContainer::ELFType::TYPE_X86_32 || Type == ELFLoader::ELFContainer::ELFType::TYPE_X86_64);
|
||||
|
||||
// We are trying to execute an ELF of a different architecture
|
||||
// We can't know if we can support this without architecture specific checks and binfmt_misc parsing
|
||||
@@ -328,7 +332,7 @@ uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname
|
||||
// - seccomp inheritance
|
||||
// - FEXServer FD inheritance (unshare(CLONE_NEWNET))
|
||||
// - FD_CLOEXEC set on FD on anonymous file FD.
|
||||
const bool NeedsEnvpCopy = (IsFDExec && !(IsBinfmtCompatible || IsOtherELF)) || HasSeccomp || NeedsFDCopy;
|
||||
const bool NeedsEnvpCopy = (IsFDExec && !(IsBinfmtCompatible || IsOtherELF)) || HasSeccomp || NeedsFDCopy || NeedsLoaderFD;
|
||||
|
||||
// We are trying to execute a shebang handled by a different architecture interpreter (e.g. /usr/bin/python from the host FS).
|
||||
// In this case we just defer to the kernel.
|
||||
@@ -350,6 +354,13 @@ uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname
|
||||
// so duplicate the FD if FD_CLOEXEC is set, which removes the FD_CLOEXEC flag.
|
||||
Args.dirfd = dup(Args.dirfd);
|
||||
FDExecCopy = true;
|
||||
} else if (NeedsLoaderFD) {
|
||||
Args.dirfd = open(Filename.c_str(), O_RDONLY);
|
||||
if (Args.dirfd == -1) {
|
||||
CloseSeccompFD();
|
||||
return -errno;
|
||||
}
|
||||
FDExecCopy = true;
|
||||
}
|
||||
|
||||
// Remove AT_EMPTY_PATH flag now.
|
||||
@@ -363,6 +374,10 @@ uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname
|
||||
|
||||
// Insert the FD for FEX to track.
|
||||
EnvpArgs.emplace_back(FDExecEnv.data());
|
||||
if (NeedsLoaderFD) {
|
||||
// Distinguish this path-backed FD from an anonymous memfd exec.
|
||||
EnvpArgs.emplace_back("FEX_EXECVEFD_PATH=1");
|
||||
}
|
||||
}
|
||||
|
||||
if (HasSeccomp) {
|
||||
@@ -424,8 +439,11 @@ uint64_t ExecveHandler(FEXCore::Core::CpuStateFrame* Frame, const char* pathname
|
||||
|
||||
// It is valid to provide nullptr first argument.
|
||||
if (*OldArgv) {
|
||||
// Skip filename argument
|
||||
++OldArgv;
|
||||
// The direct fallback uses the executable path as argv[0]. An FD-backed load uses the
|
||||
// binfmt preserve-argv0 layout, so the caller-supplied argv[0] has to stay in the vector.
|
||||
if (!NeedsLoaderFD) {
|
||||
++OldArgv;
|
||||
}
|
||||
while (*OldArgv) {
|
||||
// Append the arguments together
|
||||
ExecveArgs.emplace_back(*OldArgv);
|
||||
@@ -887,6 +905,10 @@ void SyscallHandler::HandleSyscall(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
} else {
|
||||
HandleSyscallImpl<false>(Frame, JITPC);
|
||||
}
|
||||
|
||||
// Skip past the `syscall` or `int 0x80` instruction. Both of which are 2-bytes.
|
||||
auto Thread = FEX::HLE::ThreadManager::GetStateObjectFromCPUState(Frame);
|
||||
Thread->Thread->CurrentFrame->State.rip += 2;
|
||||
}
|
||||
|
||||
#ifdef DEBUG_STRACE
|
||||
|
||||
@@ -205,6 +205,7 @@ public:
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(NeedsSeccomp, NEEDSSECCOMP);
|
||||
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
|
||||
FEX_CONFIG_OPT(EnableDiskCache, DISKCACHE);
|
||||
|
||||
uint32_t GetHostKernelVersion() const {
|
||||
return HostKernelVersion;
|
||||
|
||||
@@ -17,6 +17,7 @@ $end_info$
|
||||
#include <sys/mman.h>
|
||||
#include <sys/personality.h>
|
||||
#include <sys/shm.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
#include "LinuxSyscalls/Syscalls.h"
|
||||
#include "LinuxSyscalls/SignalDelegator.h"
|
||||
@@ -231,6 +232,7 @@ FEXCore::HLE::ExecutableRangeInfo SyscallHandler::QueryGuestExecutableRange(FEXC
|
||||
struct ReadELFHeadersResult {
|
||||
fextl::vector<Elf64_Phdr> ProgramHeaders;
|
||||
fextl::robin_map<uint32_t, FEXCore::GuestRelocationType> Relocations;
|
||||
fextl::vector<uint8_t> BuildID;
|
||||
bool HasCodeRelocations;
|
||||
};
|
||||
|
||||
@@ -256,7 +258,8 @@ static ReadELFHeadersResult ReadELFHeaders(int FD, std::span<std::byte> HeaderDa
|
||||
|
||||
auto Relocations = Parser.PopulateRelocations();
|
||||
auto HasCodeRelocations = Parser.HasCodeRelocations();
|
||||
return ReadELFHeadersResult {std::move(Parser.phdrs), std::move(Relocations), HasCodeRelocations};
|
||||
auto buildid = Parser.GetBuildID();
|
||||
return ReadELFHeadersResult {std::move(Parser.phdrs), std::move(Relocations), buildid, HasCodeRelocations};
|
||||
}
|
||||
|
||||
static fextl::unique_ptr<FEXCore::MappedCodeCacheFile>
|
||||
@@ -574,6 +577,35 @@ uint64_t SyscallHandler::GuestShmdt(bool Is64Bit, FEXCore::Core::InternalThreadS
|
||||
return Result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Computes a unique identifier for the referenced binary file to be used for
|
||||
* generating the code map.
|
||||
* This identifier is independent of FEX build/runtime configuration and
|
||||
* stable across FEX updates.
|
||||
*/
|
||||
static uint64_t ComputeCodeMapId(std::string_view Filename, std::optional<fextl::vector<uint8_t>*> BuildID = std::nullopt) {
|
||||
// Use a combination of filename and buildid if it exists.
|
||||
// Not all executables have buildid so this isn't an all encompassing solution.
|
||||
// `Crypt of the Necrodancer` as an example doesn't have a BuildID on Linux.
|
||||
|
||||
uint64_t Hash = ~0ULL;
|
||||
if (!Filename.empty()) {
|
||||
Hash = XXH3_64bits(Filename.data(), Filename.size());
|
||||
}
|
||||
|
||||
if (BuildID) {
|
||||
auto Data = (*BuildID)->data();
|
||||
auto Size = (*BuildID)->size();
|
||||
if (Size == 8) {
|
||||
auto ID64Bit = reinterpret_cast<uint64_t*>(Data);
|
||||
Hash ^= *ID64Bit;
|
||||
} else {
|
||||
Hash ^= XXH3_64bits(Data, Size);
|
||||
}
|
||||
}
|
||||
return Hash;
|
||||
}
|
||||
|
||||
// MMan Tracking
|
||||
std::optional<SyscallHandler::LateApplyExtendedVolatileMetadata>
|
||||
SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t addr, size_t length, int prot, int flags, int fd,
|
||||
@@ -611,11 +643,12 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
if ((prot & PROT_READ) && Inserted) {
|
||||
Resource->MappedFile = fextl::make_unique<VMATracking::ExecutableFileState>();
|
||||
Resource->MappedFile->Filename = fextl::string(Tmp, PathLength);
|
||||
Resource->MappedFile->FileId = CTX->GetCodeCache().ComputeCodeMapId(Resource->MappedFile->Filename, fd);
|
||||
|
||||
// Read ELF headers if applicable and needed for code caching.
|
||||
// For performance, skip ELF checks if we're not mapping the file header
|
||||
bool CheckForElfFile = (offset == 0) && EnableCodeCaching;
|
||||
bool CheckForElfFileDiskCache = (offset == 0) && EnableDiskCache;
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
CheckForElfFile = true;
|
||||
#endif
|
||||
@@ -624,6 +657,7 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
Resource->ProgramHeaders = std::move(ELFResult.ProgramHeaders);
|
||||
Resource->MappedFile->Relocations = std::move(ELFResult.Relocations);
|
||||
Resource->RequiresDelayedCacheLoad = ELFResult.HasCodeRelocations;
|
||||
Resource->MappedFile->FileId = ComputeCodeMapId(Resource->MappedFile->Filename, &ELFResult.BuildID);
|
||||
|
||||
// GuestRelocationType::Skip indicates to FEXOfflineCompiler that
|
||||
// any blocks covered by the relocation may not be cached.
|
||||
@@ -636,8 +670,29 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t MinAddress = 0;
|
||||
uint64_t MaxAddress = 0;
|
||||
for (auto& ProgramHeader : Resource->ProgramHeaders) {
|
||||
if (ProgramHeader.p_type == PT_LOAD) {
|
||||
if (!MinAddress && !MaxAddress) {
|
||||
MinAddress = ProgramHeader.p_vaddr;
|
||||
MaxAddress = ProgramHeader.p_vaddr + ProgramHeader.p_memsz;
|
||||
}
|
||||
MinAddress = std::min(MinAddress, ProgramHeader.p_vaddr);
|
||||
MaxAddress = std::max(MaxAddress, ProgramHeader.p_vaddr + ProgramHeader.p_memsz);
|
||||
}
|
||||
}
|
||||
if (MaxAddress > MinAddress) {
|
||||
Resource->MappedFile->MappedSize = MaxAddress - MinAddress;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Resource->ProgramHeaders.empty() || offset == 0, "Expected file offset 0 for the first mapping of an ELF "
|
||||
"file");
|
||||
} else if (CheckForElfFileDiskCache) {
|
||||
auto ELFResult = ReadELFHeaders(fd, std::span {reinterpret_cast<std::byte*>(addr), length});
|
||||
Resource->MappedFile->FileId = ComputeCodeMapId(Resource->MappedFile->Filename, &ELFResult.BuildID);
|
||||
} else {
|
||||
Resource->MappedFile->FileId = ComputeCodeMapId(Resource->MappedFile->Filename);
|
||||
}
|
||||
} else if (ResourceIt->second.ProgramHeaders.empty()) {
|
||||
// Not an ELF file, so we don't need to distinguish between different base addresses
|
||||
|
||||
@@ -197,7 +197,7 @@ namespace PThreads {
|
||||
|
||||
class PThread final : public FEXCore::Threads::Thread {
|
||||
public:
|
||||
PThread(StackTracker* STracker, FEXCore::Threads::ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags)
|
||||
PThread(StackTracker* STracker, FEXCore::Threads::ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags, const char* ThreadName)
|
||||
: STracker {STracker}
|
||||
, UserFunc {Func}
|
||||
, UserArg {Arg}
|
||||
@@ -226,6 +226,9 @@ namespace PThreads {
|
||||
HLE::ThreadManager::SetSignalMask(OldMask);
|
||||
}
|
||||
pthread_attr_destroy(&Attr);
|
||||
if (ThreadName) {
|
||||
pthread_setname_np(Thread, ThreadName);
|
||||
}
|
||||
}
|
||||
|
||||
bool joinable() override {
|
||||
@@ -375,8 +378,8 @@ namespace PThreads {
|
||||
static StackTracker* STracker {};
|
||||
|
||||
static fextl::unique_ptr<FEXCore::Threads::Thread>
|
||||
CreateThread_PThread(FEXCore::Threads::ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags) {
|
||||
return fextl::make_unique<PThread>(STracker, Func, Arg, Flags);
|
||||
CreateThread_PThread(FEXCore::Threads::ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags, const char* ThreadName) {
|
||||
return fextl::make_unique<PThread>(STracker, Func, Arg, Flags, ThreadName);
|
||||
}
|
||||
|
||||
static void CleanupAfterFork_PThread() {
|
||||
|
||||
@@ -253,7 +253,7 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
});
|
||||
|
||||
// launch a new process under fex
|
||||
// currently does not propagate argv[0] correctly
|
||||
// the ELF self-reexec fallback preserves the caller-supplied argv[0]
|
||||
REGISTER_SYSCALL_IMPL_X32(execve, [](FEXCore::Core::CpuStateFrame* Frame, const char* pathname, uint32_t* argv, uint32_t* envp) -> uint64_t {
|
||||
fextl::vector<const char*> Args;
|
||||
fextl::vector<const char*> Envp;
|
||||
|
||||
@@ -285,7 +285,7 @@ void RegisterThread(FEX::HLE::SyscallHandler* Handler) {
|
||||
});
|
||||
|
||||
// launch a new process under fex
|
||||
// currently does not propagate argv[0] correctly
|
||||
// the ELF self-reexec fallback preserves the caller-supplied argv[0]
|
||||
REGISTER_SYSCALL_IMPL_X64(execve, [](FEXCore::Core::CpuStateFrame* Frame, const char* pathname, char* const argv[], char* const envp[]) -> uint64_t {
|
||||
fextl::vector<const char*> Args;
|
||||
fextl::vector<const char*> Envp;
|
||||
|
||||
@@ -287,6 +287,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
const bool SupportsRDPID = Feature.Feat_rdpid;
|
||||
const bool SupportsCLFLOPT = Feature.Feat_clflopt;
|
||||
const bool SupportsFSGSBase = Feature.Feat_fsgsbase;
|
||||
const bool SupportsAVXVNNI = Feature.Feat_avx_vnni;
|
||||
|
||||
TestUnsupported |=
|
||||
(!Supports3DNow && Loader.Requires3DNow()) || (!SupportsSSE4A && Loader.RequiresSSE4A()) || (!SupportsBMI1 && Loader.RequiresBMI1()) ||
|
||||
@@ -295,6 +296,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
(!SupportsAES && Loader.RequiresAES()) || (!SupportsPCLMUL && Loader.RequiresPCLMUL()) || (!SupportsMOVBE && Loader.RequiresMOVBE()) ||
|
||||
(!SupportsADX && Loader.RequiresADX()) || (!SupportsXSAVE && Loader.RequiresXSAVE()) || (!SupportsRDPID && Loader.RequiresRDPID()) ||
|
||||
(!SupportsCLFLOPT && Loader.RequiresCLFLOPT()) || (!SupportsFSGSBase && Loader.RequiresFSGSBase()) || Loader.RequiresEMMI();
|
||||
TestUnsupported |= !SupportsAVXVNNI && Loader.RequiresAVXVNNI();
|
||||
#endif
|
||||
|
||||
#ifdef _WIN32
|
||||
|
||||
@@ -101,9 +101,24 @@ ExitFunctionSuspendPoint:
|
||||
brk #0xCAFE
|
||||
// Resume will jump back to `ExitFunctionSuspendResumePoint`
|
||||
|
||||
// This table will contain all of the possible SVC #x instructions that we might need.
|
||||
// We index into this table in DIRECT_SYSCALL_WRAPPER based on the syscall number we find
|
||||
// by parsing the table __arm64ec_syscall_ffs from NTDLL.
|
||||
#define SYSCALL_TABLE_ENTRIES 512
|
||||
.global SyscallTable
|
||||
SyscallTable:
|
||||
.set SyscallId, 0
|
||||
.rept SYSCALL_TABLE_ENTRIES
|
||||
svc #SyscallId
|
||||
ret
|
||||
.set SyscallId, SyscallId + 1
|
||||
.endr
|
||||
.global SyscallTableEnd
|
||||
SyscallTableEnd:
|
||||
|
||||
// Makes a wrapper for calling a system call directly, skipping the usual ntdll thunks
|
||||
#define HASH #
|
||||
#define DIRECT_SYSCALL_WRAPPER(Name, WineIdName, WindowsId) \
|
||||
#define DIRECT_SYSCALL_WRAPPER(Name, WineIdName) \
|
||||
.global Name; \
|
||||
Name:; \
|
||||
adrp x16, WineSyscallDispatcher; \
|
||||
@@ -115,19 +130,23 @@ ExitFunctionSuspendPoint:
|
||||
blr x16; \
|
||||
ret; \
|
||||
1:; \
|
||||
svc HASH WindowsId; \
|
||||
ret
|
||||
adrp x16, SyscallTable; \
|
||||
add x16, x16, HASH:lo12:SyscallTable; \
|
||||
adrp x17, WineIdName; \
|
||||
ldr x17, [x17, HASH:lo12:WineIdName]; \
|
||||
add x16, x16, x17, lsl HASH 3; \
|
||||
br x16
|
||||
|
||||
// Allows for continuing from a full native context, as the NTDLL NtContinue export takes in an x64 context with EC and
|
||||
// the conversion to that loses the ARM64EC ABI-disallowed registers that FEX uses.
|
||||
DIRECT_SYSCALL_WRAPPER("#NtContinueNative", WineNtContinueSyscallId, 0x43)
|
||||
DIRECT_SYSCALL_WRAPPER("#NtRaiseExceptionNative", WineNtRaiseExceptionSyscallId, 0x174)
|
||||
DIRECT_SYSCALL_WRAPPER("#NtContinueNative", WineNtContinueSyscallId)
|
||||
DIRECT_SYSCALL_WRAPPER("#NtRaiseExceptionNative", WineNtRaiseExceptionSyscallId)
|
||||
|
||||
// Both of these are wrapped as FEX needs them to setup its call checker at startup time and their NTDLL thunks could
|
||||
// already be patched by then (and because the call checker isn't installed, their patched x86 versions would be invoked
|
||||
// when called by FEX).
|
||||
DIRECT_SYSCALL_WRAPPER("#NtAllocateVirtualMemoryNative", WineNtAllocateVirtualMemorySyscallId, 0x18)
|
||||
DIRECT_SYSCALL_WRAPPER("#NtProtectVirtualMemoryNative", WineNtProtectVirtualMemorySyscallId, 0x50)
|
||||
DIRECT_SYSCALL_WRAPPER("#NtAllocateVirtualMemoryNative", WineNtAllocateVirtualMemorySyscallId)
|
||||
DIRECT_SYSCALL_WRAPPER("#NtProtectVirtualMemoryNative", WineNtProtectVirtualMemorySyscallId)
|
||||
|
||||
// A replacement for the standard ARM64EC call checker that ignores any FFS patches and always redirects to a function's
|
||||
// native implementation. As the only library FEX calls into is NTDLL, this is done using a LUT generated at init time.
|
||||
|
||||
@@ -74,6 +74,8 @@ extern void* ExitFunctionEC;
|
||||
extern void* CheckCall;
|
||||
extern void* ExitFunctionSuspendPoint;
|
||||
extern void* ExitFunctionSuspendResumePoint;
|
||||
extern uint64_t SyscallTable[];
|
||||
extern uint64_t SyscallTableEnd[];
|
||||
|
||||
void* X64ReturnInstr; // See Module.S
|
||||
uintptr_t NtDllBase;
|
||||
@@ -82,7 +84,7 @@ uintptr_t NtDllBase;
|
||||
uint32_t* NtDllRedirectionLUT;
|
||||
uint32_t NtDllRedirectionLUTSize;
|
||||
|
||||
// Wine doesn't support issuing direct system calls with SVC, and unlike Windows it doesn't have a 'stable' syscall number for NtContinue
|
||||
// Wine doesn't support issuing direct system calls with SVC, and like Windows it also doesn't have 'stable' syscall numbers
|
||||
void* WineSyscallDispatcher;
|
||||
uint64_t WineNtContinueSyscallId;
|
||||
uint64_t WineNtAllocateVirtualMemorySyscallId;
|
||||
@@ -268,11 +270,51 @@ void ParseWineSyscallNumbers(HMODULE NtDll) {
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t GetNTSyscallNo(HMODULE NtDll, const char* ExportName) {
|
||||
ULONG LoadConfigSize;
|
||||
const _IMAGE_LOAD_CONFIG_DIRECTORY64* NtDllLoadConfig = reinterpret_cast<_IMAGE_LOAD_CONFIG_DIRECTORY64*>(
|
||||
RtlImageDirectoryEntryToData(NtDll, true, IMAGE_DIRECTORY_ENTRY_LOAD_CONFIG, &LoadConfigSize));
|
||||
|
||||
const IMAGE_ARM64EC_METADATA* NtDllARM64ECMetadata = reinterpret_cast<IMAGE_ARM64EC_METADATA*>(NtDllLoadConfig->CHPEMetadataPointer);
|
||||
|
||||
// This redirection table stores a mapping from x64 entry points for fast forward sequences,
|
||||
// to the real ARM64EC implementation of the function.
|
||||
const IMAGE_ARM64EC_REDIRECTION_ENTRY* RedirectionTableBegin =
|
||||
reinterpret_cast<IMAGE_ARM64EC_REDIRECTION_ENTRY*>(NtDllBase + NtDllARM64ECMetadata->RedirectionMetadata);
|
||||
const IMAGE_ARM64EC_REDIRECTION_ENTRY* RedirectionTableEnd = RedirectionTableBegin + NtDllARM64ECMetadata->RedirectionMetadataCount;
|
||||
|
||||
// Walk through the table until we find an entry matching the x64 FFS.
|
||||
const uintptr_t x64FFS = reinterpret_cast<uintptr_t>(GetProcAddress(NtDll, ExportName)) - NtDllBase;
|
||||
uintptr_t ARM64ECImplementation = 0;
|
||||
for (const IMAGE_ARM64EC_REDIRECTION_ENTRY* RedirectionEntry = RedirectionTableBegin; RedirectionEntry != RedirectionTableEnd;
|
||||
RedirectionEntry++) {
|
||||
if (RedirectionEntry->Source == x64FFS) {
|
||||
ARM64ECImplementation = NtDllBase + RedirectionEntry->Destination;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Now walk through the syscall FFS table and find which entry points to the ARM64EC implementation
|
||||
// we found in the previous table. The index of this entry in that table will be the syscall number.
|
||||
const uintptr_t* SyscallImplementationTable = reinterpret_cast<uintptr_t*>(GetProcAddress(NtDll, "__arm64ec_syscall_ffs"));
|
||||
const uint32_t* SyscallImplementationCount = reinterpret_cast<uint32_t*>(GetProcAddress(NtDll, "__arm64ec_syscall_ffs_size"));
|
||||
if (!ARM64ECImplementation || !SyscallImplementationTable || !SyscallImplementationCount) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
for (uint64_t SyscallNo = 0; SyscallNo < *SyscallImplementationCount; SyscallNo++) {
|
||||
if (SyscallImplementationTable[SyscallNo] == ARM64ECImplementation) {
|
||||
return SyscallNo;
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
// Syscall thunks may have been patched before FEX has loaded, the default call checker installed by ntdll into FEX will
|
||||
// try to invoke the JIT when calling such patched syscalls but this obviously doesn't work before FEX is initalised.
|
||||
// This function parses ntdll and sets up a custom call checker to prevent this, as such it must avoid using any syscall
|
||||
// thunks itself.
|
||||
void InitSyscalls() {
|
||||
bool InitSyscalls() {
|
||||
// The ntdll exports called by GetModuleHandle/GetProcAddress aren't known to be patched before JIT init by any current
|
||||
// software so are safe to call, but if that changes the loader structures in the PEB could be parsed manually.
|
||||
const auto NtDll = GetModuleHandleW(L"ntdll.dll");
|
||||
@@ -282,10 +324,27 @@ void InitSyscalls() {
|
||||
if (WineSyscallDispatcherPtr) {
|
||||
WineSyscallDispatcher = *WineSyscallDispatcherPtr;
|
||||
ParseWineSyscallNumbers(NtDll);
|
||||
} else {
|
||||
// NT syscall numbers may change between versions, so we need to find the numbers
|
||||
// for the syscalls we need ahead of time.
|
||||
WineNtContinueSyscallId = GetNTSyscallNo(NtDll, "NtContinue");
|
||||
WineNtAllocateVirtualMemorySyscallId = GetNTSyscallNo(NtDll, "NtAllocateVirtualMemory");
|
||||
WineNtProtectVirtualMemorySyscallId = GetNTSyscallNo(NtDll, "NtProtectVirtualMemory");
|
||||
WineNtRaiseExceptionSyscallId = GetNTSyscallNo(NtDll, "NtRaiseException");
|
||||
|
||||
// Fail if the syscall number we found is beyond the bounds
|
||||
// of our static table. This is almost certainly not going
|
||||
// to happen, but failing here will make debugging easier later on.
|
||||
const uint64_t SyscallTableSize = SyscallTableEnd - SyscallTable;
|
||||
if (std::max({WineNtContinueSyscallId, WineNtAllocateVirtualMemorySyscallId, WineNtProtectVirtualMemorySyscallId,
|
||||
WineNtRaiseExceptionSyscallId}) >= SyscallTableSize) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
FillNtDllLUTs(NtDll);
|
||||
PatchCallChecker();
|
||||
return true;
|
||||
}
|
||||
|
||||
void HandleImageMap(uint64_t Address, bool MainImage = false) {
|
||||
@@ -580,7 +639,9 @@ extern "C" void SyncThreadContext(CONTEXT* Context) {
|
||||
}
|
||||
|
||||
NTSTATUS ProcessInit() {
|
||||
InitSyscalls();
|
||||
if (!InitSyscalls()) {
|
||||
return STATUS_NOT_SUPPORTED;
|
||||
}
|
||||
|
||||
FEX::Windows::InitCRTProcess();
|
||||
FEX::Windows::SetupThreadHandlers();
|
||||
@@ -612,7 +673,8 @@ NTSTATUS ProcessInit() {
|
||||
}
|
||||
|
||||
{
|
||||
auto HostFeatures = FEX::Windows::CPUFeatures::FetchHostFeatures(IsWine, FEXCore::HostFeatures::HostTypeEnum::Arm64ec);
|
||||
auto HostFeatures =
|
||||
FEX::Windows::CPUFeatures::FetchHostFeatures(IsWine, FEXCore::HostFeatures::HostTypeEnum::Arm64ec, FEX::Windows::UnixLib::GetPID());
|
||||
CTX = FEXCore::Context::Context::CreateNewContext(HostFeatures);
|
||||
}
|
||||
|
||||
@@ -1017,7 +1079,7 @@ NTSTATUS ThreadTerm(HANDLE Thread, LONG ExitCode) {
|
||||
return STATUS_ACCESS_DENIED;
|
||||
}
|
||||
|
||||
auto ThreadDup = FEX::Windows::DupHandle(Thread, THREAD_QUERY_INFORMATION | THREAD_SUSPEND_RESUME);
|
||||
auto ThreadDup = FEX::Windows::DupHandle(Thread, THREAD_QUERY_INFORMATION | THREAD_SUSPEND_RESUME | THREAD_GET_CONTEXT);
|
||||
|
||||
THREAD_BASIC_INFORMATION Info;
|
||||
if (auto Err = NtQueryInformationThread(*ThreadDup, ThreadBasicInformation, &Info, sizeof(Info), nullptr); Err) {
|
||||
@@ -1031,6 +1093,7 @@ NTSTATUS ThreadTerm(HANDLE Thread, LONG ExitCode) {
|
||||
// If we are suspending a thread that isn't ourselves, try to suspend it first so we know internal JIT locks aren't being held.
|
||||
NtSuspendThread(*ThreadDup, NULL);
|
||||
// This will wait for the thread to be suspended
|
||||
TmpContext.ContextFlags = CONTEXT_CONTROL;
|
||||
NtGetContextThread(*ThreadDup, &TmpContext);
|
||||
}
|
||||
|
||||
|
||||
@@ -56,7 +56,7 @@ public:
|
||||
}
|
||||
};
|
||||
|
||||
FEXCore::HostFeatures CPUFeatures::FetchHostFeatures(bool IsWine, FEXCore::HostFeatures::HostTypeEnum HostType) {
|
||||
FEXCore::HostFeatures CPUFeatures::FetchHostFeatures(bool IsWine, FEXCore::HostFeatures::HostTypeEnum HostType, uint32_t ProcessPID) {
|
||||
HKEY Key = OpenProcessorKey(0);
|
||||
if (!Key) {
|
||||
ERROR_AND_DIE_FMT("Couldn't detect CPU features");
|
||||
@@ -84,6 +84,7 @@ FEXCore::HostFeatures CPUFeatures::FetchHostFeatures(bool IsWine, FEXCore::HostF
|
||||
HostFeatures.SupportsCPUIndexInTPIDRRO = !IsWine;
|
||||
|
||||
HostFeatures.HostType = HostType;
|
||||
HostFeatures.ProcessPID = ProcessPID;
|
||||
|
||||
if (HostType == FEXCore::HostFeatures::HostTypeEnum::Wow64) {
|
||||
// AVX is unsupported for WOW64
|
||||
|
||||
@@ -16,7 +16,7 @@ class Context;
|
||||
namespace FEX::Windows {
|
||||
class CPUFeatures {
|
||||
public:
|
||||
static FEXCore::HostFeatures FetchHostFeatures(bool IsWine, FEXCore::HostFeatures::HostTypeEnum HostType);
|
||||
static FEXCore::HostFeatures FetchHostFeatures(bool IsWine, FEXCore::HostFeatures::HostTypeEnum HostType, uint32_t ProcessPID);
|
||||
|
||||
CPUFeatures(FEXCore::Context::Context& CTX);
|
||||
|
||||
|
||||
@@ -200,4 +200,14 @@ void* MapFile(HANDLE FileHandle, uint64_t MapSize) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
uint32_t GetPID() {
|
||||
if (Available()) {
|
||||
FEXUnixLib_GetPID Args {};
|
||||
Call(FEXUnixLibFunctions::GetPID, &Args);
|
||||
return Args.Result;
|
||||
}
|
||||
|
||||
return GetCurrentProcessId();
|
||||
}
|
||||
|
||||
} // namespace FEX::Windows::UnixLib
|
||||
@@ -48,4 +48,9 @@ SHMSlotResult AllocateSHMSlots(void* SHMBase, uint32_t MapSize, uint32_t MaxSize
|
||||
void DeleteSHMStatsFile();
|
||||
|
||||
void* MapFile(HANDLE FileHandle, uint64_t MapSize);
|
||||
|
||||
// Returns the process pid when unixlib is available.
|
||||
// Or returns the process ID otherwise.
|
||||
uint32_t GetPID();
|
||||
|
||||
} // namespace FEX::Windows::UnixLib
|
||||
@@ -32,6 +32,12 @@ static void AssertHandler(const char* Message) {
|
||||
}
|
||||
} // namespace
|
||||
|
||||
|
||||
namespace FEX::Windows::Logging {
|
||||
void UnimplementedLog(const char* Func) {
|
||||
LogMan::Msg::DFmt("Unimplemented Function in: {}", Func);
|
||||
}
|
||||
} // namespace FEX::Windows::Logging
|
||||
namespace FEX::Windows::Logging {
|
||||
void Init() {
|
||||
FEX_CONFIG_OPT(SilentLog, SILENTLOG);
|
||||
|
||||
@@ -55,11 +55,15 @@ public:
|
||||
}
|
||||
};
|
||||
|
||||
namespace FEX::Windows::Logging {
|
||||
void UnimplementedLog(const char* Func);
|
||||
}
|
||||
|
||||
#define UNIMPLEMENTED() \
|
||||
do { \
|
||||
NtTerminateProcess(NtCurrentProcess(), 0); \
|
||||
__fastfail(0); \
|
||||
#define UNIMPLEMENTED() \
|
||||
do { \
|
||||
FEX::Windows::Logging::UnimplementedLog(__func__); \
|
||||
NtTerminateProcess(NtCurrentProcess(), 0); \
|
||||
__fastfail(0); \
|
||||
} while (0)
|
||||
|
||||
#define DLLEXPORT_FUNC(Ret, Name, Args) \
|
||||
|
||||
@@ -13,7 +13,7 @@ namespace FEX::Windows {
|
||||
namespace WinThreadImpl {
|
||||
class Thread final : public FEXCore::Threads::Thread {
|
||||
public:
|
||||
Thread(FEXCore::Threads::ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags)
|
||||
Thread(FEXCore::Threads::ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags, const char* ThreadName)
|
||||
: UserFunc {Func}
|
||||
, UserArg {Arg}
|
||||
, Flags {Flags} {
|
||||
@@ -28,6 +28,17 @@ namespace WinThreadImpl {
|
||||
LogMan::Msg::EFmt("NtCreateThreadEx failed: 0x{:x}", static_cast<uint32_t>(Status));
|
||||
Handle = nullptr;
|
||||
}
|
||||
|
||||
if (ThreadName) {
|
||||
UNICODE_STRING ThreadNameW;
|
||||
if (RtlCreateUnicodeStringFromAsciiz(&ThreadNameW, ThreadName)) {
|
||||
THREAD_NAME_INFORMATION info {
|
||||
.ThreadName = ThreadNameW,
|
||||
};
|
||||
NtSetInformationThread(Handle, static_cast<THREADINFOCLASS>(38) /* ThreadNameInformation */, &info, sizeof(info));
|
||||
RtlFreeUnicodeString(&ThreadNameW);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool joinable() override {
|
||||
@@ -90,8 +101,9 @@ namespace WinThreadImpl {
|
||||
void* ReturnValue {};
|
||||
};
|
||||
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> CreateThread(FEXCore::Threads::ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags) {
|
||||
return fextl::make_unique<Thread>(Func, Arg, Flags);
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread>
|
||||
CreateThread(FEXCore::Threads::ThreadFunc Func, void* Arg, FEXCore::Threads::Flags Flags, const char* ThreadName) {
|
||||
return fextl::make_unique<Thread>(Func, Arg, Flags, ThreadName);
|
||||
}
|
||||
|
||||
void CleanupAfterFork() {}
|
||||
|
||||
Loaded 100 of 231 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user