Compare commits

...
331 Commits
Author SHA1 Message Date
Ryan Houdek e869aa644a Docs: Update for release FEX-2608 2026-08-04 16:55:43 -07:00
Ryan Houdek 68740b3c65 Merge pull request #5792 from Sonicadvance1/192
CI: Disable ranges-v3 from trying to build native
2026-08-03 15:30:52 -07:00
Ryan Houdek 4caad9bf55 Merge pull request #5800 from Sonicadvance1/195
#5795 but with clang_format
2026-08-03 15:30:14 -07:00
Iaying 7629323548 Fix an XMM register bug in SpillSRA, along with adding a test that reproduces the bug 2026-08-03 15:15:41 -07:00
Ryan Houdek 2fdbff3d1c Merge pull request #5790 from mstorsjo/libc++23
Fix building for Windows with libc++ 23
2026-07-30 14:29:13 -07:00
Ryan Houdek 5c7df98768 CI: Disable ranges-v3 from trying to build native
We don't want this.
2026-07-30 13:59:44 -07:00
Martin Storsjö efe38f2ee2 Add more function stubs for libc++ 23 on Windows
These are needed by libc++ when targeting Windows since
https://github.com/llvm/llvm-project/commit/8a531c3608c722ad529be448d6ecef06ba107228,
which is included in libc++ 23.

In a very brief test, it seems like we don't need to actually
implement them.
2026-07-28 23:37:53 +03:00
Martin Storsjö 08031a2767 Add missing includes
This fixes compilation with libc++ 23, which has removed a number
of unnecessary transitive includes in its headers.

Include <cstdlib> in StringConv.h for std::strtoll and std::strtoull.

Include <cstdlib> for the declarations of malloc/free/realloc/calloc
in Alloc.cpp. (Without this, the functions we define end up with
C++ name mangling.)

Include <stdarg.h> in IO.cpp for va_start/va_end.
2026-07-28 23:17:43 +03:00
Ryan Houdek d295d9f08e Merge pull request #5789 from lioncash/sig
SignalDelegator: Remove unused Required parameter in handler setting
2026-07-27 13:14:14 -07:00
Ryan Houdek 61d033f21e Merge pull request #5788 from lioncash/config
Config: Minor cleanup
2026-07-27 13:09:21 -07:00
Ryan Houdek 27315e33ab Merge pull request #5784 from FrontMage/fix/instruction-fetch-fault-priority
Frontend: Prioritize instruction fetch faults
2026-07-27 12:54:53 -07:00
LC e8b5cd18bd SignalDelegator: Remove unused Required parameter in handler setting
The required flag is set by the subsequent frontend functions that
follow the calls to these functions.
2026-07-27 15:06:24 -04:00
LC 1a77f41846 Config: Add missing override specifier 2026-07-27 14:01:11 -04:00
LC d1947f715c Config: Move strings in constructor where applicable
Same thing, just a little less memory churn.
2026-07-27 14:00:06 -04:00
LC 3f7971b7d2 Config: Mark internally linked where applicable
Makes it obvious these aren't supposed to be exposed.
2026-07-27 13:56:53 -04:00
LC 2cbd5cb3e6 Config: Amend prototypes where applicable
Previously, these didn't match up with the implementation (luckily it's
only used internally at the moment).
2026-07-27 13:54:25 -04:00
FrontMage 151b4d4c2d Frontend: Prioritize instruction fetch faults 2026-07-25 09:01:21 +08:00
Ryan Houdek 7a3fdefafb Merge pull request #5786 from OFFTKP/fist
Extend FIST tests to check for indefinite value
2026-07-24 09:27:22 -07:00
Paris Oplopoios 86c20d0519 Extend FIST tests to check for indefinite value 2026-07-24 17:04:10 +03:00
Ryan Houdek 464ec9d0bc Merge pull request #5783 from FrontMage/fix/inactive-jit-guard-range
FEXCore: Ignore inactive JIT guard ranges
2026-07-23 17:50:51 -07:00
FrontMage 35518a5fa0 FEXCore: Ignore inactive JIT guard ranges 2026-07-24 07:57:48 +08:00
Ryan Houdek d028c7942b Merge pull request #5782 from lioncash/validation
IRValidation: Minor cleanups
2026-07-23 15:44:48 -07:00
LC 585286a617 IRValidation: Remove unused members from BlockInfo
HasExit is assigned to but never used, but we check this condition a
different way right after leaving the main loop anyway.
2026-07-24 16:00:20 -04:00
LC 4904fd43e9 IRValidation: Make BlockInfo private
This isn't used outside the context of the pass.
2026-07-24 16:00:20 -04:00
LC 9e3f287c1f IRValidation: Use C instead of CW
This op isn't mutated anywhere in the pass.
2026-07-24 16:00:20 -04:00
LC f5e4e26e08 IRValidation: Move var closer to usage
Same behavior, just more compact.
2026-07-24 16:00:20 -04:00
LC 6c5a39e164 IRValidation: Turn ORs with true into assignment
These are just unconditional setting to true anyway.
2026-07-24 16:00:17 -04:00
Ryan Houdek 7469fdb0d6 Merge pull request #5781 from FrontMage/fix/multiblock-block-local-errors
FEXCore: Isolate multiblock error state per block
2026-07-23 15:11:13 -07:00
FrontMage fcf9fd77d7 FEXCore: Isolate multiblock error state per block 2026-07-23 16:52:19 +08:00
Ryan Houdek 0589d9b872 Merge pull request #5779 from lioncash/x87
x87StackOptimizationPass: Minor cleanup
2026-07-22 18:58:17 -07:00
LC 6d4c80adff x87StackOptimizationPass: Remove IR member
This is only used in the store helpers, so we can just pass it in
directly
2026-07-23 02:26:46 -04:00
LC 56dd470528 x87StackOptimizationPass: Remove unnecesary return in Run()
It's a void function, so we don't need this at the end
2026-07-23 02:21:26 -04:00
LC 6741f53d87 x87StackOptimizationPass: Mark getValidMask()/getInvalidMask() as const
These don't modify instance state.
2026-07-23 02:18:33 -04:00
LC 19550c5417 x87StackOptimizationPass: Pass by const reference in setTop()
Avoids redundant copies. Just a minor codegen saving.
2026-07-23 02:17:02 -04:00
Ryan Houdek cfa3dfaac7 Merge pull request #5780 from mrpippy/unicode
Windows: Fixes around using Unicode functions
2026-07-22 18:48:52 -07:00
Brendan Shanks 0ed0bc1dd5 CMake: Define UNICODE when building for Windows 2026-07-22 15:17:17 -07:00
Brendan Shanks 074743d6e8 Windows: Explicitly use *A/*W Win32 functions 2026-07-22 15:16:44 -07:00
Brendan Shanks 8b446de059 Windows: Use GetModuleHandleW() to avoid unnecessary string conversions 2026-07-22 15:16:44 -07:00
Ryan Houdek f2e35f336f Merge pull request #5778 from lioncash/buf
SharedCodeBufferManager: Minor header tidying
2026-07-21 09:18:25 -07:00
LC 10a0fe2e71 SharedCodeBufferManager: Make AllocateNew() signature consistent with declaration 2026-07-23 01:42:41 -04:00
LC c9add0d292 SharedCodeBufferManager: Hoist prctl define into util header
Same behavior, but just moves the potential define to be alongside all
of the others in the wrapper header.
2026-07-23 01:40:49 -04:00
LC c239d09ea0 SharedCodeBufferManager: Add missing header
Ensures the page size define is always visible.
2026-07-23 00:16:56 -04:00
LC a652a5811b Merge pull request #5777 from Sonicadvance1/191
FEXCore: Split out CodeBuffer management to its own file
2026-07-21 07:31:13 -04:00
Ryan Houdek d2c92808f5 FEXCore: Split out CodeBuffer management to its own file
NFC

- Renames CodeBufferManager to SharedCodeBufferManager to be more
  explicit about it being shared between threads
- Renames `CodeBuffers` to `SharedCodeBuffers` to make it more explicit
  about sharing these buffers between threads.
- Separates the Manager to its own file so it is distinct from the rest
  of the CPUBackend code

Makes it easier to parse ownership and lifetime semantics of these
buffers.
2026-07-20 18:09:29 -07:00
LC 2464633431 Merge pull request #5776 from Sonicadvance1/190
JIT: Remove JIT detection string
2026-07-20 21:07:23 -04:00
LC 15e76e88b4 Merge pull request #5775 from Sonicadvance1/189
JIT: Rename temporary CPU buffer allocator
2026-07-20 20:50:56 -04:00
Ryan Houdek fe1ac1bc1d JIT: Remove JIT detection string
Now that we have VMA region naming enabled on JIT buffers, this is no
longer used. Confirming a region is a JIT buffer is now just a case of
comparing the name that shows up in `/procfs/maps` rather than dumping
the first bytes of an unknown region.
2026-07-20 17:44:15 -07:00
Ryan Houdek 9edd27b214 JIT: Rename temporary CPU buffer allocator
`TempAllocator` was a bit too opaque as to what the allocator was for,
so I kept needing to lookup its usage every couple of months. Rename it
to `TempCodeBufferAllocator` so I can remember that it is a temporary
allocator for the staging JIT code buffer more easily.

NFC
2026-07-20 17:32:03 -07:00
Ryan Houdek eb7e02ea1d Merge pull request #5772 from lioncash/pass
PassManager: Simplify initialization interface
2026-07-19 16:28:07 -07:00
LC 53befc68c9 PassManager: Ensure GetPass() only queries the underlying pass mappings
Previously this would create an entry in the map if it didn't exist.
2026-07-21 12:10:26 -04:00
LC 19f95d89ec PassManager: Add basic documentation 2026-07-21 12:10:26 -04:00
LC ecb9b3b7f8 PassManager: Constrain GetPass() template to Pass-derived objects
Makes the particular conversion types constrained to catch any trivial
misuses.
2026-07-21 12:10:26 -04:00
LC d619e36523 PassManager: Pass string by const reference where applicable
Gets rid of potential extraneous copies. We also add handling for cases
where two passes with the same name are unintentionally added.
Previously we'd blindly overwrite the mapping.
2026-07-21 12:09:16 -04:00
LC fa80d11960 PassManager: Remove SyscallHandler member
This isn't used anymore, so we can get rid of it to further simplify
initialization.
2026-07-21 11:28:12 -04:00
LC 2893d2b64f PassManager: Simplify pass initialization
We don't conditionally add any passes, so we can simplify the interface
so that we just add all existing passes at once. Makes the core
initialization process a little more straightforward.
2026-07-21 11:28:09 -04:00
Ryan Houdek 99b8df4e6f Merge pull request #5773 from lioncash/fdres
ThreadManager: Fix error return values in FrontendAllocateSlots()
2026-07-19 16:14:50 -07:00
LC 331655182d ThreadManager: Fix error return values in FrontendAllocateSlots()
If ftruncate or the mmap ever fail for whatever reason, then we need to
return the current size, rather than the new size.
2026-07-21 15:12:41 -04:00
Ryan Houdek ec95330dcd Merge pull request #5765 from lioncash/signal
SignalDelegator: Group members together
2026-07-19 16:13:26 -07:00
LC 3959863462 SignalDelegator: Group members together
Hides all public members and situates all of them together to make for
an easier overview.
2026-07-19 21:10:35 -04:00
LC c00b2c9584 Merge pull request #5774 from Sonicadvance1/93
gitlab: Fixes CI
2026-07-19 18:43:37 -04:00
Ryan Houdek 6c354f3987 gitlab: Fixes CI 2026-07-19 12:23:46 -07:00
Ryan Houdek 3bd4d244a4 Merge pull request #5771 from lioncash/fmt
Externals: Update fmt to 12.2.0
2026-07-19 11:50:58 -07:00
LC 9888de25fe Externals: Update fmt to 12.2.0
Keeps fmt up to date.
2026-07-21 09:32:23 -04:00
Ryan Houdek 58c247b30c Merge pull request #5770 from lioncash/cast
CPUBackend: Remove unnecessary reinterpret_casts
2026-07-19 00:57:29 -07:00
LC 22bd10f3b1 CPUBackend: Remove unnecessary reinterpret_casts
This both take a void*, so the casting is unnecessary to begin with,
since this would occur anyway without it. We can also avoid a
duplication to reduce line noise.
2026-07-21 08:22:17 -04:00
Ryan Houdek f374b4775a Merge pull request #5769 from lioncash/bound
Core: Remove unnecessary bounds check in GenerateIR()
2026-07-18 21:35:34 -07:00
LC ff213bbc5e Core: Move vars closer to usage scope in GenerateIR()
Makes it so their purpose is more easily seen
2026-07-21 04:44:47 -04:00
LC 56a4ca6e6a Core: Remove unnecessary bounds check in GenerateIR()
We already check the bounds in the loop prior to calling at().
2026-07-21 04:40:22 -04:00
Ryan Houdek d6b38b6b1c Merge pull request #5768 from lioncash/stream
IRDumper: stringstream -> ostringstream
2026-07-18 21:33:42 -07:00
LC 04d06d386f IRDumper: stringstream -> ostringstream
These are purely output operations, so we don't need to use the more
heavyweight class.
2026-07-21 04:29:03 -04:00
LC aa26a780ed Merge pull request #5767 from Sonicadvance1/188
AVX128: Optimize 256-bit vmovmaskpd as well
2026-07-17 16:56:23 -04:00
Ryan Houdek f73b93dbc2 InstcountCI: Update 2026-07-17 13:17:39 -07:00
Ryan Houdek c4a5ac892f AVX128: Optimize 256-bit vmovmaskpd as well
Similar to #5757, but once the elements have been zipped together, we
can treat it identically to the 128-bit 32-bit element path.

Closes #3782
2026-07-17 13:15:47 -07:00
Ryan Houdek f129ca0c61 unittests/vmovmskpd: Extend test to have different lower and upper results between 128-bit lanes. 2026-07-17 13:12:27 -07:00
Ryan Houdek 941f0fbf8d Merge pull request #5764 from lioncash/alloc
LinuxAllocator: Reduce MemAllocator32Bit size by 16 bytes
2026-07-17 08:00:25 -07:00
LC 6e37dca566 LinuxAllocator: Reduce MemAllocator32Bit size by 16 bytes
These constants don't need to be member vars.
2026-07-19 18:37:38 -04:00
Ryan Houdek 1cffa009fe Merge pull request #5763 from lioncash/const
IREmitter: Mark some helpers as const
2026-07-17 07:59:47 -07:00
Tony Wasserka 83f4c9d101 Merge pull request #5760 from neobrain/fix_emitter_constants
Arm64Emitter: Fix incorrect condition for constant NOP padding
2026-07-17 13:05:50 +02:00
Tony Wasserka c0c95da796 Arm64Emitter: Fix incorrect condition for constant NOP padding
This needs to be enabled when *generating* caches, not at runtime when we're
loading them (unless we're compiling for validation).

Previous code would incorrectly disable NOP padding in FEXOfflineCompiler and
instead enable it at runtime when it wasn't needed.
2026-07-17 12:47:51 +02:00
LC 8b612a87c6 IREmitter: Mark some helpers as const
These don't modify internal state.
2026-07-17 03:20:36 -04:00
Ryan Houdek 2ac95f446b Merge pull request #5762 from lioncash/core
FEXCore: Resolve missing prototype warnings
2026-07-17 00:04:47 -07:00
LC c5eddd922d FEXCore: Resolve missing prototype warnings
Makes sure we mark everything internally linked as necessary, or make
declarations visible to their implementation.
2026-07-17 02:48:45 -04:00
Ryan Houdek b58be2c073 Merge pull request #5761 from lioncash/sys
LinuxEmulation: Resolve missing prototype warnings
2026-07-16 22:00:10 -07:00
LC a3d0b67777 LinuxEmulation: Resolve missing prototype warnings
Ensures all functions are marked whether they're intended to be
internally linked or not.
2026-07-17 00:12:28 -04:00
Ryan Houdek 0467d523c0 Merge pull request #5759 from neobrain/fix_codebuffer_max_size
CodeCache: Use maximal code buffer size when generating code caches, too
2026-07-16 13:53:42 -07:00
Ryan Houdek b0af054a95 Merge pull request #5757 from MoonFlowww/avx128-vmovmsk-256
AVX_128: Optimize VMOVMSKPS from 11 to 7 instructions
2026-07-16 13:53:00 -07:00
Ryan Houdek 6846f10510 Merge pull request #5758 from lioncash/x87
x87StackOptimizationPass: Make use of std::array for FixedSizeStack
2026-07-16 12:39:08 -07:00
Ryan Houdek e24f232f52 Merge pull request #5756 from lioncash/fill
Arm64Emitter: Pull FillSpecialRegs bools into a struct
2026-07-16 12:36:31 -07:00
Ryan Houdek a1cc5d034e Merge pull request #5755 from lioncash/host
HostRunner: Tidy up interface
2026-07-16 12:35:18 -07:00
Ryan Houdek b478f54aea Merge pull request #5754 from lioncash/vdso
VDSO_Emulation: Mark relevant members as internally linked
2026-07-16 12:34:27 -07:00
Ryan Houdek 7bb380a086 Merge pull request #5753 from lioncash/pipe
FEXServer: Fix some missing declaration warnings
2026-07-16 12:33:54 -07:00
Tony Wasserka 228c351396 CodeCache: Use maximal code buffer size when generating code caches, too
This is less likely to happen, but will still be required for very large libraries.
2026-07-16 16:37:18 +02:00
LC 4bb675a530 x87StackOptimizationPass: Reduce noise in slow push/pop paths
Deduplicates the repeated rotate behavior.
2026-07-16 10:31:40 -04:00
LC 8d62773570 x87StackOptimizationPass: Make helpers internally linked
Makes it obvious they're only used in this TU and allows the compiler to
warn if they ever become unused.
2026-07-16 09:56:14 -04:00
LC 9e26c57643 x87StackOptimizationPass: Fix isValid()
Previously this wouldn't have worked, since .first isn't a valid member.
The only reason it wasn't caught is because the function is never
instantiated.
2026-07-16 09:56:14 -04:00
LC 35bc502062 x87StackOptimizationPass: Make use of std::array for FixedSizeStack
Reduces the overall generated code for state management.

Drops the overall text size from 11447294 to 11441918
2026-07-16 09:56:05 -04:00
moonfloww 7189e1e280 InstcountCI: Update 2026-07-16 14:44:04 +02:00
LC 4b4aa1cdbe Arm64Emitter: Pull FillSpecialRegs bools into a struct
Makes this easily expandable over time without modifying the prototype,
and lets us be a little more informative at call sites.
2026-07-16 08:07:57 -04:00
moonfloww 3680282b30 new vmovmsk from 11 to 7 ins. 2026-07-16 13:57:12 +02:00
LC a4d5fbae63 HostRunner: Tidy up interface
We've accumulated a bunch of forward declarations that are no longer
necessary. We also don't need to pass the signal delegator as a
reference, since we're not modifying the pointer itself, it's just
passed in to register a signal handler.
2026-07-16 07:39:03 -04:00
LC 2943b87c83 VDSO_Emulation: Mark relevant members as internally linked
Silences missing prototype warnings and makes it obvious they're only
used within the translation unit.
2026-07-16 07:28:50 -04:00
LC 9575506391 FEXServer: Fix some missing declaration warnings
Makes sure prototypes are visible to their implementation. Also marks
functions internally linked where applicable.

Also makes it a little more visibly obvious which bits are exposed for
use elsewhere.
2026-07-16 07:10:32 -04:00
Ryan Houdek 31c2449d6e Merge pull request #5752 from lioncash/config
FEXGetConfig: Add convenience option for dumping system/tso info
2026-07-15 09:48:49 -07:00
LC 70eafc4f6f FEXGetConfig: Mark helpers as static where applicable
Makes them internally linked, and also lets them be caught by the
compiler when they're unused.
2026-07-15 12:22:12 -04:00
LC 38cbd2aeb2 FEXGetConfig: Add convenience option for dumping system/tso info
Just lets you get a broad overview all at once instead of needing to
type out every long command.

Now it's easier to be lazy and just pass "-e", or "--all-emu-info".
2026-07-15 12:22:10 -04:00
LC ba762a9326 Merge pull request #5747 from Sonicadvance1/187
HostFeatures: Pull MMFR3 identification register
2026-07-15 12:20:52 -04:00
Ryan Houdek 921ce59054 Merge pull request #5750 from lioncash/pred
VectorOps: Make use of unpredicated shifts
2026-07-15 09:15:04 -07:00
Ryan Houdek a7627ba39a Merge pull request #5751 from lioncash/sq
VectorOps: Add trivial case handling in VSQXTN2
2026-07-15 09:06:59 -07:00
Ryan Houdek 34ef28dea5 Merge pull request #5749 from lioncash/calc
RedundantFlagCalculationElimination: Minor tidying
2026-07-15 09:05:45 -07:00
Ryan Houdek 0f2463dda2 Merge pull request #5748 from lioncash/invariant
RegisterAllocationPass: Ensure pair reg invariant
2026-07-15 09:04:38 -07:00
Ryan Houdek 5ec852637c Merge pull request #5745 from lioncash/addv
VectorOps: Simplify 256-bit VAddV
2026-07-15 09:03:57 -07:00
LC ba8b0afe7a VectorOps: Use unpredicated shifts where applicable for 256-bit scalar shifts
Lets us trim some output
2026-07-15 09:10:57 -04:00
LC 98d45a6a9b VectorOps: Make use of unpredicated immediate shifts
Same behavior, just without introducing a predicate register dependency.
2026-07-15 08:00:58 -04:00
LC 6bc808cb06 VectorOps: Add trivial case handling in VSQXTN2
Lets us generate much more optimal code in the event the destination and
lower source are the same.
2026-07-15 07:49:50 -04:00
LC 1325fef703 RFCE: Prefer accessing ops with C instead of CW
CW is only intended when the op needs to be writable, but most of these
are only reading data.
2026-07-15 07:06:23 -04:00
LC d2d0ef1803 RFCE: Remove unnecessary std::invoke()
We can just call this normally (and also make the constituent helper
function internally linked).
2026-07-15 07:03:09 -04:00
LC d6fb60d512 RegisterAllocationPass: Ensure pair reg invariant
Allows us to actually catch if this requirement ever gets broken in
the future.
2026-07-15 06:46:24 -04:00
LC ef35474f88 VectorOps: Simplify 256-bit VAddV
Didn't read the manual close enough on the first read award.
2026-07-15 04:43:47 -04:00
Ryan Houdek 372891361c HostFeatures: Pull MMFR3 identification register
This has the S1POE flag that we will want to use in the future.
2026-07-14 20:09:31 -07:00
Ryan Houdek 50c75d1f43 Move Linux version calculation to common code 2026-07-14 20:06:37 -07:00
Ryan Houdek 30f2a7b23b Merge pull request #5746 from lioncash/shadow
x87StackOptimizationPass: Remove shadowing variable in PUSHSTACK case
2026-07-14 13:32:13 -07:00
LC f2212a497b x87StackOptimizationPass: Remove shadowing variable in PUSHSTACK case
No behavioral change, since the one in the outer scope does the same thing.
2026-07-14 08:01:30 -04:00
Ryan Houdek 12e8cf008a Merge pull request #5744 from lioncash/telem
AtomicOps: Avoid constrained unpredictable case in TelemetrySetValue()
2026-07-13 16:37:51 -07:00
LC 9e8e87bbb2 AtomicOps: Avoid constrained unpredictable case in TelemetrySetValue()
STLXR cannot use the same register as both the status register and the
value register, otherwise it's architecturally unpredictable
behavior.

Only applies to hardware without FEAT_LSE, so this only meaningfully
affects hardware using the v8.0 spec, since FEAT_LSE becomes mandatory
in v8.1 and newer.
2026-07-13 19:08:17 -04:00
Ryan Houdek 76c4ebb36f Merge pull request #5743 from lioncash/str
StringUtils: Handle strings entirely composed of whitespace in trims
2026-07-13 15:40:50 -07:00
Ryan Houdek 9ff322eeed Merge pull request #5742 from lioncash/sema
x32/Semaphore: Fix storing of message type in msgrcv
2026-07-13 15:40:07 -07:00
LC 4afa49824e StringUtils: Handle strings entirely composed of whitespace in trims
Previously this wouldn't handle fully whitespaced strings.
2026-07-13 17:37:09 -04:00
LC 23402bf31b x32/Semaphore: Fix storing of message type in msgrcv
This was previously storing into the local compat handler, not the
actual managed message.
2026-07-13 17:30:27 -04:00
LC 50be718b72 x32/Semaphore: Mark _ipc as static
This isn't used outside of the translation unit.
2026-07-13 17:30:24 -04:00
Ryan Houdek 903e7db427 Merge pull request #5741 from lioncash/file 2026-07-13 14:13:13 -07:00
LC 5e5e9e0803 Utils/File: Fix handle releasing
ShouldClose was never being set in the event we opened a regular file.
The only time it was set (to false) is when it's used to encapsulate
stderr and stdout.

So anything opened by a File instance was essentially held open.
2026-07-13 16:01:28 -04:00
Ryan Houdek ce27754b9d Merge pull request #5740 from lioncash/ra
RegisterAllocationPass: Function cleanup
2026-07-13 12:42:45 -07:00
LC 4254c0f5a9 RegisterAllocationPass: Function cleanup
Marks a few functions const or static to clarify usage a little more.
2026-07-13 15:26:51 -04:00
Ryan Houdek 7efc3ecaba Merge pull request #5739 from lioncash/zero
Vector: Indicate 128-bit zero vector in DefaultX87State()
2026-07-13 11:59:22 -07:00
LC 2934b01d58 Vector: Indicate 128-bit zero vector in DefaultX87State()
Same functional behavior, just makes it visually match the store size
below. Technically also avoids delegating off to the 64-bit element
path if a 128-bit constant zero is already loaded.
2026-07-13 14:07:12 -04:00
LC 192e363701 Merge pull request #5738 from Sonicadvance1/186
64BitAllocator: Removes unused additional size argument
2026-07-13 13:44:11 -04:00
Ryan Houdek 1287365616 64BitAllocator: Removes unused additional size argument
This used to be used for the intrusively allocated `LiveVMARegion` but
that is all handled internally to the object now, making this
unnecessary. It was always receiving zero and doing nothing so just
remove it.
2026-07-13 10:18:46 -07:00
Ryan Houdek 4fa539fbb2 Merge pull request #5733 from lioncash/alloc
64BitAllocator: Avoid madvising more than necessary in InitializeVMARegionsUsed()
2026-07-13 10:16:57 -07:00
Ryan Houdek 28cdae4687 Merge pull request #5737 from lioncash/bsl
VectorOps: Simplify SVE 256-bit VOrn with BSL2N
2026-07-13 10:05:08 -07:00
LC 842e22915c VectorOps: Simplify SVE 256-bit VOrn with BSL2N
Lets us shave off an instruction and also avoid using a temporary
register in some cases. We can also tweak our worst case that requires a
predicate to eliminate the temporary as well.

We can also expand our cmpps cases, so that we can reflect the
BSL2N usages in instcountci.
2026-07-13 12:27:54 -04:00
Ryan Houdek bd150233ce Merge pull request #5736 from lioncash/ushrni
VectorOps: Make SVE shift==0 case symmetric with ASIMD
2026-07-13 07:56:25 -07:00
LC 24720b67da VectorOps: Make SVE shift==0 case symmetric with ASIMD
Ensures that we have consistent behavior.
2026-07-13 10:26:42 -04:00
Ryan Houdek 356d461123 Merge pull request #5734 from lioncash/bytes
Common/BitSet: Amend byte size retrieval
2026-07-13 07:07:00 -07:00
Ryan Houdek 3cbcc7b9f8 Merge pull request #5735 from lioncash/ir
IR: Enclose straggler Desc comments in brackets
2026-07-13 07:06:06 -07:00
LC 329f12a888 json_ir_generator: Join successive write calls together for allocator helpers
We can just write these out as cohesive units. Also makes adding to them
less annoying.
2026-07-13 09:22:55 -04:00
LC c0b2eec5de IR: Enclose straggler Desc comments in brackets
Ensures the comments get rendered properly in output. We can also
make sure that the IR generation script catches this in the future.
2026-07-13 08:59:45 -04:00
LC 9b8ae25491 Common/BitSet: Amend byte size retrieval
This needs to divide by 8 to get a proper byte size for all type sizes.
The only usage of this is currently a uint64_t, so it worked by
coincidence, since sizeof(uint64_t) == 8.
2026-07-13 08:17:27 -04:00
LC c0ee865e3e 64BitAllocator: Avoid madvising more than necessary in InitializeVMARegionsUsed
Because our bitset type is uint64_t, then that means Memory + ManagedSize
is more like: Memory + (ManagedSize * 8), which is way larger of a base
than we need.
2026-07-12 20:38:19 -04:00
Ryan Houdek 46ec2797ff Merge pull request #5732 from lioncash/vec
Crypto: Clarify zero vector size in SHA1RNDS4Op()
2026-07-12 16:14:26 -07:00
LC fb2cdc8541 Crypto: Clarify zero vector size in SHA1RNDS4Op()
This ends up zeroing out the whole 128-bit vector.
2026-07-12 18:49:27 -04:00
Ryan Houdek 850ef70496 Merge pull request #5731 from lioncash/xar
Crypto: Make use of XAR in SHA1NEXTE when available
2026-07-12 14:26:14 -07:00
LC 9d3c388664 Crypto: Make use of XAR in SHA1NEXTE when available
Lets us shave an instruction off on hardware that supports XAR.

Closes #5730
2026-07-12 16:02:12 -04:00
Ryan Houdek f2b679f602 Merge pull request #5728 from lioncash/halves
x32/FD: Combine offset halves directly
2026-07-12 11:15:10 -07:00
Ryan Houdek b9d97dffe7 Merge pull request #5727 from lioncash/vmsplice
x32/FD: Make use of SanitizeIOCount for vector construction in vmsplice
2026-07-12 11:14:31 -07:00
Ryan Houdek 7ff0466c2c Merge pull request #5726 from lioncash/file
Utils/File: Handle dual read/write case
2026-07-12 11:09:00 -07:00
Ryan Houdek a135325185 Merge pull request #5725 from lioncash/dead
Signals: Preprocessor disable intentional dead code
2026-07-12 11:06:30 -07:00
LC a6c8f0d300 x32/FD: Combine offset halves directly
Shortens these up a little.
2026-07-12 13:31:48 -04:00
LC 046750354e x32/FD: Make use of SanitizeIOCount for vector construction in vmsplice
Makes this consistent with the other fd syscalls that make temporary
buffers.
2026-07-12 13:01:58 -04:00
LC e23d703873 Utils/File: Handle dual read/write case
According to POSIX open docs, this is a completely separate flag that
isn't a combination of O_RDONLY and O_WRONLY, so we need to handle this
separately.

Makes the codepath behaviorally symmetric with the Windows one.
2026-07-12 12:41:02 -04:00
LC ca2d2520d2 Signals: Preprocessor disable intentional dead code in userfaultfd
Noticed this when going through the syscalls. Avoids potential warnings.
2026-07-12 12:16:07 -04:00
Ryan Houdek bc16f902d1 Merge pull request #5724 from lioncash/file
WinAPI/IO: Fix handling of end of file offset in SetFilePointerEx
2026-07-11 22:44:55 -07:00
LC f5ae888597 WinAPI/IO: Fix handling of end of file offset in SetFilePointerEx
This just means the end of the file is being used as the base offset.

Also note that according to the documentation for SetFilePositionEx,
that setting the position beyond the current file size is not considered
an error as far as the API is concerned.
2026-07-12 01:26:23 -04:00
Ryan Houdek 376e3af058 Merge pull request #5723 from lioncash/gdb 2026-07-11 21:54:57 -07:00
Ryan Houdek af63c0a9e1 Merge pull request #5722 from lioncash/container 2026-07-11 21:54:20 -07:00
LC 7d149ebec4 GdbServer: Add missing log format argument 2026-07-12 00:28:29 -04:00
LC 37c809471b ElfContainer: Amend entry iteration in GetDynamicLibs()
These were using i in the termination condition, which is for section
headers, not entries.
2026-07-12 00:19:28 -04:00
Ryan Houdek 12baceb859 Merge pull request #5721 from lioncash/win
AllocatorHooks: Amend VirtualProtect for Windows
2026-07-11 20:49:16 -07:00
Ryan Houdek 3e6b3c8c40 Merge pull request #5720 from lioncash/hdr
64BitAllocator: Remove duplicate headers
2026-07-11 20:33:09 -07:00
LC 25f2021711 AllocatorHooks: Amend VirtualProtect for Windows
VirtualProtect returns non-zero on success, also the old protection flag
parameter isn't allowed to be null.
2026-07-11 23:24:30 -04:00
LC 8b30f7dbb5 64BitAllocator: Remove duplicate headers
These are already included.
2026-07-11 23:08:29 -04:00
Ryan Houdek 76b35dfeb0 Merge pull request #5719 from lioncash/small
64BitAllocator: Avoid overwriting Region[0] in Create64BitAllocatorWithRegions
2026-07-11 19:30:08 -07:00
Ryan Houdek 4518d5831b Merge pull request #5718 from lioncash/absolute
Filesystem: Fix Absolute() on Windows
2026-07-11 17:11:18 -07:00
Ryan Houdek 066250851f Merge pull request #5717 from lioncash/reg
RegisterAllocationPass: Amend type cast in DecodeSRANode()
2026-07-11 17:10:27 -07:00
Ryan Houdek 341a5196ba Merge pull request #5701 from lioncash/thread
Thread: Amend new thread handling in HandleNewClone()
2026-07-11 17:09:59 -07:00
LC c878a89e95 Filesystem: Fix Absolute() on Windows
sizeof(*Fill) will only ever be 1, so we wouldn't actually copy much of
anything.
2026-07-11 17:25:26 -04:00
LC c72d59a3f1 64BitAllocator: Avoid overwriting Region[0] in Create64BitAllocatorWithRegions
Since this was a reference, this would end up overwriting Region[0] with
whatever the smallest region was instead of just being a running pointer
to what happened to be the current smallest region.

We can switch over to a pointer to avoid obliterating the first memory
region.
2026-07-11 16:52:01 -04:00
LC 995e2657bb RegisterAllocationPass: Amend type cast in DecodeSRANode()
Uses the proper type for StoreRegister. Same behavior though, due to
layout.
2026-07-11 16:39:35 -04:00
Ryan Houdek fe6d6397d6 Merge pull request #5716 from lioncash/vec 2026-07-11 13:29:47 -07:00
LC c7f52bcee4 Vector: Centralize masking in InsertScalarFCMPOp
Ensures that even if someone threw bogus constants in the upper bits of
the immediate, that the special-cased comparison types would still be
handled properly.

We can move the masking in the AVX variant too, just to be consistent.
2026-07-11 16:07:19 -04:00
Ryan Houdek fb4d8a6d14 Merge pull request #5715 from lioncash/singlestep
Dispatcher: Avoid loading unnecessary reg in vixl single step
2026-07-11 12:25:13 -07:00
LC 929f9a7ad2 Dispatcher: Avoid loading unnecessary reg in vixl single step
CompileSingleStep only takes one uint64_t, not two.
2026-07-11 15:03:29 -04:00
Ryan Houdek 48ce5bf6e6 Merge pull request #5714 from lioncash/readahead
x32/FD: Fix readahead upper offset type
2026-07-11 11:20:49 -07:00
Ryan Houdek 617a518714 Merge pull request #5713 from lioncash/select
x32/FD: Correct total word calculation in select() variants
2026-07-11 11:19:29 -07:00
Ryan Houdek 480f45f2f3 Merge pull request #5712 from lioncash/bpf
BPFEmitter: Amend instruction class checking in HandleStore()
2026-07-11 11:08:58 -07:00
Ryan Houdek 0e8b01c1b5 Merge pull request #5711 from lioncash/fault
x64/Thread: Amend faulting copy handling related to LDTs
2026-07-11 11:04:47 -07:00
Ryan Houdek b750d6772f Merge pull request #5710 from lioncash/close
Common/Async: Handle fd closing a little better
2026-07-11 11:03:10 -07:00
Ryan Houdek 7ce172a497 Merge pull request #5709 from lioncash/size
FlexBitSet: Simplify MemClear/MemSet
2026-07-11 10:57:58 -07:00
Ryan Houdek 821dfe5b98 Merge pull request #5708 from lioncash/mrs
MiscOps: Fix round mode clearing for RP/RM modes in PushRoundingMode
2026-07-11 10:56:27 -07:00
Ryan Houdek 6c57b2f8f9 Merge pull request #5707 from lioncash/offset
MemoryOps: Avoid double application of base offset in {Load,Store}ContextIndexed case
2026-07-11 10:49:27 -07:00
Ryan Houdek 6e0c9d159d Merge pull request #5706 from lioncash/odd
SignalDelegator: Remove odd double negation in GuestSigProcMask
2026-07-11 10:48:42 -07:00
LC a90a7dbec7 x32/FD: Fix readahead upper offset type
This should be a uint32_t
2026-07-11 13:33:11 -04:00
LC 2250bd58a2 x32/FD: Deduplicate guest and host fd set management
Same behavior, but less copy pastey
2026-07-11 13:23:19 -04:00
LC 126be45ed1 x32/FD: Correct total word calculation in select() variants
Previously this would result in a larger amount of words specified than
necessary.

e.g. Given nfds = 1:

With AlignUp(1, 32) / 4, we'd end up with supposedly eight words, when it
should only be one word.

On the other extreme, given a full fd set of 1024 fds, then we'd end up
with 256 words, when it should only be 32.
2026-07-11 12:53:08 -04:00
LC e021a55abd BPFEmitter: Amend instruction class checking in HandleStore()
This was previously checking for a load class, which would result in ST
clobbering the index register.
2026-07-11 12:26:11 -04:00
LC 2ad6254894 x64/Thread: Amend faulting copy handling related to LDTs
CopyToUser doesn't return the number of bytes copied, but rather returns
0 to indicate success, otherwise a fault has occurred (and the SIGSEGV
handler has set X0 to EFAULT)
2026-07-11 11:57:38 -04:00
LC 5bd97c2f63 Common/Async: Handle fd closing a little better
We should be checking against -1, rather than just anything non-zero.
2026-07-11 10:49:08 -04:00
LC 570d1f2271 FlexBitSet: Simplify MemClear/MemSet
We can just make use of the helpers already in the interface.
2026-07-11 10:33:47 -04:00
LC 613e9ef701 MiscOps: Fix round mode clearing for RP/RM modes in PushRoundingMode
Previously this had the potential to not clear rounding bits properly
depending on incoming FPCR state.
2026-07-11 10:24:07 -04:00
LC 5080c6ffc5 MemoryOps: Avoid double application of base offset in {Load,Store}ContextIndexed unaligned case 2026-07-11 09:56:56 -04:00
LC f16bc12b7d SignalDelegator: Remove odd double negation in GuestSigProcMask
We can just reduce it to normal null comparisons.
2026-07-11 05:42:48 -04:00
Ryan Houdek d2f096187c Merge pull request #5705 from lioncash/thread3 2026-07-10 22:34:04 -07:00
Ryan Houdek bd1e61befd Merge pull request #5704 from lioncash/host 2026-07-10 22:33:30 -07:00
Ryan Houdek b97165bec9 Merge pull request #5703 from lioncash/size 2026-07-10 22:32:45 -07:00
LC 8b0f07b2dc Syscalls: Remove unnecessary usages of namespace FEXCore::IR
These aren't necessary.
2026-07-10 21:02:12 -04:00
LC c0ba45f6de HostFeatures: Shrink feature setting in FillFeatureFlags
Allows us to unify most of the flag setting, so the flag name only needs
to be stated once, reducing likelihood of typos.
2026-07-10 20:34:46 -04:00
LC 017c898ed0 x64/Signals: Amend set size in rt_sigtimedwait
We should be checking the size passed in, not the sizeof of it.
2026-07-10 19:40:05 -04:00
Ryan Houdek b193a0c9fd Merge pull request #5702 from lioncash/sbss
HostFeatures: Amend SSBS2 signifying
2026-07-10 16:38:07 -07:00
LC db0c7b0562 HostFeatures: Amend SSBS2 signifying 2026-07-10 19:13:29 -04:00
LC 85b8e91b57 Thread: Amend new thread handling in HandleNewClone()
Ensures that newly cloned threads get tracked properly.
2026-07-10 18:35:58 -04:00
Ryan Houdek 87301ca154 Merge pull request #5700 from lioncash/bitset
Common/Bitset: Minor API changes
2026-07-10 10:35:45 -07:00
Ryan Houdek 650f5b784d Merge pull request #5699 from lioncash/spillops
Arm64Emitter: Wire up conditional FPR spilling in SpillForPreserveAllABICall
2026-07-10 10:35:19 -07:00
Ryan Houdek c8dd9eefa6 Merge pull request #5698 from lioncash/dead
ConversionOps: Remove redundant code in Vector_FToS
2026-07-10 10:35:06 -07:00
Ryan Houdek 0bdf15977f Merge pull request #5697 from lioncash/thread
x32/Thread: Move writability check around in waitpid
2026-07-10 10:34:56 -07:00
Ryan Houdek 8a844f9cca Merge pull request #5695 from lioncash/socket
Socket: Pass size by reference in getsockopt
2026-07-10 10:34:33 -07:00
Ryan Houdek 8fbde84380 Merge pull request #5696 from lioncash/time
x32/Time: Correct sizeof expression in utimensat
2026-07-10 10:31:47 -07:00
Ryan Houdek 00826d3327 Merge pull request #5694 from lioncash/rlimit
x32/Info: Only modify output in getrlimit/ugetrlimit if successful
2026-07-10 10:27:51 -07:00
Ryan Houdek d5572e322f Merge pull request #5693 from lioncash/ir
IR: Use begin block type in operator--
2026-07-10 10:27:13 -07:00
Ryan Houdek 37814111de Merge pull request #5692 from lioncash/unary
Vector: Use unary handler for scalar unary insertions
2026-07-10 10:26:44 -07:00
Ryan Houdek 1417888a89 Merge pull request #5691 from lioncash/fpr
Vector: LoadSourceGPR -> LoadSourceFPR for MASKMOVOp
2026-07-10 10:25:26 -07:00
Ryan Houdek 5d846c3ca9 Merge pull request #5690 from lioncash/cache
MemoryOps: Make use of current working reg for cache operations
2026-07-10 10:23:55 -07:00
Ryan Houdek 5dd0477440 Merge pull request #5689 from lioncash/msg
x32/Msg: Fix result comparison in mq_getsetattr
2026-07-10 10:15:15 -07:00
Ryan Houdek c6f823f855 Merge pull request #5688 from lioncash/pidfd
Thread: Amend pidfd_open check
2026-07-10 10:14:49 -07:00
Ryan Houdek 2af7c24e79 Merge pull request #5687 from lioncash/limit
x32/Thread: Fix off-by-one in get_thread_area
2026-07-10 10:14:29 -07:00
Ryan Houdek 94ccfefa84 Merge pull request #5686 from lioncash/fd
x32/FD: Ensure sendfile updates offset if set
2026-07-10 10:14:18 -07:00
LC 46f3bec37e Common/BitSet: Ensure internal pointer is always initialized
Provides deterministic state.
2026-07-10 10:31:45 -04:00
LC f5d2e0db29 Common/BitSet: Mark getters as const
These don't modify internal state.
2026-07-10 10:31:05 -04:00
LC 88ee56f471 Common/BitSet: Amend Clear() behavior
Ensures the bits are actually being unset.
2026-07-10 10:26:07 -04:00
LC 8436154276 Arm64Emitter: Wire up conditional FPR spilling in SpillForPreserveAllABICall
Technically, this parameter wasn't being used at all. It was wired up
for filling, but not spilling.
2026-07-10 10:10:25 -04:00
LC 150b25f29e ConversionOps: Remove redundant code in Vector_FToS
These are already defined in an outer scope.
2026-07-10 09:16:35 -04:00
LC f4c50105ba x32/Thread: Move writability check around in waitpid
Same behavior, but catches the write before it actually occurs.
2026-07-10 08:08:45 -04:00
LC 3395bedc2b x32/Time: Correct sizeof expression in utimensat
Ensures we check the proper type.
2026-07-10 08:04:56 -04:00
LC 7a14210e2b Socket: Pass size by reference in getsockopt
Previously this was passing by value.
2026-07-10 08:01:29 -04:00
LC d206b67ca8 x32/Info: Only modify output in getrlimit/ugetrlimit if successful
Avoids trampling over input data.
2026-07-10 07:49:44 -04:00
LC 0bfef9008b IR: Use begin block type in operator--
Same behavior, just more correct from a descriptive PoV
2026-07-10 07:31:25 -04:00
LC 401e542fcd Vector: Use unary handler for scalar unary insertions
Same behavior, but just uses a more proper handler.
2026-07-10 07:27:54 -04:00
LC 210514f74b Vector: LoadSourceGPR -> LoadSourceFPR for MASKMOVOp 2026-07-10 07:24:39 -04:00
LC dd838d4ad3 MemoryOps: Make use of current working reg for cache operations
TMP1 technically isn't initialized properly here until after the first
iteration.
2026-07-10 07:13:17 -04:00
Ryan Houdek 5f2d19c7aa Merge pull request #5685 from lioncash/faddv 2026-07-10 03:33:43 -07:00
LC d5ae87b5ca Thread: Amend pidfd_open check
Checks for success.
2026-07-10 06:25:43 -04:00
Ryan Houdek 95e7c866cf Merge pull request #5682 from lioncash/pid 2026-07-10 03:20:23 -07:00
LC 02028eb1ad x32/Msg: Fix result comparison in mq_getsetattr
Checks against failure instead of 1.
2026-07-10 06:19:11 -04:00
Ryan Houdek 2effdb04bd Merge pull request #5684 from lioncash/midr 2026-07-10 03:18:25 -07:00
Ryan Houdek ab31e3e3bb Merge pull request #5683 from lioncash/mul 2026-07-10 03:18:09 -07:00
LC 37b1432514 x32/Thread: Fix off-by-one in get_thread_area
12, 13, and 14 are the only valid TLS areas.
2026-07-10 06:06:30 -04:00
LC 6e8bc337aa x32/FD: Ensure sendfile updates offset if set 2026-07-10 05:58:49 -04:00
Ryan Houdek 8347566815 Merge pull request #5681 from lioncash/timer 2026-07-10 02:51:20 -07:00
LC ede09a03db VectorOps: Fix 256-bit FADDV path
Avoids falling down to the SVE-128 path.
2026-07-10 05:50:22 -04:00
LC 849d60253c CPUID: Fix MIDR walking in SetupHostHybridFlag() 2026-07-10 05:46:16 -04:00
LC b5660c8a92 MiscOps: Avoid stack misalignment in ProcessorID
This needs to be an add.
2026-07-10 05:41:54 -04:00
LC 54263bb5a7 ALUOps: Fix 32-bit MulH case
These need to be 64-bit multiply and ubfx. Thankfully this case wasn't
actually hit in practice.
2026-07-10 05:39:14 -04:00
LC 4069a9f7d5 Timer: Fix typo in timer_gettime
This should be passed by reference rather than by value.
2026-07-10 05:33:06 -04:00
Ryan Houdek 77467eaf0c Merge pull request #5680 from lioncash/usrai
IR: Remove unused VUShraI
2026-07-10 01:50:51 -07:00
LC caa030714b IR: Remove unused VUShraI
Given that this is currently unused and that we don't have the signed
equivalent implemented, we can just remove this for now.
2026-07-10 04:15:52 -04:00
Ryan Houdek facbc78e0a Merge pull request #5679 from lioncash/macro
ALUOps: Remove unused macros
2026-07-10 00:50:52 -07:00
LC 6488dcbb01 ALUOps: Remove unused macros
These are now unused.
2026-07-10 03:30:03 -04:00
LC 37265b109a Merge pull request #5676 from Sonicadvance1/185
FEX: Remove FEXInterpreter binary
2026-07-09 18:39:36 -04:00
Ryan Houdek ebe7342d10 Merge pull request #5677 from mrpippy/protontso
Windows/UnixLib: Fix enabling TSO through legacy Proton codepath
2026-07-09 15:33:27 -07:00
Ryan Houdek 523bbef034 FEX: Remove FEXInterpreter binary
It's been ten months, a bit longer than than I was expecting to keep
this around. Go ahead and remove it now.
2026-07-09 15:11:30 -07:00
Brendan Shanks 84e127a637 Windows/UnixLib: Fix enabling TSO through legacy Proton codepath 2026-07-09 15:00:50 -07:00
Ryan Houdek 4a091df8cc Merge pull request #5672 from neobrain/feature_woa_code_cache_bitness
CodeCache/WoA: Support mixed WoW64/ARM64EC processing
2026-07-09 14:55:21 -07:00
Ryan Houdek 92a171ce53 Merge pull request #5675 from lioncash/branch
VectorOps: Join identical branches in VFMLS/VFNMLS
2026-07-09 13:58:54 -07:00
LC 1b1e46ff6c VectorOps: Join identical branches in VFMLS/VFNMLS
Same thing, just a little less redundant.
2026-07-09 16:33:26 -04:00
Ryan Houdek 3370d9af15 Merge pull request #5670 from simon902/MOVDoverride
Fix movd when prefixed with 0x66
2026-07-09 13:02:08 -07:00
Ryan Houdek 9306de79ad Merge pull request #5667 from simon902/CVTTSS2SIOverride
Fix cvttss2si when prefixed with 0x66
2026-07-09 12:52:00 -07:00
Ryan Houdek ff7a54add8 Merge pull request #5668 from OFFTKP/inf
Fix element getting overwritten in 66_5B test
2026-07-09 12:26:46 -07:00
Ryan Houdek c3d1157696 Merge pull request #5669 from OFFTKP/lzcnt
Fix LZCNT tests reading out of bounds
2026-07-09 12:24:43 -07:00
Ryan Houdek d0f03cb148 Merge pull request #5674 from lioncash/insertq
Vector: Trim one instruction off insertq
2026-07-09 12:24:06 -07:00
LC 9b7c9f0fb6 Vector: Trim one instruction off insertq
We can fold a bitwise not and and pair into a bic
2026-07-09 14:58:04 -04:00
Ryan Houdek e508b6df0d Merge pull request #5673 from lioncash/vbitwise
IR: Remove need to specify element size for vector bitwise ops
2026-07-09 11:42:37 -07:00
LC 710b85be70 IR: Remove need to specify element size for vector bitwise ops
Element size doesn't really mean anything here, considering all bits are
acted upon independently of segmentation.

Makes using these ops a little bit less noisy.
2026-07-09 13:26:51 -04:00
Tony Wasserka 66455b708a CodeCache: Switch between 32-/64-bit compilers during cache generation 2026-07-09 17:17:19 +02:00
Tony Wasserka 8b8000b98a CodeCache/WoA: Run cache generation in a subprocess to improve robustness 2026-07-09 17:16:16 +02:00
Tony Wasserka 54236df6e0 CodeCache: Record main executable bitness in code map
Code maps already contain the main executable they were recorded from, so
it's convenient to capture the executable's bitness along the way.
2026-07-09 17:12:45 +02:00
LC 6cd2a48910 Merge pull request #5666 from Sonicadvance1/184
FEXCore: Fixes a crash with multiblock if `ProcessorID` IR op is encountered
2026-07-09 10:16:06 -04:00
Simon Scherer fe08b96844 OpcodeDispatcher: Fix cvttss2si when prefixed with 0x66 2026-07-09 11:58:18 +02:00
Simon Scherer 4d78901420 OpcodeDispatcher: Fix movd when prefixed with 0x66 2026-07-09 11:46:48 +02:00
Simon Scherer c06468025c unittests/ASM: Test movd prefixed with 0x66 2026-07-09 11:46:10 +02:00
Paris Oplopoios 148e539025 Fix LZCNT tests reading out of bounds 2026-07-09 12:36:18 +03:00
Paris Oplopoios 92b96ff30d Fix element getting overwritten in 66_5B test 2026-07-09 11:54:38 +03:00
Simon Scherer 778df0c93b unittests/ASM: Test cvttss2si prefixed with 0x66 2026-07-09 09:24:59 +02:00
Ryan Houdek 9d18ecc5cb FEXCore: Fixes a crash with multiblock if ProcessorID IR op is encountered
If during multiblock code discovery a RDTSCP/RDPID instruction was
encountered then ProcessorID has an assert at JIT compile time. Make
sure to early exit with an illegal instruction encoding early instead.
Also make sure to correctly report RDPID support in CPUID, it's
technically a different bit than RDTSCP.

Fixes a crash in Crusader Kings 3's Paradox Launcher installer. Although
the installer seems to fail otherwise for some reason.
2026-07-08 17:41:03 -07:00
Ryan Houdek 5f2455c502 Merge pull request #5665 from lioncash/blendop
[SVE256] Handle 256-bit blend operations much more efficiently
2026-07-08 16:43:20 -07:00
LC 6bc67609a3 [SVE256] Handle 256-bit blend operations much more efficiently
We can massage a given selector into a valid predicate register bitmask
and then simply perform a merging move, which eliminates most busywork
around optimizing 256-bit blends.

In the future, once we drop SVE2.1 support in, we can use PMOV to
eliminate the load from memory and related constant management.
2026-07-08 17:35:12 -04:00
LC 1bd3945dd9 Merge pull request #5577 from Sonicadvance1/168
Context: Add support for single-step RIP ranges
2026-07-08 15:47:28 -04:00
Ryan Houdek 168f4b1e6b Context: Add support for single-step RIP ranges
Useful when debugging a range.
2026-07-08 12:21:16 -07:00
LC 8a8827c980 Merge pull request #5664 from simon902/CMPXCHGZeroing
OpcodeDispatcher: Fix 32bit cmpxchg zero extension with eax as first operand
2026-07-08 15:14:12 -04:00
Ryan Houdek 21a968b84d Merge pull request #5663 from simon902/PDEPoverlap
JIT/ALUOps: Fix operand overlapping bug for pdep
2026-07-08 11:08:12 -07:00
Simon Scherer 84fab84b3f InstcountCI: Update 2026-07-08 15:00:18 +02:00
Simon Scherer 3d65c030a8 OpcodeDispatcher: Fix 32bit cmpxchg zero extension with eax as destination operand and remove incorrect comment. 2026-07-08 14:58:32 +02:00
Simon Scherer 59097bab20 unittests/ASM: Test cmpxchg with eax as destination 2026-07-08 14:52:19 +02:00
Simon Scherer 4cbacd9261 InstcountCI: Update 2026-07-08 10:10:30 +02:00
Simon Scherer 655102fc7d JIT/ALUOps: Fix operand overlapping bug for pdep 2026-07-08 10:09:47 +02:00
Simon Scherer f718f46545 unittests/ASM: Test overlapping operands for pdep 2026-07-08 09:47:12 +02:00
LC 71d4e2c320 Merge pull request #5660 from Sonicadvance1/183
Wow64: Spin loop on atomic with WFE
2026-07-07 22:53:50 -04:00
Ryan Houdek 41241d7500 Wow64: Spin loop on atomic with WFE
Instead of burning roughly a million watts, put this spinloop on a WFE.
This tends to occur on a crash during shutdown that isn't fully able to
be avoided. The least we can do is not consume all the power in the
world.
2026-07-07 16:09:00 -07:00
Ryan Houdek b90c9836cb Merge pull request #5662 from lioncash/alias
OpcodeDispatcher: Remove asterisk from BMI source args
2026-07-07 11:22:59 -07:00
Ryan Houdek aff3fcf76d Merge pull request #5658 from neobrain/fix_woa_code_cache_ec
CodeCache: Mark executable memory as EC code on ARM64EC
2026-07-07 11:22:15 -07:00
Ryan Houdek ec2aa4063a Merge pull request #5661 from lioncash/blend
unittests: Add stress tests for VBLEND{PD, PS}
2026-07-07 11:13:56 -07:00
LC 718f2e01f7 OpcodeDispatcher: Remove asterisk from BMI source args
Keeps it consistent with the rest of the code and prevents breakages
whenever the Ref alias gets turned into its own value type.
2026-07-07 14:07:38 -04:00
LC ba9f7fb1b5 unittests: Add stress tests for VBLEND{PD, PS}
Forgot about these two
2026-07-07 13:49:09 -04:00
Tony Wasserka 0ac6b3e8f3 CodeCache: Mark executable memory as EC code on ARM64EC
See bd5b817c3a.
2026-07-07 15:55:27 +02:00
Ryan Houdek dddad1c2ca Merge pull request #5659 from lioncash/shuf
unittests: Add stress tests for VSHUF{PD, PS}
2026-07-06 15:50:08 -07:00
LC 95bfff20a4 unittests: Add stress tests for VSHUF{PD, PS}
Covers the remaining shuffle paths
2026-07-06 18:01:52 -04:00
LC 7a6f0def85 Merge pull request #5653 from Sonicadvance1/182
Config: Fixes AppOverrides with FEX_APP_CONFIG
2026-07-06 17:01:55 -04:00
Ryan Houdek 103d4d76be Config: Fixes AppOverrides with FEX_APP_CONFIG 2026-07-06 12:43:14 -07:00
Ryan Houdek db9414a756 Merge pull request #5655 from neobrain/feature_woa_cache_loading
Windows/ImageTracker: Adapt code cache loading logic to FEXOfflineCompiler
2026-07-06 12:40:40 -07:00
Ryan Houdek 5a0bf1bb5f Merge pull request #5657 from neobrain/fix_foc_syscall_abi_woa
FEXOfflineCompiler: Fix improper syscall ABI on WoA
2026-07-06 12:37:52 -07:00
LC 444c37fe2e Merge pull request #5656 from neobrain/fix_invalid_iterator_deref
Core: Fix dereference of invalid iterator
2026-07-06 10:13:04 -04:00
Tony Wasserka d3d735370f FEXOfflineCompiler: Fix improper syscall ABI on WoA 2026-07-06 15:16:30 +02:00
Tony Wasserka 852e93aa74 Core: Fix dereference of invalid iterator 2026-07-06 15:07:37 +02:00
Tony Wasserka dc1be2efe8 Windows/ImageTracker: Unindent refactored code 2026-07-06 14:58:16 +02:00
Tony Wasserka f2734ac608 Windows/ImageTracker: Adapt code cache loading logic to FEXOfflineCompiler 2026-07-06 14:58:16 +02:00
Tony Wasserka 57ca49dc5c Merge pull request #5501 from bylaws/finishloadwin
Windows/ImageTracker: Wire up LoadCache and EnableLoadedSection
2026-07-06 14:58:01 +02:00
Billy Laws 5ef3134a4d Windows/ImageTracker: Wire up LoadCache and EnableLoadedSection
The lazy code loading refactor replaced LoadData with the new
LoadCache/EnableLoadedSection API but left the Windows path as TODOs.
Implement the wiring: LoadAOTImages now calls LoadCache +
RegisterMappedCodeBuffer for each mapped cache file, and HandleImageMap
calls EnableLoadedSection (with nullptr thread since lazy mapping is not
yet implemented on Windows).
2026-07-06 14:44:59 +02:00
Ryan Houdek 5b91642883 Merge pull request #5654 from ShadowCurse/cache_va_size
Allocator: fix the caching of host va size
2026-07-05 18:32:10 -07:00
Egor Lazarchuk 1eb5abb8db Allocator: rename DetermineVASize to GetHostVABits
`DetermineVASize` does not return the size of VA, but the number of bits
it can use. Change the naming to make it more self explanatory.
In the mean time also move `HostVASize` global into `GetHostVABits`
since it is not and should not be used directly.
2026-07-05 12:46:41 +01:00
Egor Lazarchuk a65e1bf7e5 Allocator: fix the caching of host va size
Commit abf9724 ("Allocator: Fix and optimize VA range detection")
removed assignment to the `HostVASize` global thus making each call to
`DetermineVASize` redo all the work with potential to produce incorrect
results. Set the global again to fix this.
2026-07-05 12:46:29 +01:00
Ryan Houdek 3f98202b0f Merge pull request #5651 from lioncash/perm
unittests: Add stress tests for VPERMIL{PD, PS}
2026-07-04 12:13:24 -07:00
LC e7727c39f3 unittests: Add stress tests for VPERMIL{PD, PS}
While unlikely to be used in practice over other kind of
shuffling and blending, these should also have stress tests
to make sure they do the right thing.
2026-07-04 15:00:30 -04:00
Ryan Houdek 17e5637664 Merge pull request #5650 from lioncash/aes256
[SVE256] Handle 256-bit AES operations
2026-07-03 19:06:36 -07:00
LC 7fd9b897c2 [SVE256] Handle 256-bit AES operations
Currently we split these into two 128-bit operations since VIXL doesn't
have support for the unified SVE operations yet.

Now we fully support VAES on SVE256.
2026-07-03 21:52:59 -04:00
Ryan Houdek f8967aa207 Merge pull request #5649 from lioncash/pclmul256
[SVE256] EncryptionOps: Handle 256-bit VPCLMULQDQ
2026-07-03 18:33:53 -07:00
LC 5145324806 [SVE256] EncryptionOps: Handle 256-bit VPCLMULQDQ
Since vixl now handles this, we can drop this support right in.
2026-07-03 21:22:05 -04:00
Ryan Houdek 4233fb6270 Merge pull request #5648 from lioncash/aes
[SVE256] Ensure SSE insertion behavior for AES/SHA/PCLMUL operations
2026-07-03 17:33:19 -07:00
LC d11b19fd2b [SVE256] Ensure insertion behavior for PCLMUL SSE operations
Also includes accompanying test to ensure it never breaks.
2026-07-03 20:15:34 -04:00
LC 684c568033 [SVE256] Ensure insertion behavior for SHA SSE operations
These slipped through, so now we can add tests for them to prevent that
from happening again.
2026-07-03 20:09:50 -04:00
LC ab4fb7b3ad [SVE256] Ensure insertion behavior for AES operations on SSE
These slipped through, so now we can add tests for them to prevent that
from happening again.
2026-07-03 19:35:31 -04:00
Ryan Houdek 91017dbedb Merge pull request #5647 from lioncash/perm128
unittests: Add stress test for vperm2f128/vperm2i128
2026-07-03 15:32:34 -07:00
LC f5e8a051a8 unittests: Add stress test for vperm2f128/vperm2i128
Lets us test all possible immediate encodings for proper behavior.
2026-07-03 16:17:12 -04:00
LC 1d695f6db4 Merge pull request #5646 from Sonicadvance1/181
Github: More dependabot things
2026-07-02 22:22:45 -04:00
Ryan Houdek d4bdfd0592 Github: More dependabot things
They just never stop.
2026-07-02 19:05:12 -07:00
200 changed files with 11240 additions and 2830 deletions

No files matched your search

+1 -1
View File
@@ -43,7 +43,7 @@ jobs:
distrobox upgrade steamrt4
distrobox enter --name steamrt4 -- sudo apt-get install -y \
git cmake ninja-build ccache \
lld clang \
lld clang clang-tools \
libclang-dev llvm-dev \
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross \
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
+1 -1
View File
@@ -24,7 +24,7 @@ runs:
cmake -S . -B build_${{ inputs.target }} -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=Data/CMake/toolchain_mingw.cmake \
-DMINGW_TRIPLE=${_cc}-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja \
-DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False \
-DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=generic -DTUNE_CPU=none
-DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=generic -DTUNE_CPU=none -DRANGES_NATIVE=OFF
- name: Build
shell: bash
+2 -2
View File
@@ -31,12 +31,12 @@ build:
- apt-get -y update
- apt-get install -y
git cmake ninja-build ccache
lld clang
lld clang clang-tools
libclang-dev llvm-dev
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
- cmake -E make_directory build/
- cmake -DCMAKE_BUILD_TYPE=Release -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=armv8.2-a -DTUNE_CPU=none . -B build/
- cmake -DCMAKE_BUILD_TYPE=Release -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr -DTUNE_ARCH=armv8.2-a -DTUNE_CPU=none -DRANGES_NATIVE=OFF . -B build/
- cmake --build build/ --config Release
- DESTDIR=$(pwd)/install/ cmake --build build/ --config Release -t install
+3
View File
@@ -195,6 +195,9 @@ if (ENABLE_GDB_SYMBOLS)
endif()
add_compile_definitions(_LARGEFILE64_SOURCE)
if (WIN32)
add_compile_definitions(UNICODE _UNICODE)
endif()
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
+50 -53
View File
@@ -210,56 +210,53 @@ click==8.1.7 \
--hash=sha256:ae74fb96c20a0277a1d615f1e4d73c8414f5a98db8b799a7931d1582f3390c28 \
--hash=sha256:ca9853ad459e787e2192211578cc907e7594e294c7ccc834310722b41b9ca6de
# via black
cryptography==48.0.0 \
--hash=sha256:0890f502ddf7d9c6426129c3f49f5c0a39278ed7cd6322c8755ffca6ee675a13 \
--hash=sha256:0c558d2cdffd8f4bbb30fc7134c74d2ca9a476f830bb053074498fbc86f41ed6 \
--hash=sha256:16cd65b9330583e4619939b3a3843eec1e6e789744bb01e7c7e2e62e33c239c8 \
--hash=sha256:18349bbc56f4743c8b12dc32e2bccb2cf83ee8b69a3bba74ef8ae857e26b3d25 \
--hash=sha256:1e2d54c8be6152856a36f0882ab231e70f8ec7f14e93cf87db8a2ed056bf160c \
--hash=sha256:22a5cb272895dce158b2cacdfdc3debd299019659f42947dbdac6f32d68fe832 \
--hash=sha256:27241b1dc9962e056062a8eef1991d02c3a24569c95975bd2322a8a52c6e5e12 \
--hash=sha256:2b4d59804e8408e2fea7d1fbaf218e5ec984325221db76e6a241a9abd6cdd95c \
--hash=sha256:2eb992bbd4661238c5a397594c83f5b4dc2bc5b848c365c8f991b6780efcc5c7 \
--hash=sha256:369a6348999f94bbd53435c894377b20ab95f25a9065c283570e70150d8abc3c \
--hash=sha256:3cb07a3ed6431663cd321ea8a000a1314c74211f823e4177fefa2255e057d1ec \
--hash=sha256:40ba1f85eaa6959837b1d51c9767e230e14612eea4ef110ee8854ada22da1bf5 \
--hash=sha256:4defde8685ae324a9eb9d818717e93b4638ef67070ac9bc15b8ca85f63048355 \
--hash=sha256:55b7718303bf06a5753dcdccf2f3945cf18ad7bffde41b61226e4db31ab89a9c \
--hash=sha256:561215ea3879cb1cbbf272867e2efda62476f240fb58c64de6b393ae19246741 \
--hash=sha256:58d00498e8933e4a194f3076aee1b4a97dfec1a6da444535755822fe5d8b0b86 \
--hash=sha256:59baa2cb386c4f0b9905bd6eb4c2a79a69a128408fd31d32ca4d7102d4156321 \
--hash=sha256:5a5ed8fde7a1d09376ca0b40e68cd59c69fe23b1f9768bd5824f54681626032a \
--hash=sha256:5b012212e08b8dd5edc78ef54da83dd9892fd9105323b3993eff6bea65dc21d7 \
--hash=sha256:5c3932f4436d1cccb036cb0eaef46e6e2db91035166f1ad6505c3c9d5a635920 \
--hash=sha256:614d0949f4790582d2cc25553abd09dd723025f0c0e7c67376a1d77196743d6e \
--hash=sha256:76341972e1eff8b4bea859f09c0d3e64b96ce931b084f9b9b7db8ef364c30eff \
--hash=sha256:77a2ccbbe917f6710e05ba9adaa25fb5075620bf3ea6fb751997875aff4ae4bd \
--hash=sha256:7995ef305d7165c3f11ae07f2517e5a4f1d5c18da1376a0a9ed496336b69e5f3 \
--hash=sha256:7ce4bfae76319a532a2dc68f82cc32f5676ee792a983187dac07183690e5c66f \
--hash=sha256:7e8eac43dfca5c4cccc6dad9a80504436fca53bb9bc3100a2386d730fbe6b602 \
--hash=sha256:84cf79f0dc8b36ac5da873481716e87aef31fcfa0444f9e1d8b4b2cece142855 \
--hash=sha256:8c7378637d7d88016fa6791c159f698b3d3eed28ebf844ac36b9dc04a14dae18 \
--hash=sha256:8cd666227ef7af430aa5914a9910e0ddd703e75f039cef0825cd0da71b6b711a \
--hash=sha256:906cbf0670286c6e0044156bc7d4af9cbb0ef6db9f73e52c3ec56ba6bdde5336 \
--hash=sha256:9071196d81abc88b3516ac8cdfad32e2b66dd4a5393a8e68a961e9161ddc6239 \
--hash=sha256:9249e3cd978541d665967ac2cb2787fd6a62bddf1e75b3e347a594d7dacf4f74 \
--hash=sha256:984a20b0f62a26f48a3396c72e4bc34c66e356d356bf370053066b3b6d54634a \
--hash=sha256:9be5aafa5736574f8f15f262adc81b2a9869e2cfe9014d52a44633905b40d52c \
--hash=sha256:9c459db21422be75e2809370b829a87eb37f74cd785fc4aa9ea1e5f43b47cda4 \
--hash=sha256:9ccdac7d40688ecb5a3b4a604b8a88c8002e3442d6c60aead1db2a89a041560c \
--hash=sha256:a0e692c683f4df67815a2d258b324e66f4738bd7a96a218c826dce4f4bd05d8f \
--hash=sha256:a5da777e32ffed6f85a7b2b3f7c5cbc88c146bfcd0a1d7baf5fcc6c52ee35dd4 \
--hash=sha256:a64697c641c7b1b2178e573cbc31c7c6684cd56883a478d75143dbb7118036db \
--hash=sha256:ad64688338ed4bc1a6618076ba75fd7194a5f1797ac60b47afe926285adb3166 \
--hash=sha256:bd72e68b06bb1e96913f97dd4901119bc17f39d4586a5adf2d3e47bc2b9d58b5 \
--hash=sha256:c17dfe85494deaeddc5ce251aebd1d60bbe6afc8b62071bb0b469431a000124f \
--hash=sha256:c18684a7f0cc9a3cb60328f496b8e3372def7c5d2df39ac267878b05565aaaae \
--hash=sha256:cc90c0b39b2e3c65ef52c804b72e3c58f8a04ab2a1871272798e5f9572c17d20 \
--hash=sha256:db63bf618e5dea46c07de12e900fe1cdd2541e6dc9dbae772a70b7d4d4765f6a \
--hash=sha256:ea8990436d914540a40ab24b6a77c0969695ed52f4a4874c5137ccf7045a7057 \
--hash=sha256:ecde28a596bead48b0cfd2a1b4416c3d43074c2d785e3a398d7ec1fc4d0f7fbb \
--hash=sha256:f5333311663ea94f75dd408665686aaf426563556bb5283554a3539177e03b8c \
--hash=sha256:fdfef35d751d510fcef5252703621574364fec16418c4a1e5e1055248401054b
cryptography==49.0.0 \
--hash=sha256:026ac7423e6fa66872d3bf889be5974507da3944f866f704fa200eadacd00001 \
--hash=sha256:07cab27cc7b7e0fd28e5e26bb9eeedde5c135c868b46de4a27845abe94af6122 \
--hash=sha256:084ef1af862eb07ec46d25f68689f2102a9fc0e05ce7b80f14f5fe51e4eef0f6 \
--hash=sha256:0b82e28ee398a386f0807bba7884d30f25218855690f45115831bcce5d90822c \
--hash=sha256:0e959b578856a3924bc0cbb710fc12c387b9412a951389f3ca61704a9e25f325 \
--hash=sha256:0f21641cf4b30fca7aee061ced0ec7ad7b073518088b7c9969a297c0ae796c69 \
--hash=sha256:196ecd6a36e4e9aa10270393bb98d8df88fccee0bf1e5128b91ae4eb4375896d \
--hash=sha256:2400ef9c9e2299a25614eb1dea3db54a69b1349efd043bfac9c67630d136df36 \
--hash=sha256:28d8b15e6275f12c8a207dc309dfa957903c927d08d0cc937ee3f63f200693cc \
--hash=sha256:2afe9051da7ae7bd5905da5a949280c7d2bb75682e188f650a9d0f2756b834c6 \
--hash=sha256:2eda353d8a27bcbcaa4cbed18994a74ab4d19a2ca897db188ea269ab9b71419b \
--hash=sha256:32703d93296f5c1f4b53349ad3a250c2cae0fdecd3a3dd5d47e616d8d616af27 \
--hash=sha256:33cd0565932807baddb67b96dbee92f2c374b5c89dee09fd74079aeb8c8dba61 \
--hash=sha256:35b151772baff2c74cba7fa290ceaff4c3b11c0c881eb93eb5dbc05a7cfbba18 \
--hash=sha256:36d1709f992593689b45bda411498d62c6e365f2ca00b84657d4dadd24de16db \
--hash=sha256:42b0684e0e40cf26122427802486f6d93aea593612603a94fbf260c7eb1e9c1b \
--hash=sha256:4ae387c9cb68ea569ca17e490d66d8142b81c3cc814bf179974b7d146e490bbb \
--hash=sha256:53ecee2e23f7169b6117e99fc8a944e5e50f79e69758a83b52a00cb98ab2b2d2 \
--hash=sha256:66ec79c3904820572d7e987abdf304281f141d37ad9a489b8e97066e7b9b6459 \
--hash=sha256:67e1d20ad9ef3a563c59ef22e7a8a0b8210bd26604369ea4a30a7c66aefe504e \
--hash=sha256:6f2debedf9ca60cf1d5bd466475638af5130f89965605cd818484d19987d3a21 \
--hash=sha256:6fc361c34fb6aac015ce19435876635e5c6d21db31998b0920f675f131e043b8 \
--hash=sha256:73a205dce83953d131a4aa1e0fd917a2fd1c5b1eef251e9d7152efefcbf5caf7 \
--hash=sha256:7abcee80084cda3f7691f3eb1ce480d8df49cec637b429aa35986c1de71738aa \
--hash=sha256:8c25ceb16df5b9435f3f6a9829204985b0e0cbee3b48aacd432c7d2c850b44d9 \
--hash=sha256:966fe0e9c67490071f14c0d2b1cb2dfb3023c5ce39457343931415f08382f2db \
--hash=sha256:9e82dcc8e56052715fb18b2429e3bca4823b1629136a2084fc45a9a5cecb9b64 \
--hash=sha256:b20133d204d2bb56ba047642199603876c872026ca53e79c35b83772ab2cc505 \
--hash=sha256:b39efa323140595abd3ecca8529d321ae50f55f3aa3ba9cc81ea56a6011953d5 \
--hash=sha256:b47db11c2c3525083296069b98ac5221907455e989ae0c2e3008bde851921615 \
--hash=sha256:b87e65d263b3e5d3bb92a57e2a6638e2f31110fa7aa890c7b2dbba42248d0a3f \
--hash=sha256:b970c6da94d5bb18629db453d14f2a1300f6bf59b61e9b82377931ef95504866 \
--hash=sha256:be9fcb48a55f023493482827d4f459bd263cc20efde64f204b97c123201850c6 \
--hash=sha256:c2bc30226390d60ea19d9f82b19db005fe0452154a23c1c410c12ea801e43561 \
--hash=sha256:c83782480a4a9da4d0feb51950131ba32e12e70813848b3343f6e18c28a66838 \
--hash=sha256:cbc77da8c523d5abd028635ba850a6966fcee2c82e2bf65a41d1d8afe0f98be9 \
--hash=sha256:ccac2bfebc306b862133e3bb71f3f6ee8bb525240089b2d952e4144b3a6d5da7 \
--hash=sha256:d0527ce944105f257f605a827d6ebead966c752038b6e8656abb9c5edee6fc68 \
--hash=sha256:d8ecde755e2e91bf773fc94e8c9d730cd7f2007004cb492263a794ec3899a1c8 \
--hash=sha256:e3fb64c420688e5319ae25113a354015abbd8dffbfbc41781a1ea66fc7622ac3 \
--hash=sha256:e5dfc1e64de5677cec922ffa8da89c546d0415bf6efdf081842e5d44c84e1f0e \
--hash=sha256:ec5e529fb80935c94fe7b729f9972b50e351a0e6b50aa294fd5cabb109fcc29a \
--hash=sha256:f37d847238971164fdbc68ade6f6574aecc9c0af714190e2083429ff68f4ce9d \
--hash=sha256:f78ff2c9ed8dc2d036b0f4d640e22522213d047c1b14e61205a7e55c80a494d4 \
--hash=sha256:f89660a348f4f78a92366240a61404e337586ef7f5909a2fef59ca88ef505493 \
--hash=sha256:fc1e275c2f1d97b1a6450b8b0ea3ebfa6e087a611c2b26cb2404d48588abab7b
# via
# -r requirements_formatting.txt.in
# pyjwt
@@ -311,9 +308,9 @@ pygithub==2.6.1 \
--hash=sha256:6f2fa6d076ccae475f9fc392cc6cdbd54db985d4f69b8833a28397de75ed6ca3 \
--hash=sha256:b5c035392991cca63959e9453286b41b54d83bf2de2daa7d7ff7e4312cebf3bf
# via -r requirements_formatting.txt.in
pyjwt==2.12.1 \
--hash=sha256:28ca37c070cad8ba8cd9790cd940535d40274d22f80ab87f3ac6a713e6e8454c \
--hash=sha256:c74a7a2adf861c04d002db713dd85f84beb242228e671280bf709d765b03672b
pyjwt==2.13.0 \
--hash=sha256:41571c89ca91598c79e8ef18a2d07367d4810fbbd6f637794879baf1b7703423 \
--hash=sha256:66adcc2aff09b3f1bbd95fc1e1577df8ac8723c978552fd43304c8a290ac5728
# via
# -r requirements_formatting.txt.in
# pygithub
+2 -2
View File
@@ -1,10 +1,10 @@
black>=26.3.1
darker==2.1.1
PyGithub==2.6.1
cryptography>=46.0.7
cryptography>=48.0.1
urllib3>=2.7.0
requests>=2.33.0
idna>=3.15
certifi>=2024.7.4
PyNaCl>=1.6.2
PyJWT>=2.12.1
PyJWT>=2.13.0
+1 -1
+64 -60
View File
@@ -251,6 +251,10 @@ def parse_ops(ops):
if "Desc" in op_val:
OpDef.Desc = op_val["Desc"]
if not isinstance(OpDef.Desc, list):
ExitError(f"Desc field for op {OpDef.Name} must be an array of strings")
if not all(isinstance(item, str) for item in OpDef.Desc):
ExitError(f"Desc field for op {OpDef.Name} must only contain strings")
if "DynamicDispatch" in op_val:
OpDef.DynamicDispatch = bool(op_val["DynamicDispatch"])
@@ -603,77 +607,77 @@ def print_validation(op):
def print_ir_allocator_helpers():
output_file.write("#ifdef IROP_ALLOCATE_HELPERS\n")
output_file.write("\ttemplate <class T>\n")
output_file.write("\tstruct Wrapper final {\n")
output_file.write("\t\tT *first;\n")
output_file.write("\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n")
output_file.write("\n")
output_file.write("\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n")
output_file.write("\t\toperator OrderedNode *() { return Node; }\n")
output_file.write("\t\toperator const OrderedNode *() const { return Node; }\n")
output_file.write("\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n")
output_file.write("\t};\n")
output_file.write("\ttemplate <class T>\n"
"\tstruct Wrapper final {\n"
"\t\tT *first;\n"
"\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n"
"\n"
"\t\toperator Wrapper<IROp_Header>() const { return Wrapper<IROp_Header> {reinterpret_cast<IROp_Header*>(first), Node}; }\n"
"\t\toperator OrderedNode *() { return Node; }\n"
"\t\toperator const OrderedNode *() const { return Node; }\n"
"\t\toperator OpNodeWrapper () const { return Node->Header.Value; }\n"
"\t};\n")
output_file.write("\ttemplate <class T>\n")
output_file.write("\tusing IRPair = Wrapper<T>;\n\n")
output_file.write("\ttemplate <class T>\n"
"\tusing IRPair = Wrapper<T>;\n\n")
output_file.write("\tIRPair<IROp_Header> AllocateRawOp(size_t HeaderSize) {\n")
output_file.write("\t\tauto Op = reinterpret_cast<IROp_Header*>(DualListData.DataAllocate(HeaderSize));\n")
output_file.write("\t\tmemset(Op, 0, HeaderSize);\n")
output_file.write("\t\tOp->Op = IROps::OP_DUMMY;\n")
output_file.write("\t\treturn IRPair<IROp_Header>{Op, CreateNode(Op)};\n")
output_file.write("\t}\n\n")
output_file.write("\tIRPair<IROp_Header> AllocateRawOp(size_t HeaderSize) {\n"
"\t\tauto Op = reinterpret_cast<IROp_Header*>(DualListData.DataAllocate(HeaderSize));\n"
"\t\tmemset(Op, 0, HeaderSize);\n"
"\t\tOp->Op = IROps::OP_DUMMY;\n"
"\t\treturn IRPair<IROp_Header>{Op, CreateNode(Op)};\n"
"\t}\n\n")
output_file.write("\ttemplate<class T, IROps T2>\n")
output_file.write("\tT *AllocateOrphanOp() {\n")
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
output_file.write("\t\tmemset(Op, 0, Size);\n")
output_file.write("\t\tOp->Header.Op = T2;\n")
output_file.write("\t\treturn Op;\n")
output_file.write("\t}\n\n")
output_file.write("\ttemplate<class T, IROps T2>\n"
"\tT *AllocateOrphanOp() {\n"
"\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n"
"\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n"
"\t\tmemset(Op, 0, Size);\n"
"\t\tOp->Header.Op = T2;\n"
"\t\treturn Op;\n"
"\t}\n\n")
output_file.write("\ttemplate<class T, IROps T2>\n")
output_file.write("\tIRPair<T> AllocateOp() {\n")
output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n")
output_file.write("\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n")
output_file.write("\t\tmemset(Op, 0, Size);\n")
output_file.write("\t\tOp->Header.Op = T2;\n")
output_file.write("\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n")
output_file.write("\t}\n\n")
output_file.write("\ttemplate<class T, IROps T2>\n"
"\tIRPair<T> AllocateOp() {\n"
"\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n"
"\t\tauto Op = reinterpret_cast<T*>(DualListData.DataAllocate(Size));\n"
"\t\tmemset(Op, 0, Size);\n"
"\t\tOp->Header.Op = T2;\n"
"\t\treturn IRPair<T>{Op, CreateNode(&Op->Header)};\n"
"\t}\n\n")
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\t\treturn HeaderOp->Size;\n")
output_file.write("\t}\n\n")
output_file.write("\tIR::OpSize GetOpSize(const OrderedNode *Op) const {\n"
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
"\t\treturn HeaderOp->Size;\n"
"\t}\n\n")
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\t\treturn HeaderOp->ElementSize;\n")
output_file.write("\t}\n\n")
output_file.write("\tIR::OpSize GetOpElementSize(const OrderedNode *Op) const {\n"
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
"\t\treturn HeaderOp->ElementSize;\n"
"\t}\n\n")
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n")
output_file.write("\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n")
output_file.write("\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n")
output_file.write("\t}\n\n")
output_file.write("\tuint8_t GetOpElements(const OrderedNode *Op) const {\n"
"\t\tLOGMAN_THROW_A_FMT(OpHasDest(Op), \"Op {} has no dest\\n\", GetOpName(Op));\n"
"\t\treturn IR::OpSizeToSize(GetOpSize(Op)) / IR::OpSizeToSize(GetOpElementSize(Op));\n"
"\t}\n\n")
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\t\treturn GetHasDest(HeaderOp->Op);\n")
output_file.write("\t}\n\n")
output_file.write("\tbool OpHasDest(const OrderedNode *Op) const {\n"
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
"\t\treturn GetHasDest(HeaderOp->Op);\n"
"\t}\n\n")
output_file.write("\tIROps GetOpType(const OrderedNode *Op) const {\n")
output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n")
output_file.write("\t\treturn HeaderOp->Op;\n")
output_file.write("\t}\n\n")
output_file.write("\tIROps GetOpType(const OrderedNode *Op) const {\n"
"\t\tauto HeaderOp = Op->Header.Value.GetNode(DualListData.DataBegin());\n"
"\t\treturn HeaderOp->Op;\n"
"\t}\n\n")
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n")
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
output_file.write("\t}\n\n")
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n"
"\t\treturn GetRegClass(GetOpType(Op));\n"
"\t}\n\n")
output_file.write("\tstd::string_view const& GetOpName(const OrderedNode *Op) const {\n")
output_file.write("\t\treturn IR::GetName(GetOpType(Op));\n")
output_file.write("\t}\n\n")
output_file.write("\tstd::string_view const& GetOpName(const OrderedNode *Op) const {\n"
"\t\treturn IR::GetName(GetOpType(Op));\n"
"\t}\n\n")
# Generate helpers with operands
for op in IROps:
+1
View File
@@ -24,6 +24,7 @@ set(SRCS
Interface/Core/Addressing.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/SharedCodeBufferManager.cpp
Interface/Core/OpcodeDispatcher/AVX_128.cpp
Interface/Core/OpcodeDispatcher/Crypto.cpp
Interface/Core/OpcodeDispatcher/Flags.cpp
+19 -14
View File
@@ -18,7 +18,7 @@ struct BitSet final {
constexpr static size_t MinimumSize = sizeof(ElementType);
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
ElementType* Memory;
ElementType* Memory {};
void Allocate(size_t Elements) {
size_t AllocateSize = ToBytes(Elements);
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
@@ -33,14 +33,15 @@ struct BitSet final {
FEXCore::Allocator::free(Memory);
Memory = nullptr;
}
bool Get(T Element) {
[[nodiscard]]
bool Get(T Element) const {
return (Memory[Element / MinimumSizeBits] & (1ULL << (Element % MinimumSizeBits))) != 0;
}
void Set(T Element) {
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
}
void Clear(T Element) {
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
}
void MemClear(size_t Elements) {
memset(Memory, 0, ToBytes(Elements));
@@ -48,13 +49,15 @@ struct BitSet final {
void MemSet(size_t Elements) {
memset(Memory, 0xFF, ToBytes(Elements));
}
uint32_t ToBytes(size_t Elements) {
return AlignUp(Elements, MinimumSizeBits) / MinimumSize;
[[nodiscard]]
static size_t ToBytes(size_t Elements) {
return AlignUp(Elements, MinimumSizeBits) / 8;
}
// This very explicitly doesn't let you take an address
// Is only a getter
bool operator[](T Element) {
[[nodiscard]]
bool operator[](T Element) const {
return Get(Element);
}
};
@@ -62,35 +65,37 @@ struct BitSet final {
template<typename T>
struct BitSetView final {
using ElementType = T;
constexpr static size_t MinimumSize = sizeof(ElementType);
constexpr static size_t MinimumSizeBits = sizeof(ElementType) * 8;
constexpr static size_t MinimumSize = BitSet<T>::MinimumSize;
constexpr static size_t MinimumSizeBits = BitSet<T>::MinimumSizeBits;
ElementType* Memory;
ElementType* Memory {};
void GetView(BitSet<T>& Set, uint64_t ElementOffset) {
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0, "Bitset view offset needs to be aligned to size of backing element");
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
}
bool Get(T Element) {
[[nodiscard]]
bool Get(T Element) const {
return (Memory[Element / MinimumSizeBits] & (1ULL << (Element % MinimumSizeBits))) != 0;
}
void Set(T Element) {
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
}
void Clear(T Element) {
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
}
void MemClear(size_t Elements) {
memset(Memory, 0, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
memset(Memory, 0, BitSet<T>::ToBytes(Elements));
}
void MemSet(size_t Elements) {
memset(Memory, 0xFF, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
memset(Memory, 0xFF, BitSet<T>::ToBytes(Elements));
}
// This very explicitly doesn't let you take an address
// Is only a getter
bool operator[](T Element) {
[[nodiscard]]
bool operator[](T Element) const {
return Get(Element);
}
};
+1
View File
@@ -4,6 +4,7 @@
#include <concepts>
#include <string_view>
#include <cstdlib>
namespace FEXCore::StrConv {
template<std::integral T>
+3 -2
View File
@@ -1,9 +1,10 @@
// SPDX-License-Identifier: MIT
#include "Common/StringConv.h"
#include "FEXCore/Utils/EnumUtils.h"
#include "Utils/Config.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/FileLoading.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/StringUtils.h>
@@ -252,7 +253,7 @@ void Load() {
}
}
fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
static fextl::string ExpandPath(const fextl::string& ContainerPrefix, const fextl::string& PathName) {
if (PathName.empty()) {
return {};
}
@@ -55,4 +55,10 @@ FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionN
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const {
return Thread->CPUBackend->IsAddressInCodeBuffer(Address) || CodeCache.IsAddressInMappedCodeBuffer(Address);
}
bool FEXCore::Context::ContextImpl::RequiresRelocatableConstants() const {
// Support relocation when generating a cache or when generating reference code for validation
return CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHEVALIDATION();
}
} // namespace FEXCore::Context
+18 -2
View File
@@ -4,6 +4,7 @@
#include "Common/JitSymbols.h"
#include "Interface/Core/CPUBackend.h"
#include "Interface/Core/CPUID.h"
#include "Interface/Core/SharedCodeBufferManager.h"
#include <Interface/IR/IntrusiveIRList.h>
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/Context.h>
@@ -130,7 +131,7 @@ public:
uint32_t RelocationOffset, bool ForStorage);
};
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
class ContextImpl final : public FEXCore::Context::Context, public CPU::SharedCodeBufferManager {
public:
// Context base class implementation.
bool InitCore() override;
@@ -247,6 +248,7 @@ public:
struct TrackingEmpty {
// RIP stepping handling
virtual void AddSingleStepTarget(uint64_t GuestRIP) {}
virtual void AddSingleStepTargetRange(uint64_t RIPBegin, uint64_t RipEnd) {}
virtual void AllTargetSingleStep() {}
virtual void RemoveSingleStepTarget(uint64_t GuestRIP) {}
virtual bool IsSingleStepTarget(uint64_t GuestRIP) {
@@ -269,6 +271,10 @@ public:
SingleStepTargets.emplace(GuestRIP);
}
virtual void AddSingleStepTargetRange(uint64_t RIPBegin, uint64_t RIPEnd) override {
SingleStepRanges.emplace_back(Range {RIPBegin, RIPEnd});
}
void RemoveSingleStepTarget(uint64_t GuestRIP) override {
SingleStepTargets.erase(GuestRIP);
}
@@ -278,7 +284,7 @@ public:
}
bool IsSingleStepTarget(uint64_t GuestRIP) override {
return SingleStepEverything || SingleStepTargets.contains(GuestRIP);
return SingleStepEverything || SingleStepTargets.contains(GuestRIP) || IsInRange(GuestRIP);
}
void AddWriteWatchPoint(uint64_t Ptr) override {
@@ -302,6 +308,14 @@ public:
fextl::set<uint64_t> SingleStepTargets {};
fextl::set<uint64_t> WatchWriteTargets {};
fextl::set<uint64_t> WatchReadTargets {};
struct Range {
uint64_t Begin, End;
};
fextl::vector<Range> SingleStepRanges {};
bool IsInRange(uint64_t RIP) const {
return std::ranges::any_of(SingleStepRanges, [RIP](const auto& range) { return RIP >= range.Begin && RIP <= range.End; });
}
static bool ContainsRange(const fextl::set<uint64_t>& Set, uint64_t Ptr, size_t Size) {
for (auto it = Set.lower_bound(Ptr); it != Set.end(); --it) {
@@ -438,6 +452,8 @@ public:
return Config.MonoHacks && MonoDetected;
}
bool RequiresRelocatableConstants() const;
protected:
void UpdateAtomicTSOEmulationConfig() {
if (SupportsHardwareTSO) {
@@ -360,6 +360,7 @@ namespace x32 {
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr, size_t size)
: Emitter(static_cast<uint8_t*>(EmissionPtr), size)
, EmitterCTX {ctx}
, SupportCodeRelocations {ctx->RequiresRelocatableConstants()}
#ifdef VIXL_SIMULATOR
, Simulator {&SimDecoder, stdout, vixl::aarch64::SimStack(SimulatorStackSize).Allocate()}
#endif
@@ -425,7 +426,7 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
NOPPad = false;
} else if (Pad == PadType::AUTOPAD) {
// Force NOP padding to ensure relocated constants always have enough encoding space available
NOPPad = EnableCodeCaching;
NOPPad = SupportCodeRelocations;
}
bool Is64Bit = s == ARMEmitter::Size::i64Bit;
@@ -633,7 +634,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
}
}
void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs) {
void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, const FillSpecialRegsOptions& Options) {
#ifndef VIXL_SIMULATOR
if (EmitterCTX->HostFeatures.SupportsAFP) {
// Enable AFP features when filling JIT state.
@@ -649,7 +650,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
(1U << 2) | // NEP
(1U << 1)); // AH
if (SetFIZ) {
if (Options.SetFIZ) {
// Insert MXCSR.DAZ in to FIZ
ldr(TmpReg2.W(), STATE.R(), offsetof(FEXCore::Core::CPUState, mxcsr));
bfxil(ARMEmitter::Size::i64Bit, TmpReg, TmpReg2, 6, 1);
@@ -659,7 +660,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
}
#endif
if (SetPredRegs && EmitterCTX->HostFeatures.SupportsSVE()) {
if (Options.SetPredRegs && EmitterCTX->HostFeatures.SupportsSVE()) {
// Set up predicate registers.
// We don't bother spilling these in SpillStaticRegs,
// since all that matters is we restore them on a fill.
@@ -822,7 +823,7 @@ void Arm64Emitter::FillStaticRegs(FillStaticRegOptions Options) {
msr(ARMEmitter::SystemRegister::NZCV, TmpReg);
}
FillSpecialRegs(TmpReg, TmpReg2, true, Options.FPRs);
FillSpecialRegs(TmpReg, TmpReg2, {.SetFIZ = true, .SetPredRegs = Options.FPRs});
if (Options.FPRs) {
if (EmitterCTX->HostFeatures.SupportsAVX && EmitterCTX->HostFeatures.SupportsSVE256) {
@@ -1059,6 +1060,7 @@ size_t Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, boo
SpillStaticRegs(TmpReg, {
.GPRSpillMask = PreserveSRAMask,
.FPRSpillMask = PreserveSRAFPRMask,
.FPRs = FPRs,
});
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
@@ -129,7 +129,21 @@ protected:
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
uint32_t PairRegisters = 0;
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
bool SupportCodeRelocations;
struct FillSpecialRegsOptions {
// Whether or not to set the FPCR.FIZ (flush inputs to zero) bit in the FPCR to
// the current value of the emulated MXCSR.DAZ bit.
// Will only attempt to do so, even when set to true, if and only if the host system
// supports FEAT_AFP.
bool SetFIZ {};
// Whether or not FillSpecialRegs should load our SVE predicate temporaries
// with certain canned values that accelerate some operations. Will (obviously)
// not load predicates, even if set to true, on host systems that do not support SVE.
bool SetPredRegs {};
};
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, const FillSpecialRegsOptions& Options);
// Correlate an ARM register back to an x86 register index.
// Returning REG_INVALID if there was no mapping.
@@ -308,8 +322,6 @@ protected:
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
#endif
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
};
} // namespace FEXCore::CPU
+11 -107
View File
@@ -11,17 +11,8 @@
#include <cstdint>
#ifndef _WIN32
#include <sys/prctl.h>
#endif
namespace FEXCore {
namespace CPU {
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
@@ -275,9 +266,9 @@ namespace CPU {
return TotalLUT;
}()};
CPUBackend::CPUBackend(CodeBufferManager& CodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
CPUBackend::CPUBackend(SharedCodeBufferManager& SharedCodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
: ThreadState(ThreadState)
, CodeBuffers(CodeBuffers) {
, SharedCodeBuffers(SharedCodeBuffers) {
auto& Ptrs = ThreadState->CurrentFrame->Pointers;
@@ -316,11 +307,11 @@ namespace CPU {
CPUBackend::~CPUBackend() = default;
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
auto CPUBackend::GetEmptySharedCodeBuffer() -> CodeBuffer* {
auto PrevCodeBuffer = CurrentCodeBuffer;
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
CurrentCodeBuffer = SharedCodeBuffers.StartLargerCodeBuffer();
RegisterForSignalHandler(std::move(PrevCodeBuffer));
return CurrentCodeBuffer.get();
@@ -338,7 +329,7 @@ namespace CPU {
}
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
auto NewCodeBuffer = CodeBuffers.GetLatest();
auto NewCodeBuffer = SharedCodeBuffers.GetLatest();
if (CurrentCodeBuffer != NewCodeBuffer) {
RegisterForSignalHandler(CurrentCodeBuffer);
return std::exchange(CurrentCodeBuffer, NewCodeBuffer);
@@ -346,107 +337,20 @@ namespace CPU {
return nullptr;
}
GuestToHostMap& GetLookupCache(const CodeBuffer& Buffer) {
return *Buffer.LookupCache;
}
CodeBuffer::CodeBuffer(size_t Size)
: AllocatedSize(Size) {
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
FEXCore::Allocator::ProtectOptions::None)) {
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
}
FEXCore::Allocator::VirtualName("FEXMemJIT", reinterpret_cast<void*>(Ptr), Size);
// Huge-pages reduce the amount of iTLB misses dramatically when it works.
FEXCore::Allocator::VirtualTHPControl(reinterpret_cast<void*>(Ptr), Size, FEXCore::Allocator::THPControl::Enable);
LookupCache = fextl::make_unique<GuestToHostMap>();
}
CodeBuffer::~CodeBuffer() {
FEXCore::Allocator::VirtualFree(Ptr, AllocatedSize);
}
auto CodeBufferManager::AllocateNew(size_t Size) -> fextl::shared_ptr<CodeBuffer> {
#ifndef _WIN32
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
//
// MDWE prevents applications from creating RWX memory mappings.
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
//
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
//
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
//
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
// -1: The kernel doesn't support MDWE
// 0: MDWE is supported but disabled
// >0: MDWE is enabled, hence prohibiting RWX mappings
#ifndef PR_GET_MDWE
#define PR_GET_MDWE 66
#endif
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
if (MDWE != -1 && MDWE != 0) {
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
}
#endif
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
Latest = Buffer;
LatestOffset = 0;
OnCodeBufferAllocated(Buffer);
return Buffer;
}
fextl::shared_ptr<CodeBuffer> CodeBufferManager::GetLatest() {
if (!Latest) {
if (FEXCore::Config::Get_ENABLECODECACHINGWIP()) {
// Start with a larger code buffer to avoid resizes that would discard
// code loaded from caches
AllocateNew(MAX_CODE_SIZE);
} else {
AllocateNew(INITIAL_CODE_SIZE);
}
}
return Latest;
}
fextl::shared_ptr<CodeBuffer> CodeBufferManager::StartLargerCodeBuffer() {
if (!Latest) {
// Allocate initial CodeBuffer and return it
return GetLatest();
}
auto NewCodeBufferSize = GetLatest()->AllocatedSize;
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
return AllocateNew(NewCodeBufferSize);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
auto CheckCodeBuffer = [](CodeBuffer& Buffer, uintptr_t Address) {
const auto CheckCodeBuffer = [](const CodeBuffer& Buffer, uintptr_t Address) {
const auto BufferPtr = reinterpret_cast<uintptr_t>(Buffer.Ptr);
// The last page of the code buffer is protected, so we need to exclude it from the valid range
// when checking if the address is in the code buffer.
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.AllocatedSize - 1, FEXCore::Utils::FEX_PAGE_SIZE);
return (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr);
const uintptr_t LastPageAddr = AlignDown(BufferPtr + Buffer.AllocatedSize - 1, FEXCore::Utils::FEX_PAGE_SIZE);
return (Address >= BufferPtr && Address < LastPageAddr);
};
if (CheckCodeBuffer(*CurrentCodeBuffer, Address)) {
return true;
}
for (auto& Buffer : SignalHandlerCodeBuffers) {
for (const auto& Buffer : SignalHandlerCodeBuffers) {
if (CheckCodeBuffer(*Buffer, Address)) {
return true;
}
+5 -56
View File
@@ -8,6 +8,8 @@ $end_info$
#pragma once
#include "Interface/Core/SharedCodeBufferManager.h"
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
@@ -41,63 +43,10 @@ namespace CodeSerialize {
struct GuestToHostMap;
namespace CPU {
struct CodeBuffer {
uint8_t* Ptr;
size_t AllocatedSize; // including guard page; see UsableSize()
fextl::unique_ptr<GuestToHostMap> LookupCache;
CodeBuffer(size_t Size);
CodeBuffer(const CodeBuffer&) = delete;
CodeBuffer& operator=(const CodeBuffer&) = delete;
CodeBuffer(CodeBuffer&& oth) = delete;
CodeBuffer& operator=(CodeBuffer&&) = delete;
~CodeBuffer();
/// Returns the number of bytes available for storing code
size_t UsableSize() const {
return AllocatedSize - FEXCore::Utils::FEX_PAGE_SIZE;
}
};
/**
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
*
* The CodeBuffer is managed as a partially persistent data structure:
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
*/
class CodeBufferManager {
public:
// Get the CodeBuffer that was most recently allocated.
// This is the only CodeBuffer that data may be written to.
fextl::shared_ptr<CodeBuffer> GetLatest();
// Allocate a new CodeBuffer with geometric growth up to an internal maximum.
// Subsequent calls to GetLatest will point to the returned buffer.
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer();
// Write offset into the latest CodeBuffer
std::size_t LatestOffset {};
// Protects writes to the latest CodeBuffer and changes to LatestOffset
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
private:
fextl::shared_ptr<CodeBuffer> Latest;
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
};
class CPUBackend {
public:
CPUBackend(CodeBufferManager&, FEXCore::Core::InternalThreadState*);
CPUBackend(SharedCodeBufferManager&, FEXCore::Core::InternalThreadState*);
virtual ~CPUBackend();
@@ -190,7 +139,7 @@ namespace CPU {
FEXCore::Core::InternalThreadState* ThreadState;
[[nodiscard]]
CodeBuffer* GetEmptyCodeBuffer();
CodeBuffer* GetEmptySharedCodeBuffer();
// This is the code buffer containing the main code under execution by this thread.
// CheckCodeBufferUpdate must be used before compiling new code.
@@ -199,7 +148,7 @@ namespace CPU {
// Old CodeBuffer generations required to be valid until returning from signal handlers
fextl::vector<fextl::shared_ptr<CodeBuffer>> SignalHandlerCodeBuffers;
CodeBufferManager& CodeBuffers;
SharedCodeBufferManager& SharedCodeBuffers;
private:
void RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer>);
+11 -5
View File
@@ -100,7 +100,7 @@ namespace ProductNames {
#endif
} // namespace ProductNames
uint32_t GetCPUID_Syscall() {
static uint32_t GetCPUID_Syscall() {
uint32_t CPU {};
FHU::Syscalls::getcpu(&CPU, nullptr);
return CPU;
@@ -148,7 +148,7 @@ uint64_t GetCycleCounterFrequency() {
return Result;
}
uint32_t GetCPUID_TPIDRRO() {
static uint32_t GetCPUID_TPIDRRO() {
uint64_t Result {};
__asm("mrs %[Res], TPIDRRO_EL0" : [Res] "=r"(Result));
return Result;
@@ -316,9 +316,8 @@ void CPUIDEmu::SetupHostHybridFlag() {
// Walk our list of CPUMIDRs to find the most little core
for (size_t j = LowestMIDRIdx; j < CPUMIDRs.size(); ++j) {
auto& MIDROption = CPUMIDRs[i];
const auto& MIDROption = CPUMIDRs[j];
if ((MIDROption.Implementer == Implementer && MIDROption.Part == Part) || (MIDROption.Implementer == 0 && MIDROption.Part == 0)) {
LowestMIDRIdx = j;
LowestMIDR = MIDR;
break;
@@ -650,6 +649,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) const {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
FEXCore::CPUID::FunctionResults Res {};
if (Leaf == 0) {
#ifndef _WIN32
constexpr uint32_t SUPPORTS_RDPID = 1;
#else
// RDPID under WIN32 is only supported if CPUIndex is available in TPIDRRO.
const uint32_t SUPPORTS_RDPID = SupportsCPUIndexInTPIDRRO;
#endif
// Disable Enhanced REP MOVS when TSO is enabled.
// vcruntime140 memmove will use `rep movsb` in this case which completely destroys perf in Hades(appId 1145360)
// This is due to LRCPC performance on Cortex being abysmal.
@@ -715,7 +721,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 19) | // MPX MAWAU
(0 << 20) | // MPX MAWAU
(0 << 21) | // MPX MAWAU
(1 << 22) | // RDPID Read Processor ID
(SUPPORTS_RDPID << 22) | // RDPID Read Processor ID
(0 << 23) | // AES Key Locker
(1 << 24) | // bus-lock-detect
(0 << 25) | // CLDEMOTE
+29 -7
View File
@@ -50,6 +50,10 @@ MappedCodeCacheFile::~MappedCodeCacheFile() {
if (!CodeBuffer.empty()) {
FEXCore::Allocator::munmap(CodeBuffer.data(), CodeBuffer.size_bytes());
}
#elif defined(_M_ARM64EC)
if (!CodeBuffer.empty()) {
FEXCore::Allocator::VirtualFree(CodeBuffer.data(), CodeBuffer.size_bytes());
}
#endif
}
@@ -107,16 +111,22 @@ fextl::map<CodeMapFileId, CodeMap::ParsedContents> CodeMap::ParseCodeMap(std::if
break;
}
Ret[Info.ExternalFileId].Filename = std::move(Filename);
} else if (Entry.FileId == SetExecutableFileId {}.Marker.FileId && Entry.BlockOffset == SetExecutableFileId {}.Marker.BlockOffset) {
} else if ((Entry.FileId == SetExecutableFileId::Marker32.FileId && Entry.BlockOffset == SetExecutableFileId::Marker32.BlockOffset) ||
(Entry.FileId == SetExecutableFileId::Marker64.FileId && Entry.BlockOffset == SetExecutableFileId::Marker64.BlockOffset)) {
CodeMapFileId ExecutableFileId;
File.read(reinterpret_cast<char*>(&ExecutableFileId), sizeof(ExecutableFileId));
if (!File) {
break;
}
Ret[ExecutableFileId].IsExecutable = true;
Ret[ExecutableFileId].ExecutableBitness =
(Entry.FileId == SetExecutableFileId::Marker32.FileId && Entry.BlockOffset == SetExecutableFileId::Marker32.BlockOffset) ? 32 : 64;
} else {
if (!Ret.contains(Entry.FileId)) {
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
if (Entry.FileId == 0xffff'ffff'ffff'ffff) {
ERROR_AND_DIE_FMT("Malformed code map");
} else {
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
}
} else {
Ret[Entry.FileId].Blocks.insert(Entry.BlockOffset);
}
@@ -222,8 +232,8 @@ void CodeMapWriter::AppendLibraryLoad(const FEXCore::ExecutableFileInfo& FileInf
AppendData(std::as_bytes(std::span {Data, TotalSize}));
}
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo) {
CodeMap::SetExecutableFileId Data {.ExecutableFileId = FileInfo.FileId};
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo, bool Is64Bit) {
CodeMap::SetExecutableFileId Data {Is64Bit ? CodeMap::SetExecutableFileId::Marker64 : CodeMap::SetExecutableFileId::Marker32, FileInfo.FileId};
AppendData(std::span {reinterpret_cast<const std::byte*>(&Data), sizeof(Data)});
}
@@ -466,7 +476,7 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
if (tail->RIP >= Section.BeginVA && tail->RIP < Section.EndVA) {
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length, _] =
ValidationCTX->GenerateIR(ValidationThread.get(), tail->RIP, false, FEXCore::Config::Get_MAXINST());
fextl::stringstream ss;
fextl::ostringstream ss;
FEXCore::IR::Dump(&ss, &*IRView);
LogMan::Msg::EFmt("IR:\n{}", ss.str());
} else {
@@ -600,7 +610,16 @@ CodeCache::LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo& F
return nullptr;
}
auto CodeBuffer = std::span {static_cast<std::byte*>(CodeBufferAllocation), header.CodeBufferSize};
#else
#elif defined(_M_ARM64EC)
// TODO: Implement lazy mapping on Windows
// NOTE: The executed code must have MEM_EXTENDED_PARAMETER_EC_CODE set, so we can't operate on the mapped cache file directly
void* CodeBufferAllocation = Allocator::VirtualAlloc(header.CodeBufferSize, true);
if (!CodeBufferAllocation) {
LogMan::Msg::EFmt("Failed to allocate code cache memory");
return nullptr;
}
auto CodeBuffer = std::span {reinterpret_cast<std::byte*>(CodeBufferAllocation), header.CodeBufferSize};
#else // WoW64
// TODO: Implement lazy mapping on Windows
auto CodeBuffer = CodeDataInFile;
#endif
@@ -870,6 +889,9 @@ void CodeCache::FinalizeCodePages(MappedCodeCacheFile& Code, std::span<std::byte
Allocator::VirtualDontNeed(Code.CodeBufferInFile.data() + StartOffset, Size);
#else
// TODO: Implement lazy mapping on Windows
#ifdef _M_ARM64EC
memcpy(Code.CodeBuffer.data() + StartOffset, Code.CodeBufferInFile.data() + StartOffset, Size);
#endif
for (size_t i = StartPage; i < EndPage; ++i) {
auto PageRelocations = SpanPageRelocations(Code, i);
(void)ApplyCodeRelocations(Code.GuestBase, Code.CodeBuffer, PageRelocations, 0, false);
+26 -25
View File
@@ -342,6 +342,11 @@ void ContextImpl::SetFlagsFromCompactedEFLAGS(FEXCore::Core::InternalThreadState
}
bool ContextImpl::InitCore() {
if (CodeCache.IsGeneratingCache || FEXCore::Config::Get_ENABLECODECACHINGWIP()) {
// Start with a larger code buffer to avoid resizes that would discard code
StartMaximalCodeBuffer();
}
// Initialize the CPU core signal handlers & DispatcherConfig
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
@@ -390,7 +395,7 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>(this);
Thread->CurrentFrame->State.L1Pointer = Thread->LookupCache->GetL1Pointer();
Thread->CurrentFrame->State.L1Mask = Thread->LookupCache->GetScaledL1PointerMask();
@@ -399,15 +404,11 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
Dispatcher->InitThreadPointers(Thread);
Thread->PassManager->AddDefaultPasses(this);
Thread->PassManager->AddDefaultValidationPasses();
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
// Create CPU backend
Thread->PassManager->InsertRegisterAllocationPass(this);
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
// We finalize *after* the CPU backend is initialized, as the CPU backend will
// provide necessary register information to the register allocation pass.
Thread->PassManager->Finalize();
}
@@ -505,11 +506,11 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, boo
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
FEXCore::File::File FD = FEXCore::File::File::GetStdERR();
fextl::stringstream out;
fextl::ostringstream out;
auto NewIR = IREmitter->ViewIR();
FEXCore::IR::Dump(&out, &NewIR);
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", NewIR.PostRA() ? "post" : "pre", GuestRIP, out.str());
};
}
bool ContextImpl::CheckIfBlockIsCacheable(FEXCore::Core::InternalThreadState& Thread, uint64_t GuestRIP, uint64_t MaxInst) {
return Thread.FrontendDecoder->CheckIfCacheable(Thread, reinterpret_cast<const uint8_t*>(GuestRIP), GuestRIP, MaxInst);
@@ -538,18 +539,14 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
}
if (!HasCustomIR) {
const uint8_t* GuestCode {};
GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
bool HadDispatchError {false};
bool HadInvalidInst {false};
const auto* GuestCode = reinterpret_cast<const uint8_t*>(GuestRIP);
Thread->FrontendDecoder->DecodeInstructionsAtEntry(Thread, GuestCode, GuestRIP, MaxInst);
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
auto CodeBlocks = &BlockInfo->Blocks;
const auto* BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
const auto& CodeBlocks = BlockInfo->Blocks;
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
Thread->OpDispatcher->BeginFunction(GuestRIP, &CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
AreMonoHacksActive() && MonoBackpatcherBlock.load(std::memory_order_relaxed) == GuestRIP);
const auto GPRSize = Thread->OpDispatcher->GetGPROpSize();
@@ -563,11 +560,17 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
}
#endif
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
for (size_t j = 0; j < CodeBlocks.size(); ++j) {
const auto& Block = CodeBlocks[j];
// Dispatch failures and invalid instructions terminate only the decoded
// block that contains them. Other block targets in the same multiblock
// compilation unit are independent entry paths.
bool HadDispatchError {false};
bool HadInvalidInst {false};
#ifdef ZYDIS_DISASSEMBLER
if (FEXCore::Config::Get_X86DISASSEMBLE() && CodeBlocks->size() > 1) {
if (FEXCore::Config::Get_X86DISASSEMBLE() && CodeBlocks.size() > 1) {
LogMan::Msg::IFmt(" Block {} Entry={:#x} NumInsts={}", j, Block.Entry, Block.NumInstructions);
}
#endif
@@ -575,7 +578,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
bool BlockInForceTSOValidRange = false;
auto InstForceTSOIt = ForceTSOInstructions.end();
if (ForceTSOValidRanges.Contains({Block.Entry, Block.Entry + Block.Size})) {
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); *It < Block.Entry + Block.Size) {
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); It != ForceTSOInstructions.end() && *It < Block.Entry + Block.Size) {
InstForceTSOIt = It;
BlockInForceTSOValidRange = true;
}
@@ -584,18 +587,16 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
// Set the block entry point
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
uint64_t BlockInstructionsLength {};
// Reset any block-specific state
Thread->OpDispatcher->StartNewBlock();
uint64_t InstsInBlock = Block.NumInstructions;
const uint64_t InstsInBlock = Block.NumInstructions;
if (InstsInBlock == 0) {
// Special case for an empty instruction block.
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry - GuestRIP));
}
uint64_t BlockInstructionsLength {};
for (size_t i = 0; i < InstsInBlock; ++i) {
uint64_t InstAddress = Block.Entry + BlockInstructionsLength;
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
@@ -121,7 +121,7 @@ void Dispatcher::EmitDispatcher() {
ldr(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
FillSpecialRegs(TMP1, TMP2, false, true);
FillSpecialRegs(TMP1, TMP2, {.SetFIZ = false, .SetPredRegs = true});
// As ARM64EC uses this as an entrypoint for both guest calls and host returns, opportunistically try to return
// using the call-ret stack to avoid unbalancing it.
@@ -357,7 +357,7 @@ void Dispatcher::EmitDispatcher() {
ldr(ARMEmitter::XReg::x4, &l_CompileSingleStep);
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
GenerateIndirectRuntimeCall<uintptr_t, void*, void*, uint64_t>(ARMEmitter::Reg::r4);
} else {
blr(ARMEmitter::Reg::r4); // { CTX, Frame, RIP }
}
+13 -5
View File
@@ -1091,13 +1091,21 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
if (ErrorDuringDecoding != DecodedBlockStatus::SUCCESS || HitNonExecutableRange || HitBadRelocation) [[unlikely]] {
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
// Error while decoding instruction. We don't know the table or instruction size
const auto InstSize = DecodeInst->InstSize;
DecodeInst->TableInfo = nullptr;
auto Result = ErrorDuringDecoding != DecodedBlockStatus::SUCCESS ? ErrorDuringDecoding :
DecodeInst->InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST :
HitNonExecutableRange ? DecodedBlockStatus::NOEXEC_INST :
DecodedBlockStatus::BAD_RELOCATION;
DecodeInst->InstSize = 0;
return Result;
// A decode error can be caused by substituting zero for an inaccessible
// instruction byte, so the instruction fetch fault takes priority.
if (HitNonExecutableRange) {
return InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST : DecodedBlockStatus::NOEXEC_INST;
}
if (HitBadRelocation) {
return DecodedBlockStatus::BAD_RELOCATION;
}
return ErrorDuringDecoding;
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
return DecodedBlockStatus::INVALID_INST;
+3 -6
View File
@@ -13,9 +13,6 @@ $end_info$
namespace FEXCore::CPU {
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
#define DEF_BINOP_WITH_CONSTANT(FEXOp, VarOp, ConstOp) \
DEF_OP(FEXOp) { \
auto Op = IROp->C<IR::IROp_##FEXOp>(); \
@@ -421,8 +418,8 @@ DEF_OP(MulH) {
if (OpSize == IR::OpSize::i32Bit) {
sxtw(TMP1, Src1.W());
sxtw(TMP2, Src2.W());
mul(ARMEmitter::Size::i32Bit, Dst, TMP1, TMP2);
ubfx(ARMEmitter::Size::i32Bit, Dst, Dst, 32, 32);
mul(ARMEmitter::Size::i64Bit, Dst, TMP1, TMP2);
ubfx(ARMEmitter::Size::i64Bit, Dst, Dst, 32, 32);
} else {
smulh(Dst.X(), Src1.X(), Src2.X());
}
@@ -774,7 +771,7 @@ DEF_OP(PDep) {
// Now, they're copied, so we can start setting Dest (even if it overlaps with
// one of them). Handle early exit case
mov(EmitSize, Dest, 0);
(void)cbz(EmitSize, OrigMask, &Done);
(void)cbz(EmitSize, Mask, &Done);
// Setup for first iteration
neg(EmitSize, T0, Mask);
@@ -342,8 +342,8 @@ DEF_OP(TelemetrySetValue) {
(void)Bind(&LoopTop);
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP4, TMP3, TMP2);
(void)cbnz(ARMEmitter::Size::i32Bit, TMP4, &LoopTop);
}
#endif
}
@@ -292,8 +292,6 @@ DEF_OP(Vector_FToS) {
frinti(SubEmitSize, Dst.Z(), Mask.Merging(), Vector.Z());
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Dst.Z(), SubEmitSize);
} else {
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector);
if (OpSize == IR::OpSize::i64Bit) {
frinti(SubEmitSize, Dst.D(), Vector.D());
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
@@ -324,24 +324,62 @@ DEF_OP(PCLMUL) {
const auto Op = IROp->C<IR::IROp_PCLMUL>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Src1 = GetVReg(Op->Src1);
const auto Src2 = GetVReg(Op->Src2);
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
switch (Op->Selector) {
case 0b00000000: pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D()); break;
case 0b00000001:
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
break;
case 0b00010000:
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
break;
case 0b00010001: pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q()); break;
default: LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector); break;
if (HostSupportsSVE256 && Is256Bit) {
switch (Op->Selector) {
case 0b00000000: {
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), Src2.Z());
break;
}
case 0b00000001: {
trn2(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), Src1.Z(), Src1.Z());
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), VTMP1.Z(), Src2.Z());
break;
}
case 0b00010000:
trn2(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), Src2.Z(), Src2.Z());
pmullb(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), VTMP1.Z());
break;
case 0b00010001: {
pmullt(ARMEmitter::SubRegSize::i128Bit, Dst.Z(), Src1.Z(), Src2.Z());
break;
}
default: {
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
break;
}
}
} else {
switch (Op->Selector) {
case 0b00000000: {
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), Src1.D(), Src2.D());
break;
}
case 0b00000001: {
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src1.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src2.D());
break;
}
case 0b00010000: {
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), Src2.Q(), 1);
pmull(ARMEmitter::SubRegSize::i128Bit, Dst.D(), VTMP1.D(), Src1.D());
break;
}
case 0b00010001: {
pmull2(ARMEmitter::SubRegSize::i128Bit, Dst.Q(), Src1.Q(), Src2.Q());
break;
}
default: {
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
break;
}
}
}
}
+13 -21
View File
@@ -622,7 +622,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES}
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP}
, CTX {ctx}
, TempAllocator(ctx->CPUBackendAllocator, 0) {
, TempCodeBufferAllocator(ctx->CPUBackendAllocator, 0) {
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
@@ -630,7 +630,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
RAPass->AddRegisters(IR::RegClass::GPRFixed, StaticRegisters.size());
RAPass->AddRegisters(IR::RegClass::FPR, GeneralFPRegisters.size());
RAPass->AddRegisters(IR::RegClass::FPRFixed, StaticFPRegisters.size());
RAPass->PairRegs = PairRegisters;
RAPass->SetNumPairRegs(PairRegisters);
{
// Set up pointers that the JIT needs to load
@@ -666,24 +666,17 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
Ptrs.LDIV = reinterpret_cast<uint64_t>(LDIV);
}
CurrentCodeBuffer = CodeBuffers.GetLatest();
CurrentCodeBuffer = SharedCodeBuffers.GetLatest();
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
}
void Arm64JITCore::EmitDetectionString() {
const char JITString[] = "FEXJIT::Arm64JITCore::";
EmitString(JITString);
Align();
}
void Arm64JITCore::ClearCache() {
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
auto PrevCodeBuffer = CurrentCodeBuffer;
auto lk = PrevCodeBuffer->LookupCache->AcquireWriteLock();
auto CodeBuffer = GetEmptyCodeBuffer();
auto CodeBuffer = GetEmptySharedCodeBuffer();
SetBuffer(CodeBuffer->Ptr, CodeBuffer->AllocatedSize);
EmitDetectionString();
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache, lk);
}
@@ -843,7 +836,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
case RestartOptions::Control::NeedsLargerJITSpace:
// Get rid of the claimed buffer immediately, we can't fit in it at all.
TempAllocator.UnclaimBuffer();
TempCodeBufferAllocator.UnclaimBuffer();
SSANodeMultiplier *= 2;
break;
default: LOGMAN_MSG_A_FMT("Unhandled Arm64 restart condition!");
@@ -864,7 +857,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
// This minimizes lock contention of CodeBufferWriteMutex.
auto TempCodeBufferInfo = TempAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
auto TempCodeBufferInfo = TempCodeBufferAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
auto TempCodeBuffer = TempCodeBufferInfo.Ptr;
const uint32_t UsableBufferRange = TempCodeBufferInfo.Size - FEXCore::Utils::FEX_PAGE_SIZE;
@@ -909,7 +902,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
PendingCallReturnTargetLabel = nullptr;
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
using namespace FEXCore::IR;
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
@@ -1080,7 +1072,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Migrate the compile output from temporary storage to the actual CodeBuffer.
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
{
auto CodeBufferLock = std::unique_lock {CodeBuffers.CodeBufferWriteMutex};
auto CodeBufferLock = std::unique_lock {SharedCodeBuffers.CodeBufferWriteMutex};
// Query size of generated code
const auto TempSize = GetCursorOffset();
@@ -1097,13 +1089,13 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
SetBuffer(CurrentCodeBuffer->Ptr, CurrentCodeBuffer->AllocatedSize);
SetCursorOffset(CodeBuffers.LatestOffset);
SetCursorOffset(SharedCodeBuffers.LatestOffset);
Align16B();
if ((GetCursorOffset() + TempSize) > CurrentCodeBuffer->UsableSize()) {
CTX->ClearCodeCache(ThreadState);
}
CodeBuffers.LatestOffset = GetCursorOffset();
SharedCodeBuffers.LatestOffset = GetCursorOffset();
}
// Adjust host addresses
@@ -1115,17 +1107,17 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
CodeBegin += Delta;
for (std::size_t Idx = PrevNumAllocations; Idx != Relocations.size(); ++Idx) {
Relocations[Idx].Header.Offset += CodeBuffers.LatestOffset;
Relocations[Idx].Header.Offset += SharedCodeBuffers.LatestOffset;
}
// Copy over CodeBuffer contents
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
SetCursorOffset(CodeBuffers.LatestOffset + TempSize);
SetCursorOffset(SharedCodeBuffers.LatestOffset + TempSize);
CodeBuffers.LatestOffset = GetCursorOffset();
SharedCodeBuffers.LatestOffset = GetCursorOffset();
}
TempAllocator.DelayedDisownBuffer();
TempCodeBufferAllocator.DelayedDisownBuffer();
ClearICache(CodeBegin, CodeOnlySize);
+1 -3
View File
@@ -105,7 +105,7 @@ private:
};
fextl::vector<PendingJumpThunk> PendingJumpThunks;
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempAllocator;
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempCodeBufferAllocator;
static uint64_t ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
@@ -526,8 +526,6 @@ private:
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
}
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
IR::RegisterAllocationPass* RAPass {};
FEXCore::Core::DebugData* DebugData {};
@@ -267,7 +267,7 @@ DEF_OP(LoadContextIndexed) {
ldr(Dst.Q(), TMP1, Op->BaseOffset);
} else {
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, Op->BaseOffset);
ldur(Dst.Q(), TMP1, Op->BaseOffset);
ldur(Dst.Q(), TMP1);
}
break;
case IR::OpSize::i256Bit:
@@ -333,7 +333,7 @@ DEF_OP(StoreContextIndexed) {
str(Value.Q(), TMP1, Op->BaseOffset);
} else {
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, Op->BaseOffset);
stur(Value.Q(), TMP1, Op->BaseOffset);
stur(Value.Q(), TMP1);
}
break;
case IR::OpSize::i256Bit:
@@ -2406,7 +2406,7 @@ DEF_OP(CacheLineClear) {
} else {
auto CurrentWorkingReg = MemReg.X();
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheLineSize); ++i) {
dc(ARMEmitter::DataCacheOperation::CIVAC, TMP1);
dc(ARMEmitter::DataCacheOperation::CIVAC, CurrentWorkingReg);
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
CurrentWorkingReg = TMP1;
}
@@ -2435,7 +2435,7 @@ DEF_OP(CacheLineClean) {
} else {
auto CurrentWorkingReg = MemReg.X();
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheLineSize); ++i) {
dc(ARMEmitter::DataCacheOperation::CVAC, TMP1);
dc(ARMEmitter::DataCacheOperation::CVAC, CurrentWorkingReg);
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
CurrentWorkingReg = TMP1;
}
@@ -168,7 +168,7 @@ DEF_OP(PushRoundingMode) {
} else {
LOGMAN_THROW_A_FMT(Op->RoundMode == 1 || Op->RoundMode == 2, "expect a valid round mode");
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(Op->RoundMode << 22));
and_(ARMEmitter::Size::i64Bit, TMP1, Dest, ~(3 << 22));
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, (Op->RoundMode == 2 ? 1 : 2) << 22);
}
@@ -282,7 +282,7 @@ DEF_OP(ProcessorID) {
// Load the values returned by the kernel
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::WReg::w0, ARMEmitter::WReg::w1, ARMEmitter::Reg::rsp);
// Deallocate stack space
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
// Now that we are done in the syscall we need to carefully peel back the state
// First unspill the registers from before
+161 -121
View File
@@ -1139,9 +1139,16 @@ DEF_OP(VOrn) {
const auto Vector2 = GetVReg(Op->Vector2);
if (HostSupportsSVE256 && Is256Bit) {
const auto Pred = PRED_TMP_32B.Merging();
not_(ARMEmitter::SubRegSize::i8Bit, VTMP1.Z(), Pred, Vector2.Z());
orr(Dst.Z(), Vector1.Z(), VTMP1.Z());
if (Dst == Vector1) {
bsl2n(Dst.Z(), Dst.Z(), Vector2.Z(), Dst.Z());
} else if (Dst == Vector2) {
const auto Pred = PRED_TMP_32B.Merging();
not_(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Pred, Dst.Z());
orr(Dst.Z(), Vector1.Z(), Dst.Z());
} else {
movprfx(Dst.Z(), Vector1.Z());
bsl2n(Dst.Z(), Dst.Z(), Vector2.Z(), Vector1.Z());
}
} else if (Is128Bit) {
orn(Dst.Q(), Vector1.Q(), Vector2.Q());
} else {
@@ -1165,8 +1172,7 @@ DEF_OP(VFAddV) {
if (HostSupportsSVE256 && Is256Bit) {
const auto Pred = PRED_TMP_32B.Merging();
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
}
if (HostSupportsSVE128) {
} else if (HostSupportsSVE128) {
const auto Pred = PRED_TMP_16B.Merging();
faddv(SubRegSize.Vector, Dst, Pred, Vector.Z());
} else {
@@ -1193,20 +1199,16 @@ DEF_OP(VAddV) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
// SVE doesn't have an equivalent ADDV instruction, so we make do
// by performing two Adv. SIMD ADDV operations on the high and low
// 128-bit lanes and then sum them up.
const auto Mask = PRED_TMP_32B.Zeroing();
const auto CompactPred = ARMEmitter::PReg::p0;
// Select all our upper elements to run ADDV over them.
not_(CompactPred, Mask, PRED_TMP_16B);
compact(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), CompactPred, Vector.Z());
addv(SubRegSize.Vector, VTMP2.Q(), Vector.Q());
addv(SubRegSize.Vector, VTMP1.Q(), VTMP1.Q());
add(SubRegSize.Vector, Dst.Q(), VTMP1.Q(), VTMP2.Q());
if (ElementSize == IR::OpSize::i64Bit) {
const auto Mask = PRED_TMP_32B.Zeroing();
uaddv(SubRegSize.Vector, Dst.D(), Mask, Vector.Z());
} else {
const auto Mask = ARMEmitter::PReg::p0;
uaddv(SubRegSize.Vector, VTMP1.D(), Mask, Vector.Z());
mov_imm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), 0);
ptrue(SubRegSize.Vector, Mask, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Mask.Merging(), VTMP1.Z());
}
} else {
if (ElementSize == IR::OpSize::i64Bit) {
addp(SubRegSize.Scalar, Dst, Vector);
@@ -2785,17 +2787,17 @@ DEF_OP(VUShrSWide) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
if (ElementSize == IR::OpSize::i64Bit) {
const auto Mask = PRED_TMP_32B.Merging();
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
} else {
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
lsr_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
}
} else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging();
@@ -2851,17 +2853,17 @@ DEF_OP(VSShrSWide) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
if (ElementSize == IR::OpSize::i64Bit) {
const auto Mask = PRED_TMP_32B.Merging();
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
} else {
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
asr_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
}
} else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging();
@@ -2917,17 +2919,17 @@ DEF_OP(VUShlSWide) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
if (ElementSize == IR::OpSize::i64Bit) {
const auto Mask = PRED_TMP_32B.Merging();
if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z());
}
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
} else {
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
lsl_wide(SubRegSize, Dst.Z(), Vector.Z(), VTMP1.Z());
}
} else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging();
@@ -3175,19 +3177,12 @@ DEF_OP(VUShrI) {
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
} else {
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
if (BitShift == 0) {
if (Dst != Vector) {
mov(Dst.Z(), Vector.Z());
}
} else {
// SVE LSR is destructive, so lets set up the destination if
// Vector doesn't already alias it.
if (Dst != Vector) {
movprfx(Dst.Z(), Vector.Z());
}
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), BitShift);
lsr(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
}
} else {
if (BitShift == 0) {
@@ -3201,48 +3196,6 @@ DEF_OP(VUShrI) {
}
}
DEF_OP(VUShraI) {
const auto Op = IROp->C<IR::IROp_VUShraI>();
const auto OpSize = IROp->Size;
const auto BitShift = Op->BitShift;
const auto SubRegSize = ConvertSubRegSize8(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto DestVector = GetVReg(Op->DestVector);
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
if (Dst == DestVector) {
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
} else {
if (Dst != Vector) {
mov(Dst.Z(), DestVector.Z());
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
} else {
mov(VTMP1.Z(), DestVector.Z());
usra(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
mov(Dst.Z(), VTMP1.Z());
}
}
} else {
if (Dst == DestVector) {
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
} else {
if (Dst != Vector) {
mov(Dst.Q(), DestVector.Q());
usra(SubRegSize, Dst.Q(), Vector.Q(), BitShift);
} else {
mov(VTMP1.Q(), DestVector.Q());
usra(SubRegSize, VTMP1.Q(), Vector.Q(), BitShift);
mov(Dst.Q(), VTMP1.Q());
}
}
}
}
DEF_OP(VSShrI) {
const auto Op = IROp->C<IR::IROp_VSShrI>();
const auto OpSize = IROp->Size;
@@ -3258,19 +3211,12 @@ DEF_OP(VSShrI) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
if (Shift == 0) {
if (Dst != Vector) {
mov(Dst.Z(), Vector.Z());
}
} else {
// SVE ASR is destructive, so lets set up the destination if
// Vector doesn't already alias it.
if (Dst != Vector) {
movprfx(Dst.Z(), Vector.Z());
}
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), Shift);
asr(SubRegSize, Dst.Z(), Vector.Z(), Shift);
}
} else {
if (Shift == 0) {
@@ -3300,19 +3246,12 @@ DEF_OP(VShlI) {
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
} else {
if (HostSupportsSVE256 && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
if (BitShift == 0) {
if (Dst != Vector) {
mov(Dst.Z(), Vector.Z());
}
} else {
// SVE LSL is destructive, so lets set up the destination if
// Vector doesn't already alias it.
if (Dst != Vector) {
movprfx(Dst.Z(), Vector.Z());
}
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), BitShift);
lsl(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
}
} else {
if (BitShift == 0) {
@@ -3339,8 +3278,13 @@ DEF_OP(VUShrNI) {
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
if (BitShift == 0) {
mov_imm(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), 0);
uzp1(SubRegSize, Dst.Z(), Dst.Z(), VTMP1.Z());
} else {
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
uzp1(SubRegSize, Dst.Z(), Dst.Z(), Dst.Z());
}
} else {
if (BitShift == 0) {
xtn(SubRegSize, Dst.D(), Vector.D());
@@ -3593,9 +3537,13 @@ DEF_OP(VSQXTN2) {
mov(Dst.Q(), VectorLower.Q());
ins(ARMEmitter::SubRegSize::i32Bit, Dst, 1, VTMP2, 0);
} else {
mov(VTMP1.Q(), VectorLower.Q());
sqxtn2(SubRegSize, VTMP1, VectorUpper);
mov(Dst.Q(), VTMP1.Q());
if (Dst == VectorLower) {
sqxtn2(SubRegSize, VectorLower, VectorUpper);
} else {
mov(VTMP1.Q(), VectorLower.Q());
sqxtn2(SubRegSize, VTMP1, VectorUpper);
mov(Dst.Q(), VTMP1.Q());
}
}
}
}
@@ -4451,13 +4399,9 @@ DEF_OP(VFMLS) {
if (Is128Bit) {
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
} else {
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
}
if (Is128Bit) {
fmla(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
} else {
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
fmla(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
}
@@ -4611,13 +4555,9 @@ DEF_OP(VFNMLS) {
if (Is128Bit) {
fneg(SubRegSize, DestTmp.Q(), VectorAddend.Q());
} else {
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
}
if (Is128Bit) {
fmls(SubRegSize, DestTmp.Q(), Vector1.Q(), Vector2.Q());
} else {
fneg(SubRegSize, DestTmp.D(), VectorAddend.D());
fmls(SubRegSize, DestTmp.D(), Vector1.D(), Vector2.D());
}
@@ -4631,6 +4571,106 @@ DEF_OP(VFNMLS) {
}
}
DEF_OP(VBlendImm) {
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "Host must support SVE to use {}", __func__);
auto Op = IROp->C<IR::IROp_VBlendImm>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
const auto SubRegSize = ConvertSubRegSize8(IROp);
const auto ElementSize = IROp->ElementSize;
const auto Selector = Op->Selector;
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
const auto Dst = GetVReg(Node);
const auto LHS = GetVReg(Op->LHS);
const auto RHS = GetVReg(Op->RHS);
const auto DstIsNonAliasing = Dst != LHS && Dst != RHS;
// Silly case where two blending sources are the same.
if (LHS == RHS) {
if (DstIsNonAliasing) {
mov(SubRegSize, Dst.Z(), GoverningPredicate.Merging(), LHS.Z());
}
return;
}
// We'll need to expand our selector to match its predicate equivalent.
// The lowest bit of each predicate element being set to 1 signifies
// that it's enabled.
const auto MakePredicateMask = [ElementSize, Is256Bit, OpSize](uint16_t Imm) {
if (ElementSize == IR::OpSize::i8Bit) {
// Since we use a u16 selector, we have enough bits for every byte in a
// 128-bit lane, so we don't need to do anything here except replicate the
// bits in the event of 256-bit.
return Is256Bit ? uint32_t(Imm) << 16 | Imm : Imm;
}
uint32_t Mask = 0;
const auto DataSize = IR::OpSizeToSize(ElementSize);
const auto NumElements = IR::NumElements(OpSize, ElementSize);
for (uint32_t i = 0; i < NumElements; i++) {
if (((Imm >> i) & 1) != 0) {
Mask |= 1U << (DataSize * i);
}
}
return Mask;
};
// Our predicate that we'll be firing our constructed bitmask into.
constexpr auto Predicate = ARMEmitter::PReg::p0.Merging();
// TODO: We can completely eliminate this via PMOV in SVE2.1
ARMEmitter::ForwardLabel AfterLabel;
ARMEmitter::BackwardLabel ConstantLabel;
(void)b(&AfterLabel);
(void)Bind(&ConstantLabel);
const auto PredicateMask = MakePredicateMask(Selector);
if (Dst == RHS) {
dc32(~PredicateMask);
} else {
dc32(PredicateMask);
}
(void)Bind(&AfterLabel);
(void)adr(TMP1, &ConstantLabel);
ldr(Predicate, TMP1);
if (Dst == LHS) {
mov(SubRegSize, LHS.Z(), Predicate, RHS.Z());
} else if (Dst == RHS) {
mov(SubRegSize, RHS.Z(), Predicate, LHS.Z());
} else {
mov(SubRegSize, Dst.Z(), GoverningPredicate.Merging(), LHS.Z());
mov(SubRegSize, Dst.Z(), Predicate, RHS.Z());
}
}
DEF_OP(VXar) {
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "Host must support SVE to use {}", __func__);
auto Op = IROp->C<IR::IROp_VXar>();
const auto SubRegSize = ConvertSubRegSize8(IROp);
const auto ElementSizeBits = IR::OpSizeAsBits(IROp->ElementSize);
const auto Dst = GetVReg(Node);
const auto LHS = GetVReg(Op->LHS);
const auto RHS = GetVReg(Op->RHS);
const auto Rotate = Op->Rotate;
LOGMAN_THROW_A_FMT(Rotate >= 1 && Rotate <= ElementSizeBits, "Rotate immediate must be within [1, {}]", ElementSizeBits);
if (Dst == LHS) {
xar(SubRegSize, Dst.Z(), RHS.Z(), Rotate);
} else if (Dst == RHS) {
movprfx(VTMP1.Z(), LHS.Z());
xar(SubRegSize, VTMP1.Z(), RHS.Z(), Rotate);
mov(Dst.Z(), VTMP1.Z());
} else {
movprfx(Dst.Z(), LHS.Z());
xar(SubRegSize, Dst.Z(), RHS.Z(), Rotate);
}
}
DEF_OP(VFCopySign) {
auto Op = IROp->C<IR::IROp_VFCopySign>();
const auto OpSize = IROp->Size;
@@ -1689,8 +1689,8 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
}
void OpDispatchBuilder::ANDNBMIOp(OpcodeArgs) {
auto* Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto* Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
auto Dest = _Andn(OpSizeFromSrc(Op), Src2, Src1);
@@ -1703,8 +1703,8 @@ void OpDispatchBuilder::BEXTRBMIOp(OpcodeArgs) {
// along with some edge-case handling and flag setting.
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
auto* Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto* Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
const auto Size = OpSizeFromSrc(Op);
const auto SrcSize = IR::OpSizeAsBits(Size);
@@ -1746,7 +1746,7 @@ void OpDispatchBuilder::BLSIBMIOp(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
const auto Size = OpSizeFromSrc(Op);
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto NegatedSrc = _Neg(Size, Src);
auto Result = _And(Size, Src, NegatedSrc);
@@ -1767,7 +1767,7 @@ void OpDispatchBuilder::BLSMSKBMIOp(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
const auto Size = OpSizeFromSrc(Op);
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Result = _Xor(Size, Sub(Size, Src, 1), Src);
StoreResultGPR(Op, Result);
@@ -1787,7 +1787,7 @@ void OpDispatchBuilder::BLSRBMIOp(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
const auto Size = OpSizeFromSrc(Op);
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Result = _And(Size, Sub(Size, Src, 1), Src);
StoreResultGPR(Op, Result);
@@ -1807,8 +1807,8 @@ void OpDispatchBuilder::BMI2Shift(OpcodeArgs) {
const auto Size = OpSizeFromSrc(Op);
const auto SrcSize = Op->Src[0].IsGPR() ? GPRSize : Size;
auto* Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
auto* Shift = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GPRSize, Op->Flags, {.AllowUpperGarbage = true});
auto Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
auto Shift = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GPRSize, Op->Flags, {.AllowUpperGarbage = true});
Ref Result;
if (Op->OP == 0x6F7) {
@@ -1831,9 +1831,9 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) {
// In 32-bit mode we only look at bottom 32-bit, no 8 or 16-bit BZHI so no
// need to zero-extend sources
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto* Index = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
auto Index = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
// Clear the high bits specified by the index. A64 only considers bottom bits
// of the shift, so we don't need to mask bottom 8-bits ourselves.
@@ -1878,8 +1878,8 @@ void OpDispatchBuilder::RORX(OpcodeArgs) {
return;
}
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto* Result = Src;
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Result = Src;
if (DoRotation) [[likely]] {
Result = _Ror(OpSizeFromSrc(Op), Src, _InlineConstant(Amount));
}
@@ -1916,8 +1916,8 @@ void OpDispatchBuilder::MULX(OpcodeArgs) {
void OpDispatchBuilder::PDEP(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
auto* Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto* Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
auto Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
auto Result = _PDep(OpSizeFromSrc(Op), Input, Mask);
StoreResultGPR(Op, Op->Dest, Result);
@@ -1925,8 +1925,8 @@ void OpDispatchBuilder::PDEP(OpcodeArgs) {
void OpDispatchBuilder::PEXT(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed");
auto* Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto* Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
auto Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
auto Result = _PExt(OpSizeFromSrc(Op), Input, Mask);
StoreResultGPR(Op, Op->Dest, Result);
@@ -1936,8 +1936,8 @@ void OpDispatchBuilder::ADXOp(OpcodeArgs) {
const auto OpSize = OpSizeFromSrc(Op);
// Only 32/64-bit anyway so allow garbage, we use 32-bit ops.
auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto* Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
auto Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
auto Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
// Handles ADCX and ADOX
const bool IsADCX = Op->OP == 0x1F6;
@@ -3877,7 +3877,6 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
// This allows us to only hit the ZEXT case on failure
Ref RAXResult = NZCVSelect(OpSize::i64Bit, CondClass::EQ, Src3, Src1Lower);
// When the size is 4 we need to make sure not zext the GPR when the comparison fails
StoreGPRRegister(X86State::REG_RAX, RAXResult);
} else {
StoreGPRRegister(X86State::REG_RAX, Src1Lower, Size);
@@ -3891,7 +3890,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
Src2Lower = _Bfe(GPRSize, IR::OpSizeAsBits(Size), 0, Src2);
}
Ref DestResult = Trivial ? Src2 : NZCVSelect(OpSize::i64Bit, CondClass::EQ, Src2Lower, Src1);
Ref DestResult = Trivial ? Src2Lower : NZCVSelect(OpSize::i64Bit, CondClass::EQ, Src2Lower, Src1);
// Store in to GPR Dest
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
@@ -5038,7 +5037,11 @@ void OpDispatchBuilder::RDTSCPOp(OpcodeArgs) {
// - Explicitly use an MFENCE before this instruction if you want this behaviour
// This instruction is not an execution fence, so subsequent instructions can execute after this
// - Explicitly use an LFENCE after RDTSCP if you want to block this behaviour
if (CTX->HostFeatures.HostType != FEXCore::HostFeatures::HostTypeEnum::Linux && !CTX->HostFeatures.SupportsCPUIndexInTPIDRRO) {
// RDTSCP is unsupported on Win32 platforms if TPIDRRO isn't supported.
UnimplementedOp(Op);
return;
}
auto Counter = CycleCounter(true);
auto ID = _ProcessorID();
@@ -5048,6 +5051,11 @@ void OpDispatchBuilder::RDTSCPOp(OpcodeArgs) {
}
void OpDispatchBuilder::RDPIDOp(OpcodeArgs) {
if (CTX->HostFeatures.HostType != FEXCore::HostFeatures::HostTypeEnum::Linux && !CTX->HostFeatures.SupportsCPUIndexInTPIDRRO) {
// RDTSCP is unsupported on Win32 platforms if TPIDRRO isn't supported.
UnimplementedOp(Op);
return;
}
StoreResultGPR(Op, _ProcessorID());
}
@@ -839,12 +839,12 @@ public:
void SHA256MSG2Op(OpcodeArgs);
void SHA256RNDS2Op(OpcodeArgs);
void AESImcOp(OpcodeArgs);
void AESImcOp(OpcodeArgs, bool IsAVX);
void AESEncOp(OpcodeArgs);
void AESEncLastOp(OpcodeArgs);
void AESDecOp(OpcodeArgs);
void AESDecLastOp(OpcodeArgs);
void AESKeyGenAssist(OpcodeArgs);
void AESKeyGenAssist(OpcodeArgs, bool IsAVX);
void VFMAImpl(OpcodeArgs, IROps IROp, bool Scalar, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
void VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
@@ -603,7 +603,7 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), OpSize::i128Bit,
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return _VAndn(OpSize::i128Bit, _ElementSize, Src2, Src1); });
[this](IR::OpSize, Ref Src1, Ref Src2) { return _VAndn(OpSize::i128Bit, Src2, Src1); });
}
void OpDispatchBuilder::AVX128_VPACKSS(OpcodeArgs, IR::OpSize ElementSize) {
@@ -865,13 +865,14 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
GPR = Mask4Byte(Src.Low);
}
} else if (ElementSize == OpSize::i32Bit) {
auto GPRLow = Mask4Byte(Src.Low);
auto GPRHigh = Mask4Byte(Src.High);
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 4);
Ref Fused = _VUnZip2(OpSize::i128Bit, OpSize::i16Bit, Src.Low, Src.High);
Fused = _VUShrI(OpSize::i128Bit, OpSize::i16Bit, Fused, 15);
auto ConstantUSHL = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_INCREMENTAL_U16_INDEX);
Fused = _VUShl(OpSize::i128Bit, OpSize::i16Bit, Fused, ConstantUSHL, false);
Fused = _VAddV(OpSize::i128Bit, OpSize::i16Bit, Fused);
GPR = _VExtractToGPR(OpSize::i128Bit, OpSize::i16Bit, Fused, 0);
} else {
auto GPRLow = Mask8Byte(Src.Low);
auto GPRHigh = Mask8Byte(Src.High);
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 2);
GPR = Mask4Byte(_VUnZip2(OpSize::i128Bit, OpSize::i32Bit, Src.Low, Src.High));
}
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
}
@@ -885,7 +886,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
auto Mask1Byte = [this](Ref Src, Ref VMask) {
auto VCMP = _VCMPLTZ(OpSize::i128Bit, OpSize::i8Bit, Src);
auto VAnd = _VAnd(OpSize::i128Bit, OpSize::i8Bit, VCMP, VMask);
auto VAnd = _VAnd(OpSize::i128Bit, VCMP, VMask);
auto VAdd1 = _VAddP(OpSize::i128Bit, OpSize::i8Bit, VAnd, VAnd);
auto VAdd2 = _VAddP(OpSize::i128Bit, OpSize::i8Bit, VAdd1, VAdd1);
@@ -1729,8 +1730,8 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize) {
{
// Calculate ZF first.
auto AndLow = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low);
auto AndHigh = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High);
auto AndLow = _VAnd(OpSize::i128Bit, Src2.Low, Src1.Low);
auto AndHigh = _VAnd(OpSize::i128Bit, Src2.High, Src1.High);
auto ShiftLow = _VUShrI(OpSize::i128Bit, ElementSize, AndLow, ElementSizeInBits - 1);
auto ShiftHigh = _VUShrI(OpSize::i128Bit, ElementSize, AndHigh, ElementSizeInBits - 1);
@@ -1749,8 +1750,8 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize) {
{
// Calculate CF Second
auto AndLow = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low);
auto AndHigh = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High);
auto AndLow = _VAndn(OpSize::i128Bit, Src2.Low, Src1.Low);
auto AndHigh = _VAndn(OpSize::i128Bit, Src2.High, Src1.High);
auto ShiftLow = _VUShrI(OpSize::i128Bit, ElementSize, AndLow, ElementSizeInBits - 1);
auto ShiftHigh = _VUShrI(OpSize::i128Bit, ElementSize, AndHigh, ElementSizeInBits - 1);
@@ -1788,11 +1789,11 @@ void OpDispatchBuilder::AVX128_PTest(OpcodeArgs) {
}
// For 256-bit, we need to unroll. This is nontrivial.
Ref Test1Low = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src1.Low, Src2.Low);
Ref Test2Low = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low);
Ref Test1Low = _VAnd(OpSize::i128Bit, Src1.Low, Src2.Low);
Ref Test2Low = _VAndn(OpSize::i128Bit, Src2.Low, Src1.Low);
Ref Test1High = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src1.High, Src2.High);
Ref Test2High = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High);
Ref Test1High = _VAnd(OpSize::i128Bit, Src1.High, Src2.High);
Ref Test2High = _VAndn(OpSize::i128Bit, Src2.High, Src1.High);
// Element size must be less than 32-bit for the sign bit tricks.
Ref Test1Max = _VUMax(OpSize::i128Bit, OpSize::i16Bit, Test1Low, Test1High);
@@ -2009,13 +2010,13 @@ void OpDispatchBuilder::AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Sr
ConstantEOR = LoadAndCacheNamedVectorConstant(
OpSize::i128Bit, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PSUBADDPS_INVERT : NAMED_VECTOR_PSUBADDPD_INVERT);
}
auto InvertedSourceLow = _VXor(OpSize::i128Bit, ElementSize, Sources[AddendIdx - 1].Low, ConstantEOR);
auto InvertedSourceLow = _VXor(OpSize::i128Bit, Sources[AddendIdx - 1].Low, ConstantEOR);
Result.Low = _VFMLA(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].Low, Sources[Src2Idx - 1].Low, InvertedSourceLow);
if (Is128Bit) {
Result.High = LoadZeroVector(OpSize::i128Bit);
} else {
auto InvertedSourceHigh = _VXor(OpSize::i128Bit, ElementSize, Sources[AddendIdx - 1].High, ConstantEOR);
auto InvertedSourceHigh = _VXor(OpSize::i128Bit, Sources[AddendIdx - 1].High, ConstantEOR);
Result.High = _VFMLA(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].High, Sources[Src2Idx - 1].High, InvertedSourceHigh);
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
@@ -26,17 +26,25 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
auto RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
Ref Result {};
if (CTX->HostFeatures.SupportsSVE128) {
auto ZeroVec = LoadZeroVector(OpSize::i128Bit);
auto Tmp = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, ZeroVec, Dest);
auto Xar = _VXar(OpSize::i128Bit, OpSize::i32Bit, ZeroVec, Tmp, 2);
Result = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, Xar);
} else {
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
// Move the element to zero, rotate, and then move back (Using duplicates).
// Saves one instruction versus that path that doesn't support SHA extension.
auto Duplicated = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
auto Sha1HRotated = _VSha1H(Duplicated);
auto RotatedNode = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Sha1HRotated, 0);
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
}
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
@@ -50,9 +58,9 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
Ref Result = _VXor(OpSize::i128Bit, Dest, NewVec);
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
@@ -70,7 +78,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
// The result is swizzled differently than expected
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
@@ -99,7 +107,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
break;
}
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
const auto ZeroRegister = LoadZeroVector(OpSize::i128Bit);
Ref Src1 = SHADataShuffle(Dest);
Ref Src2 = SHADataShuffle(Src);
@@ -112,7 +120,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
}
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
@@ -125,7 +133,7 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
auto Result = _VSha256U0(Dest, Src);
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
@@ -142,7 +150,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
auto Result = _VSha256U1(Src1, Src2);
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
@@ -177,17 +185,22 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
auto B = _VSha256H2(EFGH, ABCD, Key);
auto Result = shuffle_abcd(A, B);
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
void OpDispatchBuilder::AESImcOp(OpcodeArgs, bool IsAVX) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESImc(Src);
StoreResultFPR(Op, Result);
if (IsAVX) {
StoreResultFPR(Op, Result);
} else {
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
}
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
@@ -198,19 +211,30 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESENC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
const auto Is256Bit = DstSize == OpSize::i256Bit;
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
Ref ZeroVec = LoadZeroVector(DstSize);
Ref Result {};
if (Is256Bit) {
// TODO: Handle as one operation once vixl supports it.
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
auto Lower = _VAESEnc(OpSize::i128Bit, State, Key, ZeroVec);
auto Upper = _VAESEnc(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
} else {
Result = _VAESEnc(DstSize, State, Key, ZeroVec);
}
StoreResultFPR(Op, Result);
}
@@ -223,19 +247,30 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESENCLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
const auto Is256Bit = DstSize == OpSize::i256Bit;
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
Ref ZeroVec = LoadZeroVector(DstSize);
Ref Result {};
if (Is256Bit) {
// TODO: Handle as one operation once vixl supports it.
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
auto Lower = _VAESEncLast(OpSize::i128Bit, State, Key, ZeroVec);
auto Upper = _VAESEncLast(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
} else {
Result = _VAESEncLast(DstSize, State, Key, ZeroVec);
}
StoreResultFPR(Op, Result);
}
@@ -248,19 +283,30 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESDEC.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
const auto Is256Bit = DstSize == OpSize::i256Bit;
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
Ref ZeroVec = LoadZeroVector(DstSize);
Ref Result {};
if (Is256Bit) {
// TODO: Handle as one operation once vixl supports it.
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
auto Lower = _VAESDec(OpSize::i128Bit, State, Key, ZeroVec);
auto Upper = _VAESDec(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
} else {
Result = _VAESDec(DstSize, State, Key, ZeroVec);
}
StoreResultFPR(Op, Result);
}
@@ -273,19 +319,30 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
StoreResultFPR(Op, Result);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
const auto DstSize = OpSizeFromDst(Op);
const auto Is128Bit = DstSize == OpSize::i128Bit;
// TODO: Handle 256-bit VAESDECLAST.
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
const auto Is256Bit = DstSize == OpSize::i256Bit;
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
Ref ZeroVec = LoadZeroVector(DstSize);
Ref Result {};
if (Is256Bit) {
// TODO: Handle as one operation once vixl supports it.
auto UpperState = _VDupElement(DstSize, OpSize::i128Bit, State, 1);
auto UpperKey = _VDupElement(DstSize, OpSize::i128Bit, Key, 1);
auto Lower = _VAESDecLast(OpSize::i128Bit, State, Key, ZeroVec);
auto Upper = _VAESDecLast(OpSize::i128Bit, UpperState, UpperKey, ZeroVec);
Result = _VInsElement(DstSize, OpSize::i128Bit, 1, 0, Lower, Upper);
} else {
Result = _VAESDecLast(DstSize, State, Key, ZeroVec);
}
StoreResultFPR(Op, Result);
}
@@ -298,14 +355,19 @@ Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
return _VAESKeyGenAssist(Src, KeyGenSwizzle, LoadZeroVector(OpSize::i128Bit), RCON);
}
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs, bool IsAVX) {
if (!CTX->HostFeatures.SupportsAES) {
UnimplementedOp(Op);
return;
}
Ref Result = AESKeyGenAssistImpl(Op);
StoreResultFPR(Op, Result);
if (IsAVX) {
StoreResultFPR(Op, Result);
} else {
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
}
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
@@ -317,8 +379,8 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
StoreResultFPR(Op, Res);
auto Result = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
@@ -79,7 +79,7 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
{OPD(PF_38_NONE, 0xCC), 1, &OpDispatchBuilder::SHA256MSG1Op},
{OPD(PF_38_NONE, 0xCD), 1, &OpDispatchBuilder::SHA256MSG2Op},
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESImcOp, false>},
{OPD(PF_38_66, 0xDC), 1, &OpDispatchBuilder::AESEncOp},
{OPD(PF_38_66, 0xDD), 1, &OpDispatchBuilder::AESEncLastOp},
{OPD(PF_38_66, 0xDE), 1, &OpDispatchBuilder::AESDecOp},
@@ -37,7 +37,7 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRIOp, false>},
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESKeyGenAssist, false>},
};
return std::to_array(Table);
@@ -366,13 +366,13 @@ Ref OpDispatchBuilder::VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp,
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) {
const auto DstSize = GetGuestVectorLength();
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false);
auto Result = VectorScalarUnaryInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false);
StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize);
}
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) {
const auto DstSize = GetGuestVectorLength();
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true);
auto Result = VectorScalarUnaryInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true);
StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize);
}
@@ -523,14 +523,14 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
case VectorCompareType::NLT_US: // NGT(Swapped operand)
case VectorCompareType::NLT_UQ: {
Ref Result = _VFCMPLT(ElementSize, ElementSize, Src1, Src2);
Result = _VNot(ElementSize, ElementSize, Result);
Result = _VNot(ElementSize, Result);
// Insert the lower bits
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
}
case VectorCompareType::NLE_US: // NGE(Swapped operand)
case VectorCompareType::NLE_UQ: {
Ref Result = _VFCMPLE(ElementSize, ElementSize, Src1, Src2);
Result = _VNot(ElementSize, ElementSize, Result);
Result = _VNot(ElementSize, Result);
// Insert the lower bits
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
}
@@ -539,14 +539,14 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
case VectorCompareType::NGT_UQ:
case VectorCompareType::NGT_US: {
Ref Result = _VFCMPLT(ElementSize, ElementSize, Src2, Src1);
Result = _VNot(ElementSize, ElementSize, Result);
Result = _VNot(ElementSize, Result);
// Insert the lower bits
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
}
case VectorCompareType::NGE_UQ:
case VectorCompareType::NGE_US: {
Ref Result = _VFCMPLE(ElementSize, ElementSize, Src2, Src1);
Result = _VNot(ElementSize, ElementSize, Result);
Result = _VNot(ElementSize, Result);
// Insert the lower bits
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
}
@@ -567,10 +567,10 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
// If either of the sources are unordered, then returns true.
Ref Src1_U = _VFCMPEQ(Size, ElementSize, Src1, Src1);
Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2);
auto Ordered = _VAnd(Size, ElementSize, Src1_U, Src2_U);
auto Ordered = _VAnd(Size, Src1_U, Src2_U);
Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2);
Ref Result = _VOrn(Size, ElementSize, Compare_Ordered, Ordered);
Ref Result = _VOrn(Size, Compare_Ordered, Ordered);
// Insert the lower bits
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
@@ -582,8 +582,8 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2);
Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2);
Ref Result = _VAndn(Size, ElementSize, Src1_U, Compare_Ordered);
Result = _VAnd(Size, ElementSize, Result, Src2_U);
Ref Result = _VAndn(Size, Src1_U, Compare_Ordered);
Result = _VAnd(Size, Result, Src2_U);
// Insert the lower bits
return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result);
@@ -598,14 +598,14 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize,
}
void OpDispatchBuilder::InsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize) {
const uint8_t CompType = Op->Src[1].Literal();
const uint8_t CompType = Op->Src[1].Literal() & 0b111;
const auto DstSize = GetGuestVectorLength();
const auto SrcSize = OpSizeFromSrc(Op);
Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Dest, DstSize, Op->Flags);
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType & 0b111, false);
Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType, false);
// ARM doesn't have any instructions that handle the semantics of NLT and NLE directly.
// In fact, these are the two SSE compatison types where we cannot use VFCMPScalarInsert
@@ -623,7 +623,7 @@ void OpDispatchBuilder::InsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize) {
}
void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize) {
const uint8_t CompType = Op->Src[2].Literal();
const uint8_t CompType = Op->Src[2].Literal() & 0b11111;
const auto DstSize = GetGuestVectorLength();
const auto SrcSize = OpSizeFromSrc(Op);
@@ -633,7 +633,7 @@ void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize
Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], DstSize, Op->Flags);
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType & 0b11111, true);
Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType, true);
StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize);
}
@@ -789,7 +789,7 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
Ref VMask = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_MOVMASKB);
auto VCMP = _VCMPLTZ(SrcSize, OpSize::i8Bit, Src);
auto VAnd = _VAnd(SrcSize, OpSize::i8Bit, VCMP, VMask);
auto VAnd = _VAnd(SrcSize, VCMP, VMask);
// Since we also handle the MM MOVMSKB here too,
// we need to clamp the lower bound.
@@ -884,7 +884,7 @@ Ref OpDispatchBuilder::PSHUFBOpImpl(IR::OpSize SrcSize, Ref Src1, Ref Src2, Ref
// the lane splitting behavior, so cap the maximum size at 16.
const auto SanitizedSrcSize = std::min(SrcSize, OpSize::i128Bit);
Ref MaskedIndices = _VAnd(SrcSize, SrcSize, Src2, MaskVector);
Ref MaskedIndices = _VAnd(SrcSize, Src2, MaskVector);
Ref Low = _VTBL1(SanitizedSrcSize, Src1, MaskedIndices);
if (!Is256Bit) {
@@ -1999,7 +1999,7 @@ void OpDispatchBuilder::VANDNOp(OpcodeArgs) {
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref Dest = _VAndn(SrcSize, SrcSize, Src2, Src1);
Ref Dest = _VAndn(SrcSize, Src2, Src1);
StoreResultFPR(Op, Dest);
}
@@ -2581,9 +2581,10 @@ Ref OpDispatchBuilder::CVTGPR_To_FPRImpl(OpcodeArgs, IR::OpSize DstElementSize,
}
Ref OpDispatchBuilder::CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElementSize, bool HostRoundingMode) {
// GPR size is determined by REX.W
// Source Element size is determined by instruction
const auto GPRSize = OpSizeFromDst(Op);
// GPR size is determined by REX.W
// But instruction does not support 16bit register operands
const auto GPRSize = std::max(OpSize::i32Bit, OpSizeFromDst(Op));
if (CTX->HostFeatures.SupportsFRINTTS) {
// When we have FRINTTS, this is a two-step process. First, we round to the
@@ -2616,7 +2617,9 @@ void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSize, boo
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : SrcElementSize;
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
Ref Result = CVTFPR_To_GPRImpl(Op, Src, SrcElementSize, HostRoundingMode);
StoreResultGPR(Op, Result);
const auto DestSize = std::max(OpSize::i32Bit, OpSizeFromDst(Op));
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, DestSize);
}
Ref OpDispatchBuilder::Vector_CVT_Int_To_FloatImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen) {
@@ -2782,12 +2785,12 @@ void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs, IR::OpSiz
void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
const auto Size = OpSizeFromSrc(Op);
Ref MaskSrc = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
Ref MaskSrc = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
// Mask only cares about the top bit of each byte
MaskSrc = _VCMPLTZ(Size, OpSize::i8Bit, MaskSrc);
// Vector that will overwrite byte elements.
Ref VectorSrc = LoadSourceGPR(Op, Op->Dest, Op->Flags);
Ref VectorSrc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
// RDI source (DS prefix by default)
auto MemDest = MakeSegmentAddress(X86State::REG_RDI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
@@ -2839,11 +2842,15 @@ void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType) {
if (Op->Src[0].IsGPR()) {
// Loading from GPR and moving to Vector.
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], GetGPROpSize(), Op->Flags);
const auto SrcSize = std::max(OpSize::i32Bit, OpSizeFromSrc(Op));
// zext to 128bit
Result = _VCastFromGPR(OpSize::i128Bit, OpSizeFromSrc(Op), Src);
Result = _VCastFromGPR(OpSize::i128Bit, SrcSize, Src);
} else {
// Loading from Memory as a scalar. Zero extend
Result = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
const auto SrcSize = std::max(OpSize::i32Bit, OpSizeFromSrc(Op));
Result = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
}
StoreResult_WithAVXInsert(VectorType, RegClass::FPR, Op, Result);
@@ -2851,14 +2858,19 @@ void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType) {
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
if (Op->Dest.IsGPR()) {
const auto ElementSize = OpSizeFromDst(Op);
const auto DstSize = std::max(OpSize::i32Bit, OpSizeFromDst(Op));
// Extract element from GPR. Zero extending in the process.
Src = _VExtractToGPR(OpSizeFromSrc(Op), ElementSize, Src, 0);
Src = _VExtractToGPR(OpSizeFromSrc(Op), DstSize, Src, 0);
StoreResultGPR(Op, Op->Dest, Src);
StoreResultGPR_WithOpSize(Op, Op->Dest, Src, DstSize);
} else {
const auto DstSize = std::max(OpSize::i32Bit, OpSizeFromDst(Op));
// Storing first element to memory.
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
_StoreMemFPR(OpSizeFromDst(Op), Dest, Src, OpSize::i8Bit);
_StoreMemFPR(DstSize, Dest, Src, OpSize::i8Bit);
}
}
}
@@ -2878,24 +2890,24 @@ Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1
case VectorCompareType::NLT_US: // NGT(Swapped operand)
case VectorCompareType::NLT_UQ: {
Ref Result = _VFCMPLT(Size, ElementSize, Src1, Src2);
return _VNot(Size, ElementSize, Result);
return _VNot(Size, Result);
}
case VectorCompareType::NLE_US: // NGE(Swapped operand)
case VectorCompareType::NLE_UQ: {
Ref Result = _VFCMPLE(Size, ElementSize, Src1, Src2);
return _VNot(Size, ElementSize, Result);
return _VNot(Size, Result);
}
case VectorCompareType::ORD_Q:
case VectorCompareType::ORD_S: return _VFCMPORD(Size, ElementSize, Src1, Src2);
case VectorCompareType::NGT_UQ:
case VectorCompareType::NGT_US: {
Ref Result = _VFCMPLT(Size, ElementSize, Src2, Src1);
return _VNot(Size, ElementSize, Result);
return _VNot(Size, Result);
}
case VectorCompareType::NGE_UQ:
case VectorCompareType::NGE_US: {
Ref Result = _VFCMPLE(Size, ElementSize, Src2, Src1);
return _VNot(Size, ElementSize, Result);
return _VNot(Size, Result);
}
case VectorCompareType::GT_OQ:
case VectorCompareType::GT_OS: return _VFCMPLT(Size, ElementSize, Src2, Src1);
@@ -2906,10 +2918,10 @@ Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1
// If either of the sources are unordered, then returns true.
Ref Src1_U = _VFCMPEQ(Size, ElementSize, Src1, Src1);
Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2);
auto Ordered = _VAnd(Size, ElementSize, Src1_U, Src2_U);
auto Ordered = _VAnd(Size, Src1_U, Src2_U);
Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2);
return _VOrn(Size, ElementSize, Compare_Ordered, Ordered);
return _VOrn(Size, Compare_Ordered, Ordered);
}
case VectorCompareType::NEQ_OQ:
case VectorCompareType::NEQ_OS: {
@@ -2918,8 +2930,8 @@ Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1
Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2);
Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2);
Ref Result = _VAndn(Size, ElementSize, Src1_U, Compare_Ordered);
return _VAnd(Size, ElementSize, Result, Src2_U);
Ref Result = _VAndn(Size, Src1_U, Compare_Ordered);
return _VAnd(Size, Result, Src2_U);
}
case VectorCompareType::FALSE_OQ:
case VectorCompareType::FALSE_OS: return LoadZeroVector(Size);
@@ -3289,7 +3301,7 @@ void OpDispatchBuilder::DefaultX87State(OpcodeArgs) {
// On top of resetting the flags to a default state, we also need to clear
// all of the ST0-7/MM0-7 registers to zero.
Ref ZeroVector = LoadZeroVector(OpSize::i64Bit);
Ref ZeroVector = LoadZeroVector(OpSize::i128Bit);
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
_StoreContextFPR(OpSize::i128Bit, ZeroVector, MMBaseOffset() + i * 16);
}
@@ -3485,7 +3497,7 @@ Ref OpDispatchBuilder::ADDSUBPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Sr
} else {
auto ConstantEOR =
LoadAndCacheNamedVectorConstant(Size, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PADDSUBPS_INVERT : NAMED_VECTOR_PADDSUBPD_INVERT);
auto InvertedSource = _VXor(Size, ElementSize, Src2, ConstantEOR);
auto InvertedSource = _VXor(Size, Src2, ConstantEOR);
return _VFAdd(Size, ElementSize, Src1, InvertedSource);
}
}
@@ -4349,8 +4361,8 @@ void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs, IR::OpSize ElementSiz
}
void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) {
Ref Test1 = _VAnd(Size, OpSize::i8Bit, Dest, Src);
Ref Test2 = _VAndn(Size, OpSize::i8Bit, Src, Dest);
Ref Test1 = _VAnd(Size, Dest, Src);
Ref Test2 = _VAndn(Size, Src, Dest);
// Element size must be less than 32-bit for the sign bit tricks.
Test1 = _VUMaxV(Size, OpSize::i16Bit, Test1);
@@ -4384,11 +4396,11 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref
Ref Mask = _VDupFromGPR(SrcSize, ElementSize, Constant(MaskConstant));
Ref AndTest = _VAnd(SrcSize, OpSize::i8Bit, Src2, Src1);
Ref AndNotTest = _VAndn(SrcSize, OpSize::i8Bit, Src2, Src1);
Ref AndTest = _VAnd(SrcSize, Src2, Src1);
Ref AndNotTest = _VAndn(SrcSize, Src2, Src1);
Ref MaskedAnd = _VAnd(SrcSize, OpSize::i8Bit, AndTest, Mask);
Ref MaskedAndNot = _VAnd(SrcSize, OpSize::i8Bit, AndNotTest, Mask);
Ref MaskedAnd = _VAnd(SrcSize, AndTest, Mask);
Ref MaskedAndNot = _VAnd(SrcSize, AndNotTest, Mask);
Ref MaxAnd = _VUMaxV(SrcSize, OpSize::i16Bit, MaskedAnd);
Ref MaxAndNot = _VUMaxV(SrcSize, OpSize::i16Bit, MaskedAndNot);
@@ -4491,7 +4503,7 @@ Ref OpDispatchBuilder::DPPOpImpl(IR::OpSize DstSize, Ref Src1, Ref Src2, uint8_t
// Now mask results based on IndexMask.
if (SrcMask != SizeMask) {
auto InputMask = LoadAndCacheIndexedNamedVectorConstant(DstSize, NamedIndexMask, SrcMask * 16);
Temp = _VAnd(DstSize, ElementSize, Temp, InputMask);
Temp = _VAnd(DstSize, Temp, InputMask);
}
// Now due a float reduction
@@ -4913,7 +4925,7 @@ void OpDispatchBuilder::VPERM2Op(OpcodeArgs) {
Ref OpDispatchBuilder::VPERMDIndices(OpSize DstSize, Ref Indices, Ref IndexMask, Ref Repeating3210) {
// Get rid of any junk unrelated to the relevant selector index bits (bits [2:0])
Ref SanitizedIndices = _VAnd(DstSize, OpSize::i8Bit, Indices, IndexMask);
Ref SanitizedIndices = _VAnd(DstSize, Indices, IndexMask);
// Build up the broadcasted index mask. e.g. On x86-64, the selector index
// is always in the lower 3 bits of a 32-bit element. However, in order to
@@ -5108,15 +5120,12 @@ void OpDispatchBuilder::VPERMQOp(OpcodeArgs) {
}
Ref OpDispatchBuilder::VBLENDOpImpl(IR::OpSize VecSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint64_t Selector) {
const auto IsWordElements = ElementSize == OpSize::i16Bit;
const auto Is256Bit = VecSize == OpSize::i256Bit;
if (VecSize == OpSize::i256Bit) {
return _VBlendImm(VecSize, ElementSize, Src1, Src2, Selector);
}
const auto ElementsPerLane = uint32_t(IR::NumElements(OpSize::i128Bit, ElementSize));
// PBLENDW uses the same immediate size for 128-bit and 256-bit
// while all the others double in size.
const auto MaskSize = Is256Bit && !IsWordElements ? ElementsPerLane * 2 : ElementsPerLane;
const auto Mask = (1U << MaskSize) - 1;
const auto Mask = (1U << ElementsPerLane) - 1;
// Now, we determine which mask portion has the higher population count.
// we use this to determine which source we use as the base to insert into.
@@ -5133,7 +5142,7 @@ Ref OpDispatchBuilder::VBLENDOpImpl(IR::OpSize VecSize, IR::OpSize ElementSize,
// In the event we tie, then we can just use Src1 and only perform incoming insertions
// that come from Src2.
const auto NumSrc2Bits = uint32_t(std::popcount(Selector & Mask));
const auto NumSrc1Bits = MaskSize - NumSrc2Bits;
const auto NumSrc1Bits = ElementsPerLane - NumSrc2Bits;
const auto IsUsingSrc1 = NumSrc1Bits >= NumSrc2Bits;
Ref Result = IsUsingSrc1 ? Src1 : Src2;
Ref Source = IsUsingSrc1 ? Src2 : Src1;
@@ -5316,7 +5325,7 @@ Ref OpDispatchBuilder::VPERMILRegOpImpl(OpSize DstSize, IR::OpSize ElementSize,
// Sanitize indices first
const auto ShiftAmount = 0b11 >> static_cast<uint32_t>(IsPD);
Ref IndexMask = _VectorImm(DstSize, ElementSize, ShiftAmount);
Ref SanitizedIndices = _VAnd(DstSize, OpSize::i8Bit, Indices, IndexMask);
Ref SanitizedIndices = _VAnd(DstSize, Indices, IndexMask);
Ref IndexTrn1 = _VTrn(DstSize, OpSize::i8Bit, SanitizedIndices, SanitizedIndices);
Ref IndexTrn2 = _VTrn(DstSize, OpSize::i16Bit, IndexTrn1, IndexTrn1);
@@ -5518,7 +5527,7 @@ void OpDispatchBuilder::VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx,
LoadAndCacheNamedVectorConstant(Size, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PSUBADDPS_INVERT : NAMED_VECTOR_PSUBADDPD_INVERT);
}
auto InvertedSourc = _VXor(Size, ElementSize, Sources[AddendIdx - 1], ConstantEOR);
auto InvertedSourc = _VXor(Size, Sources[AddendIdx - 1], ConstantEOR);
Ref Result = _VFMLA(Size, ElementSize, Sources[Src1Idx - 1], Sources[Src2Idx - 1], InvertedSourc);
if (!Is256Bit) {
@@ -5665,7 +5674,7 @@ void OpDispatchBuilder::Extrq_imm(OpcodeArgs) {
const uint64_t Mask = ~0ULL >> (MaskWidth == 0 ? 0 : (64 - MaskWidth));
const Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask));
Result = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Result, MaskVector);
Result = _VAnd(OpSize::i128Bit, Result, MaskVector);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
@@ -5681,7 +5690,7 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) {
Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask));
// Mask incoming source.
Src = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Src, MaskVector);
Src = _VAnd(OpSize::i64Bit, Src, MaskVector);
// If shifting then shift source and mask in to the correct location.
if (Shift) {
@@ -5689,11 +5698,8 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) {
MaskVector = _VShlI(OpSize::i128Bit, OpSize::i64Bit, MaskVector, Shift);
}
// Negate the mask.
MaskVector = _VNot(OpSize::i64Bit, OpSize::i64Bit, MaskVector);
Dest = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Dest, MaskVector);
const Ref Result = _VOr(OpSize::i64Bit, OpSize::i64Bit, Dest, Src);
Dest = _VAndn(OpSize::i64Bit, Dest, MaskVector);
const Ref Result = _VOr(OpSize::i64Bit, Dest, Src);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
@@ -5710,15 +5716,15 @@ void OpDispatchBuilder::Extrq(OpcodeArgs) {
};
// Bits[5:0] = Mask width in bits
const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Src, ElementMask);
const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, Src, ElementMask);
// Bits[13:8] = Shift right in bits
const Ref ShiftBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, Src, 8), ElementMask);
const Ref ShiftBits = _VAnd(OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, Src, 8), ElementMask);
// First shift in to the correct position.
Ref Result = _VUShr(OpSize::i64Bit, OpSize::i64Bit, Dest, ShiftBits, false);
Result = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Result, GenerateMask(MaskWidthBits));
Result = _VAnd(OpSize::i128Bit, Result, GenerateMask(MaskWidthBits));
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
@@ -5737,21 +5743,19 @@ void OpDispatchBuilder::Insertq(OpcodeArgs) {
};
// Bits[5:0] = Mask width in bits
const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, ElementMask);
const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, SelectorBits, ElementMask);
// Bits[13:8] = Shift right in bits
const Ref ShiftBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, 8), ElementMask);
const Ref ShiftBits = _VAnd(OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, 8), ElementMask);
// Extract the source data and put in to the correct location
const Ref SrcMask = GenerateMask(MaskWidthBits);
Ref SrcData = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Src, SrcMask);
Ref SrcData = _VAnd(OpSize::i128Bit, Src, SrcMask);
SrcData = _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcData, ShiftBits, false);
// Generate a destination mask
const Ref DstMask = _VNot(OpSize::i64Bit, OpSize::i64Bit, _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcMask, ShiftBits, false));
Ref Result = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Dest, DstMask);
Result = _VOr(OpSize::i64Bit, OpSize::i64Bit, Result, SrcData);
Ref Result = _VAndn(OpSize::i64Bit, Dest, _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcMask, ShiftBits, false));
Result = _VOr(OpSize::i64Bit, Result, SrcData);
StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result);
}
@@ -575,7 +575,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
for (int i = 0; i < 7; ++i) {
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
// Mask off the top bits
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
Reg = _VAnd(OpSize::i128Bit, Reg, Mask);
if (ReducedPrecisionMode) {
// Convert to double precision
Reg = _F80CVT(OpSize::i64Bit, Reg);
@@ -0,0 +1,97 @@
// SPDX-License-Identifier: MIT
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/SharedCodeBufferManager.h"
#include <FEXCore/fextl/memory.h>
#include <FEXCore/Utils/AllocatorHooks.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#ifndef _WIN32
#include <FEXCore/Utils/PrctlUtils.h>
#endif
namespace FEXCore::CPU {
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
CodeBuffer::CodeBuffer(size_t Size)
: AllocatedSize(Size) {
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
FEXCore::Allocator::ProtectOptions::None)) {
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
}
FEXCore::Allocator::VirtualName("FEXMemJIT", Ptr, Size);
// Huge-pages reduce the amount of iTLB misses dramatically when it works.
FEXCore::Allocator::VirtualTHPControl(Ptr, Size, FEXCore::Allocator::THPControl::Enable);
LookupCache = fextl::make_unique<GuestToHostMap>();
}
CodeBuffer::~CodeBuffer() {
FEXCore::Allocator::VirtualFree(Ptr, AllocatedSize);
}
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::AllocateNew(size_t Size) {
#ifndef _WIN32
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
//
// MDWE prevents applications from creating RWX memory mappings.
// This prevents FEX from doing anything JIT related, as FEX uses RWX for JIT memory mappings.
//
// A potential workaround to make FEX work with MDWE is to call mprotect every time we need to write or modify code.
// Alternatively, FEX could use a memory mirror where one half is mapped as RW and the other is RX.
//
// Once MDWE is enabled with the prctl, the feature is sealed and it can /NOT/ be turned off.
//
// Status of MDWE is queried through prctl using `PR_GET_MDWE`:
// -1: The kernel doesn't support MDWE
// 0: MDWE is supported but disabled
// >0: MDWE is enabled, hence prohibiting RWX mappings
int MDWE = ::prctl(PR_GET_MDWE, 0, 0, 0, 0);
if (MDWE != -1 && MDWE != 0) {
LogMan::Msg::EFmt("MDWE was set to 0x{:x} which means FEX can't allocate executable memory", MDWE);
}
#endif
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
Latest = Buffer;
LatestOffset = 0;
OnCodeBufferAllocated(Buffer);
return Buffer;
}
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::GetLatest() {
if (!Latest) {
AllocateNew(INITIAL_CODE_SIZE);
}
return Latest;
}
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::StartLargerCodeBuffer() {
if (!Latest) {
// Allocate initial CodeBuffer and return it
return GetLatest();
}
auto NewCodeBufferSize = GetLatest()->AllocatedSize;
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
return AllocateNew(NewCodeBufferSize);
}
fextl::shared_ptr<CodeBuffer> SharedCodeBufferManager::StartMaximalCodeBuffer() {
return AllocateNew(MAX_CODE_SIZE);
}
} // namespace FEXCore::CPU
@@ -0,0 +1,78 @@
// SPDX-License-Identifier: MIT
/*
$info$
category: Thread shared code buffer management
tags: backend|shared
$end_info$
*/
#pragma once
#include <FEXCore/fextl/memory.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <cstddef>
#include <cstdint>
namespace FEXCore {
struct GuestToHostMap;
}
namespace FEXCore::CPU {
struct CodeBuffer {
uint8_t* Ptr;
size_t AllocatedSize; // including guard page; see UsableSize()
fextl::unique_ptr<GuestToHostMap> LookupCache;
CodeBuffer(size_t Size);
CodeBuffer(const CodeBuffer&) = delete;
CodeBuffer& operator=(const CodeBuffer&) = delete;
CodeBuffer(CodeBuffer&& oth) = delete;
CodeBuffer& operator=(CodeBuffer&&) = delete;
~CodeBuffer();
/// Returns the number of bytes available for storing code
size_t UsableSize() const {
return AllocatedSize - FEXCore::Utils::FEX_PAGE_SIZE;
}
};
/**
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
*
* The CodeBuffer is managed as a partially persistent data structure:
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
*/
class SharedCodeBufferManager {
public:
// Get the CodeBuffer that was most recently allocated.
// This is the only CodeBuffer that data may be written to.
fextl::shared_ptr<CodeBuffer> GetLatest();
// Allocate a new CodeBuffer with geometric growth up to an internal maximum.
// Subsequent calls to GetLatest will point to the returned buffer.
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer();
// Allocate a new CodeBuffer with maximum internal size.
// Subsequent calls to GetLatest will point to the returned buffer.
fextl::shared_ptr<CodeBuffer> StartMaximalCodeBuffer();
// Write offset into the latest CodeBuffer
std::size_t LatestOffset {};
// Protects writes to the latest CodeBuffer and changes to LatestOffset
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
private:
fextl::shared_ptr<CodeBuffer> Latest;
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
};
} // namespace FEXCore::CPU
@@ -808,7 +808,7 @@ namespace AVX256 {
{OPD(2, 0b01, 0xB6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, true, 2, 3, 1>}, // VFMADDSUB
{OPD(2, 0b01, 0xB7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, false, 2, 3, 1>}, // VFMSUBADD
{OPD(2, 0b01, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
{OPD(2, 0b01, 0xDB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESImcOp, true>},
{OPD(2, 0b01, 0xDC), 1, &OpDispatchBuilder::VAESEncOp},
{OPD(2, 0b01, 0xDD), 1, &OpDispatchBuilder::VAESEncLastOp},
{OPD(2, 0b01, 0xDE), 1, &OpDispatchBuilder::VAESDecOp},
@@ -860,7 +860,7 @@ namespace AVX256 {
{OPD(3, 0b01, 0x62), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRMOp, true>},
{OPD(3, 0b01, 0x63), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRIOp, true>},
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AESKeyGenAssist, true>},
};
#undef OPD
+2 -2
View File
@@ -643,7 +643,7 @@ public:
auto IROp = Node.GetNode(BaseList)->Op(IRList);
if (IROp->Op == OP_BEGINBLOCK) {
auto BeginBlock = IROp->C<IROp_EndBlock>();
auto BeginBlock = IROp->C<IROp_BeginBlock>();
Node = BeginBlock->BlockHeader;
} else if (IROp->Op == OP_CODEBLOCK) {
@@ -675,7 +675,7 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
[[nodiscard]]
bool IsBlockExit(FEXCore::IR::IROps Op);
void Dump(fextl::stringstream* out, const IRListView* IR);
void Dump(fextl::ostringstream* out, const IRListView* IR);
constexpr auto format_as(FEXCore::IR::NodeID ID) {
return ID.Value;
+72 -54
View File
@@ -1864,9 +1864,9 @@
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = VNot OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"FPR = VNot OpSize:#RegisterSize, FPR:$Vector": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
"ElementSize": "OpSize::i8Bit"
},
"FPR = VAbs OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
@@ -2004,15 +2004,6 @@
"BitShift > 0"
]
},
"FPR = VUShraI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
"TiedSource": 0,
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
]
},
"FPR = VSShrI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
"TiedSource": 0,
"DestSize": "RegisterSize",
@@ -2025,7 +2016,7 @@
"FPR = VUShrNI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
"TiedSource": 0,
"Desc": "Unsigned shifts right each element and then narrows to the next lower element size",
"Desc": ["Unsigned shifts right each element and then narrows to the next lower element size"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize >> 1",
"EmitValidation": [
@@ -2047,7 +2038,7 @@
]
},
"FPR = VSXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": "Sign extends elements from the source element size to the next size up",
"Desc": ["Sign extends elements from the source element size to the next size up"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
@@ -2059,7 +2050,7 @@
"ElementSize": "ElementSize << 1"
},
"FPR = VSSHLL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift{0}": {
"Desc": "Sign extends elements from the source element size to the next size up",
"Desc": ["Sign extends elements from the source element size to the next size up"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
@@ -2071,7 +2062,7 @@
"ElementSize": "ElementSize << 1"
},
"FPR = VUXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": "Zero extends elements from the source element size to the next size up",
"Desc": ["Zero extends elements from the source element size to the next size up"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
@@ -2153,43 +2144,55 @@
"ElementSize": "ElementSize"
},
"FPR = VAnd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"FPR = VAnd OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VAndn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"FPR = VAndn OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VOrn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"FPR = VOrn OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VOr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"FPR = VOr OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VXor OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"FPR = VXor OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": {
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i8Bit",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
]
},
"FPR = VXar OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$LHS, FPR:$RHS, u8:$Rotate": {
"Desc": [
"Performs an XOR of corresponding elements and then rotates them right by",
"an amount between [1, ElementSize]"
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
"RegisterSize == IR::OpSize::i256Bit || RegisterSize == IR::OpSize::i128Bit"
]
},
@@ -2214,7 +2217,7 @@
},
"FPR = VAddP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
"Desc": "Does a horizontal pairwise add of elements across the two source vectors",
"Desc": ["Does a horizontal pairwise add of elements across the two source vectors"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
@@ -2271,7 +2274,7 @@
"ElementSize": "ElementSize"
},
"FPR = VFAddP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
"Desc": "Does a horizontal pairwise add of elements across the two source vectors with float element types",
"Desc": ["Does a horizontal pairwise add of elements across the two source vectors with float element types"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
@@ -2321,29 +2324,27 @@
"ElementSize": "ElementSize << 1"
},
"FPR = VUMull2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": "Multiplies the high elements with size extension",
"Desc": ["Multiplies the high elements with size extension"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
"FPR = VSMull2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": "Multiplies the high elements with size extension",
"Desc": ["Multiplies the high elements with size extension"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
"FPR = VUMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": "Wide unsigned multiply returning the high results",
"Desc": ["Wide unsigned multiply returning the high results"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = VSMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": "Wide signed multiply returning the high results",
"Desc": ["Wide signed multiply returning the high results"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = VUABDL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Unsigned Absolute Difference Long"
],
"Desc": ["Unsigned Absolute Difference Long"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
@@ -2564,6 +2565,24 @@
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"TiedSource": 2
},
"FPR = VBlendImm OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$LHS, FPR:$RHS, u16:$Selector": {
"Desc": [
"Functions the same way an immediate blend operation on x86 would.",
"That is: (e.g. using 16-bit elements)",
" if (Selector[0] == 1)",
" Dst[15:0] = RHS[15:0]",
" else",
" Dst[15:0] = LHS[15:0]",
" <etc for the rest of the elements along the vector>",
"",
"Note that like x86, due to the selector size, the operation of this IR op",
"uses a 128-bit lane granularity, so each blending selector independently operates",
"on each 128-bit element that composes the vector."
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize",
"TiedSource": 0
}
},
"Conv": {
@@ -2601,7 +2620,7 @@
},
"FPR = Vector_SToF OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": "Vector op: Converts signed integer to same size float",
"Desc": ["Vector op: Converts signed integer to same size float"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
@@ -2613,12 +2632,12 @@
"ElementSize": "ElementSize"
},
"FPR = Vector_FToZS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": "Vector op: Converts float to signed integer, rounding towards zero",
"Desc": ["Vector op: Converts float to signed integer, rounding towards zero"],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize"
},
"FPR = Vector_FToF OpSize:#RegisterSize, OpSize:#DestElementSize, FPR:$Vector, OpSize:$SrcElementSize": {
"Desc": "Vector op: Converts float from source element size to destination size (fp32<->fp64)",
"Desc": ["Vector op: Converts float from source element size to destination size (fp32<->fp64)"],
"DestSize": "RegisterSize",
"ElementSize": "DestElementSize"
},
@@ -2673,75 +2692,74 @@
},
"Crypto": {
"FPR = VAESImc FPR:$Vector": {
"Desc": "Does a stage of the inverse mix column transformation",
"Desc": ["Does a stage of the inverse mix column transformation"],
"DestSize": "OpSize::i128Bit"
},
"FPR = VAESEnc OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
"Desc": "Does a step of AES encryption",
"Desc": ["Does a step of AES encryption"],
"DestSize": "RegisterSize"
},
"FPR = VAESEncLast OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
"Desc": "Does the last step of AES encryption",
"Desc": ["Does the last step of AES encryption"],
"DestSize": "RegisterSize"
},
"FPR = VAESDec OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
"Desc": "Does a step of AES decryption",
"Desc": ["Does a step of AES decryption"],
"DestSize": "RegisterSize"
},
"FPR = VAESDecLast OpSize:#RegisterSize, FPR:$State, FPR:$Key, FPR:$ZeroReg": {
"Desc": "Does the last step of AES decryption",
"Desc": ["Does the last step of AES decryption"],
"DestSize": "RegisterSize"
},
"FPR = VAESKeyGenAssist FPR:$Src, FPR:$KeyGenTBLSwizzle, FPR:$ZeroReg, u8:$RCON": {
"Desc": "Assists in key generation",
"Desc": ["Assists in key generation"],
"DestSize": "OpSize::i128Bit"
},
"FPR = VSha1H FPR:$Src": {
"Desc": "Does vector scalar SHA1H instruction",
"Desc": ["Does vector scalar SHA1H instruction"],
"DestSize": "FEXCore::IR::OpSize::i32Bit"
},
"FPR = VSha1C FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": "Does vector SHA1C instruction",
"Desc": ["Does vector SHA1C instruction"],
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha1M FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": "Does vector SHA1M instruction",
"Desc": ["Does vector SHA1M instruction"],
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha1P FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": "Does vector SHA1P instruction",
"Desc": ["Does vector SHA1P instruction"],
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha1SU1 FPR:$Src1, FPR:$Src2": {
"Desc": "Does vector scalar SHA1H instruction",
"Desc": ["Does vector scalar SHA1H instruction"],
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha256U0 FPR:$Src1, FPR:$Src2": {
"Desc": "Does vector scalar VSha256U0 instruction",
"Desc": ["Does vector scalar VSha256U0 instruction"],
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha256U1 FPR:$Src1, FPR:$Src2": {
"Desc": "Does vector scalar VSha256U1 instruction",
"Desc": ["Does vector scalar VSha256U1 instruction"],
"DestSize": "FEXCore::IR::OpSize::i128Bit"
},
"FPR = VSha256H FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": "Does vector scalar VSha256H instruction",
"Desc": ["Does vector scalar VSha256H instruction"],
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"FPR = VSha256H2 FPR:$Src1, FPR:$Src2, FPR:$Src3": {
"Desc": "Does vector scalar VSha256H2 instruction",
"Desc": ["Does vector scalar VSha256H2 instruction"],
"DestSize": "FEXCore::IR::OpSize::i128Bit",
"TiedSource": 0
},
"GPR = CRC32 GPR:$Src1, GPR:$Src2, OpSize:$SrcSize": {
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
],
"Desc": ["CRC32 using polynomial 0x1EDC6F41"],
"DestSize": "OpSize::i32Bit"
},
"FPR = PCLMUL OpSize:#RegisterSize, FPR:$Src1, FPR:$Src2, u8:$Selector": {
+19 -19
View File
@@ -30,19 +30,19 @@ namespace FEXCore::IR {
#include <FEXCore/IR/IRDefines.inc>
static void PrintArg(fextl::stringstream* out, const IRListView*, const SHA256Sum& Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, const SHA256Sum& Arg) {
*out << fextl::fmt::format("sha256:{:02x}", fmt::join(Arg.data, ""));
}
static void PrintArg(fextl::stringstream* out, const IRListView*, uint64_t Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, uint64_t Arg) {
*out << fextl::fmt::format("#{:#x}", Arg);
}
static void PrintArg(fextl::stringstream* out, const IRListView*, const char* const Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, const char* const Arg) {
*out << fextl::fmt::format("'{}'", Arg);
}
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, CondClass Arg) {
if (Arg == CondClass::AL) {
*out << "ALWAYS";
return;
@@ -55,7 +55,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg)
*out << CondNames[FEXCore::ToUnderlying(Arg)];
}
static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, MemOffsetType Arg) {
static constexpr std::array<std::string_view, 3> Names = {
"SXTX",
"UXTW",
@@ -65,7 +65,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType
*out << Names[FEXCore::ToUnderlying(Arg)];
}
static void PrintArg(fextl::stringstream* out, const IRListView*, RegClass Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, RegClass Arg) {
*out << [Arg] {
switch (Arg) {
case RegClass::Invalid: return "Invalid";
@@ -79,7 +79,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, RegClass Arg)
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
if (Arg.IsImmediate()) {
auto PhyReg = PhysicalRegister(Arg);
@@ -128,7 +128,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
}
}
static void PrintArg(fextl::stringstream* out, const IRListView*, FenceType Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, FenceType Arg) {
*out << [Arg] {
switch (Arg) {
case FenceType::Load: return "Loads";
@@ -140,7 +140,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FenceType Arg)
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, RoundMode Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, RoundMode Arg) {
*out << [Arg] {
switch (Arg) {
case RoundMode::Nearest: return "Nearest";
@@ -153,7 +153,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, RoundMode Arg)
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, ConstPad Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, ConstPad Arg) {
*out << [Arg] {
switch (Arg) {
case ConstPad::NoPad: return "NoPad";
@@ -164,7 +164,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, ConstPad Arg)
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorConstant Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, NamedVectorConstant Arg) {
*out << [Arg] {
// clang-format off
switch (Arg) {
@@ -260,7 +260,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorCon
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, IndexNamedVectorConstant Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, IndexNamedVectorConstant Arg) {
*out << [Arg] {
// clang-format off
switch (Arg) {
@@ -286,7 +286,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, IndexNamedVect
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, OpSize Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, OpSize Arg) {
*out << [Arg] {
switch (Arg) {
case OpSize::iUnsized: return "Unsized";
@@ -303,7 +303,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, OpSize Arg) {
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, FloatCompareOp Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, FloatCompareOp Arg) {
*out << [Arg] {
switch (Arg) {
case FloatCompareOp::EQ: return "FEQ";
@@ -317,14 +317,14 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FloatCompareOp
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::BreakDefinition Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, FEXCore::IR::BreakDefinition Arg) {
*out << "{" << Arg.ErrorRegister << ".";
*out << static_cast<uint32_t>(Arg.Signal) << ".";
*out << static_cast<uint32_t>(Arg.TrapNumber) << ".";
*out << static_cast<uint32_t>(Arg.si_code) << "}";
}
static void PrintArg(fextl::stringstream* out, const IRListView*, ShiftType Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, ShiftType Arg) {
*out << [Arg] {
switch (Arg) {
case ShiftType::LSL: return "LSL";
@@ -336,7 +336,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, ShiftType Arg)
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, BranchHint Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, BranchHint Arg) {
*out << [Arg] {
switch (Arg) {
case BranchHint::None: return "None";
@@ -348,11 +348,11 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, BranchHint Arg
}();
}
static void PrintArg(fextl::stringstream* out, const IRListView*, const std::array<uint8_t, 0x10>& Arg) {
static void PrintArg(fextl::ostringstream* out, const IRListView*, const std::array<uint8_t, 0x10>& Arg) {
*out << fextl::fmt::format("{:02x}", fmt::join(Arg, ""));
}
void Dump(fextl::stringstream* out, const IRListView* IR) {
void Dump(fextl::ostringstream* out, const IRListView* IR) {
auto HeaderOp = IR->GetHeader();
int8_t CurrentIndent = 0;
+3 -4
View File
@@ -33,7 +33,7 @@ bool IsBlockExit(FEXCore::IR::IROps Op) {
}
}
RegClass IREmitter::WalkFindRegClass(Ref Node) {
RegClass IREmitter::WalkFindRegClass(Ref Node) const {
auto Class = GetOpRegClass(Node);
switch (Class) {
case RegClass::GPR:
@@ -45,9 +45,8 @@ RegClass IREmitter::WalkFindRegClass(Ref Node) {
}
// Complex case, needs to be handled on an op by op basis
uintptr_t DataBegin = DualListData.DataBegin();
FEXCore::IR::IROp_Header* IROp = Node->Op(DataBegin);
const uintptr_t DataBegin = DualListData.DataBegin();
const auto* IROp = Node->Op(DataBegin);
switch (IROp->Op) {
case IROps::OP_LOADREGISTER: {
+9 -9
View File
@@ -45,7 +45,7 @@ public:
*
* @{ */
RegClass WalkFindRegClass(Ref Node);
RegClass WalkFindRegClass(Ref Node) const;
// These inlining helpers are used by IRDefines.inc so define first.
Ref InlineMem(OpSize Size, Ref Offset, MemOffsetType OffsetType, uint8_t& OffsetScale, bool TSO = false) {
@@ -310,14 +310,14 @@ public:
}
/** @} */
RegClass WalkFindRegClass(OrderedNodeWrapper ssa) {
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
RegClass WalkFindRegClass(OrderedNodeWrapper ssa) const {
auto RealNode = ssa.GetNode(DualListData.ListBegin());
return WalkFindRegClass(RealNode);
}
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t* Constant = nullptr) {
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t* Constant = nullptr) const {
auto RealNode = ssa.GetNode(DualListData.ListBegin());
const auto* IROp = RealNode->Op(DualListData.DataBegin());
if (IROp->Op == OP_CONSTANT) {
auto Op = IROp->C<IR::IROp_Constant>();
if (Constant) {
@@ -328,9 +328,9 @@ public:
return false;
}
bool IsValueInlineConstant(OrderedNodeWrapper ssa) {
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
bool IsValueInlineConstant(OrderedNodeWrapper ssa) const {
auto RealNode = ssa.GetNode(DualListData.ListBegin());
const auto* IROp = RealNode->Op(DualListData.DataBegin());
if (IROp->Op == OP_INLINECONSTANT) {
return true;
}
+34 -5
View File
@@ -13,6 +13,7 @@ $end_info$
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
namespace FEXCore::IR {
@@ -66,25 +67,42 @@ void PassManager::Finalize() {
}
}
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
void PassManager::AddDefaultPasses(Context::ContextImpl* ctx) {
FEX_CONFIG_OPT(DisablePasses, O0);
// We only specifically disable optimization passes if desired, as IR output should
// still be well-formed regardless of the modifications made to it.
if (!DisablePasses()) {
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures, ctx->Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit));
InsertPass(CreateDeadFlagCalculationEliminination());
}
}
void PassManager::AddDefaultValidationPasses() {
InsertPass(IR::CreateRegisterAllocationPass(&ctx->CPUID), "RA");
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
InsertValidationPass(Validation::CreateIRValidation(), "IRValidation");
#endif
}
void PassManager::InsertRegisterAllocationPass(FEXCore::Context::ContextImpl* ctx) {
InsertPass(IR::CreateRegisterAllocationPass(&ctx->CPUID), "RA");
Pass* PassManager::InsertPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name) {
auto* PassPtr = InsertAt(Passes.end(), std::move(Pass))->get();
AttemptNameMapping(Name, PassPtr);
return PassPtr;
}
PassManager::PassArrayType::iterator PassManager::InsertAt(PassArrayType::iterator pos, fextl::unique_ptr<Pass> Pass) {
Pass->RegisterPassManager(this);
return Passes.insert(pos, std::move(Pass));
}
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
void PassManager::InsertValidationPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name) {
Pass->RegisterPassManager(this);
auto* PassPtr = ValidationPasses.emplace_back(std::move(Pass)).get();
AttemptNameMapping(Name, PassPtr);
}
#endif
void PassManager::Run(IREmitter* IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::Run");
@@ -98,4 +116,15 @@ void PassManager::Run(IREmitter* IREmit) {
}
#endif
}
void PassManager::AttemptNameMapping(const fextl::string& Name, Pass* NewPass) {
if (Name.empty()) {
// Empty name is a 'don't care' case. e.g. Passes that just need to run,
// but don't need to be actively looked up.
return;
}
const auto Result = NameToPassMaping.emplace(Name, NewPass);
LOGMAN_THROW_A_FMT(Result.second, "Tried to insert pass with name '{}'. But name is already used", Name);
}
} // namespace FEXCore::IR
+31 -43
View File
@@ -8,23 +8,18 @@ $end_info$
#pragma once
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/ThreadPoolAllocator.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <functional>
#include <concepts>
#include <utility>
namespace FEXCore::Context {
class ContextImpl;
}
namespace FEXCore::HLE {
class SyscallHandler;
}
namespace FEXCore::IR {
class PassManager;
class IREmitter;
@@ -44,64 +39,57 @@ protected:
class PassManager final {
public:
void AddDefaultPasses(FEXCore::Context::ContextImpl* ctx);
void AddDefaultValidationPasses();
Pass* InsertPass(fextl::unique_ptr<Pass> Pass, fextl::string Name = "") {
auto PassPtr = InsertAt(Passes.end(), std::move(Pass))->get();
if (!Name.empty()) {
NameToPassMaping[Name] = PassPtr;
}
return PassPtr;
explicit PassManager(Context::ContextImpl* CTX) {
AddDefaultPasses(CTX);
}
void InsertRegisterAllocationPass(FEXCore::Context::ContextImpl* ctx);
// Executes all of the passes added to the manager.
// If assertions are enabled, this will also run all validation passes.
void Run(IREmitter* IREmit);
bool HasPass(fextl::string Name) const {
// Inserts a new pass into the manager, optionally also assigning a name to it
// for use in the lookup functions,
Pass* InsertPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name = "");
// Whether or not a pass with the given name is within the manager.
bool HasPass(const fextl::string& Name) const {
return NameToPassMaping.contains(Name);
}
template<typename T>
T* GetPass(fextl::string Name) {
return dynamic_cast<T*>(NameToPassMaping[Name]);
// Retrieves a pass from the manager that has the given name assigned to it.
// Will return nullptr if the pass doesn't exist.
template<std::derived_from<Pass> T>
T* GetPass(const fextl::string& Name) {
return dynamic_cast<T*>(GetPass(Name));
}
Pass* GetPass(fextl::string Name) {
return NameToPassMaping[Name];
}
void RegisterSyscallHandler(FEXCore::HLE::SyscallHandler* Handler) {
SyscallHandler = Handler;
Pass* GetPass(const fextl::string& Name) {
const auto Iter = NameToPassMaping.find(Name);
if (Iter == NameToPassMaping.end()) {
return nullptr;
}
return Iter->second;
}
// Finalizes the pass manager state and assumes no other passes will be added after called.
// This will reorganize the execution order of the passes if necessary.
void Finalize();
protected:
FEXCore::HLE::SyscallHandler* SyscallHandler {};
private:
void AddDefaultPasses(Context::ContextImpl* ctx);
using PassArrayType = fextl::vector<fextl::unique_ptr<Pass>>;
PassArrayType::iterator InsertAt(PassArrayType::iterator pos, fextl::unique_ptr<Pass> Pass) {
Pass->RegisterPassManager(this);
return Passes.insert(pos, std::move(Pass));
}
PassArrayType::iterator InsertAt(PassArrayType::iterator pos, fextl::unique_ptr<Pass> Pass);
PassArrayType Passes;
fextl::unordered_map<fextl::string, Pass*> NameToPassMaping;
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
fextl::vector<fextl::unique_ptr<Pass>> ValidationPasses;
void InsertValidationPass(fextl::unique_ptr<Pass> Pass, fextl::string Name = "") {
Pass->RegisterPassManager(this);
auto PassPtr = ValidationPasses.emplace_back(std::move(Pass)).get();
if (!Name.empty()) {
NameToPassMaping[Name] = PassPtr;
}
}
void InsertValidationPass(fextl::unique_ptr<Pass> Pass, const fextl::string& Name = "");
#endif
void AttemptNameMapping(const fextl::string& Name, Pass* NewPass);
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
FEX_CONFIG_OPT(PassManagerDumpIR, PASSMANAGERDUMPIR);
};
@@ -57,7 +57,7 @@ void IRDumper::Run(IREmitter* IREmit) {
}
if (FD.IsValid() || DumpToLog) {
fextl::stringstream out;
fextl::ostringstream out;
FEXCore::IR::Dump(&out, &IR);
if (FD.IsValid()) {
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", +HeaderOp->OriginalRIP, out.str());
@@ -10,6 +10,7 @@ $end_info$
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include "Interface/IR/RegisterAllocationData.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/Passes/IRValidation.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
@@ -46,7 +47,7 @@ void IRValidation::Run(IREmitter* IREmit) {
OffsetToBlockMap.clear();
EntryBlock = nullptr;
uint32_t Count = CurrentIR.GetSSACount();
const auto Count = CurrentIR.GetSSACount();
if (Count > MaxNodes) {
NodeIsLive.Realloc(Count);
}
@@ -59,7 +60,7 @@ void IRValidation::Run(IREmitter* IREmit) {
#endif
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
auto BlockIROp = BlockHeader->C<FEXCore::IR::IROp_CodeBlock>();
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
if (!EntryBlock) {
@@ -77,15 +78,15 @@ void IRValidation::Run(IREmitter* IREmit) {
const auto OpSize = IROp->Size;
if (GetHasDest(IROp->Op)) {
HadError |= OpSize == IR::OpSize::iInvalid;
// Does the op have a destination of size 0?
// Does the op have an unsized destination?
if (OpSize == IR::OpSize::iInvalid) {
HadError = true;
Errors << "%" << ID << ": Had destination but with no size" << std::endl;
}
// Does the node have zero uses? Should have been DCE'd
if (CodeNode->GetUses() == 0) {
HadWarning |= true;
HadWarning = true;
Warnings << "%" << ID << ": Destination created but had no uses" << std::endl;
}
@@ -98,27 +99,26 @@ void IRValidation::Run(IREmitter* IREmit) {
// If no register class was assigned
if (AssignedClass == IR::RegClass::Invalid) {
HadError |= true;
HadError = true;
Errors << "%" << ID << ": Had destination but with no register class assigned" << std::endl;
}
// If no physical register was assigned
if (PhyReg.IsInvalid()) {
HadError |= true;
HadError = true;
Errors << "%" << ID << ": Had destination but with no register assigned" << std::endl;
}
// Assigned class wasn't the expected class and it is a non-complex op
if (AssignedClass != ExpectedClass && ExpectedClass != IR::RegClass::Complex) {
HadWarning |= true;
HadWarning = true;
Warnings << "%" << ID << ": Destination had register class " << uint32_t(AssignedClass) << " When register class "
<< uint32_t(ExpectedClass) << " Was expected" << std::endl;
}
}
}
uint8_t NumArgs = IR::GetRAArgs(IROp->Op);
const uint8_t NumArgs = IR::GetRAArgs(IROp->Op);
for (uint32_t i = 0; i < NumArgs; ++i) {
OrderedNodeWrapper Arg = IROp->Args[i];
const auto ArgID = Arg.ID();
@@ -126,8 +126,6 @@ void IRValidation::Run(IREmitter* IREmit) {
continue;
}
IROps Op = CurrentIR.GetOp<IROp_Header>(Arg)->Op;
if (ArgID.IsValid()) {
Uses[ArgID.Value]++;
}
@@ -135,10 +133,11 @@ void IRValidation::Run(IREmitter* IREmit) {
// We do not validate the location of inline constants because it's
// irrelevant, they're ignored by RA and always inlined to where they
// need to be. This lets us pool inline constants globally.
bool Ignore = (Op == OP_IRHEADER || Op == OP_INLINECONSTANT);
const IROps Op = CurrentIR.GetOp<IROp_Header>(Arg)->Op;
const bool Ignore = (Op == OP_IRHEADER || Op == OP_INLINECONSTANT);
if (!Ignore && ArgID.IsValid() && !NodeIsLive.Get(ArgID.Value)) {
HadError |= true;
HadError = true;
Errors << "%" << ID << ": Arg[" << i << "] references invalid %" << ArgID << std::endl;
}
}
@@ -147,7 +146,6 @@ void IRValidation::Run(IREmitter* IREmit) {
switch (IROp->Op) {
case IR::OP_EXITFUNCTION: {
CurrentBlock->HasExit = true;
break;
}
case IR::OP_CONDJUMP: {
@@ -163,7 +161,7 @@ void IRValidation::Run(IREmitter* IREmit) {
const FEXCore::IR::IROp_Header* FalseTargetOp = CurrentIR.GetOp<IROp_Header>(FalseTargetNode);
if (TrueTargetOp->Op != OP_CODEBLOCK) {
HadError |= true;
HadError = true;
Errors << "CondJump %" << ID << ": True Target Jumps to Op that isn't the begining of a block" << std::endl;
} else {
auto Block = OffsetToBlockMap.try_emplace(Op->TrueBlock.ID()).first;
@@ -171,7 +169,7 @@ void IRValidation::Run(IREmitter* IREmit) {
}
if (FalseTargetOp->Op != OP_CODEBLOCK) {
HadError |= true;
HadError = true;
Errors << "CondJump %" << ID << ": False Target Jumps to Op that isn't the begining of a block" << std::endl;
} else {
auto Block = OffsetToBlockMap.try_emplace(Op->FalseBlock.ID()).first;
@@ -187,7 +185,7 @@ void IRValidation::Run(IREmitter* IREmit) {
const FEXCore::IR::IROp_Header* TargetOp = CurrentIR.GetOp<IROp_Header>(TargetNode);
if (TargetOp->Op != OP_CODEBLOCK) {
HadError |= true;
HadError = true;
Errors << "Jump %" << ID << ": Jump to Op that isn't the begining of a block" << std::endl;
} else {
auto Block = OffsetToBlockMap.try_emplace(Op->Header.Args[0].ID()).first;
@@ -204,7 +202,7 @@ void IRValidation::Run(IREmitter* IREmit) {
// Blocks can only have zero (Exit), 1 (Unconditional branch) or 2 (Conditional) successors
size_t NumSuccessors = CurrentBlock->Successors.size();
if (NumSuccessors > 2) {
HadError |= true;
HadError = true;
Errors << "%" << BlockID << " Has " << NumSuccessors << " successors which is too many" << std::endl;
}
@@ -220,7 +218,7 @@ void IRValidation::Run(IREmitter* IREmit) {
{
auto Op = GetOp(CodeCurrent);
if (Op != IR::OP_ENDBLOCK) {
HadError |= true;
HadError = true;
Errors << "%" << BlockID << " Failed to end block with EndBlock" << std::endl;
}
}
@@ -231,7 +229,7 @@ void IRValidation::Run(IREmitter* IREmit) {
{
auto Op = GetOp(CodeCurrent);
if (!IsBlockExit(Op)) {
HadError |= true;
HadError = true;
Errors << "%" << BlockID << " Didn't have a block exit IR op as its last instruction" << std::endl;
}
}
@@ -243,7 +241,7 @@ void IRValidation::Run(IREmitter* IREmit) {
for (uint32_t i = 0; i < CurrentIR.GetSSACount(); i++) {
auto [Node, IROp] = CurrentIR.at(IR::NodeID {i})();
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
HadError |= true;
HadError = true;
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
}
}
@@ -251,7 +249,7 @@ void IRValidation::Run(IREmitter* IREmit) {
HadWarning = false;
if (HadError || HadWarning) {
fextl::stringstream Out;
fextl::ostringstream Out;
FEXCore::IR::Dump(&Out, &CurrentIR);
if (HadError) {
@@ -8,20 +8,16 @@
namespace FEXCore::IR::Validation {
struct BlockInfo {
bool HasExit;
const OrderedNode* BlockNode;
fextl::vector<OrderedNode*> Predecessors;
fextl::vector<OrderedNode*> Successors;
};
class IRValidation final : public FEXCore::IR::Pass {
public:
~IRValidation();
void Run(IREmitter* IREmit) override;
private:
struct BlockInfo {
fextl::vector<OrderedNode*> Predecessors;
fextl::vector<OrderedNode*> Successors;
};
BitSet<uint64_t> NodeIsLive {};
OrderedNode* EntryBlock {};
@@ -7,6 +7,7 @@ $end_info$
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/X86Enums.h>
@@ -196,7 +197,7 @@ unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClass Cond)
}
}
constexpr FlagInfo ClassifyConst(IROps Op) {
static constexpr FlagInfo ClassifyConst(IROps Op) {
switch (Op) {
case OP_ANDWITHFLAGS:
return FlagInfo::Pack({
@@ -332,15 +333,15 @@ constexpr FlagInfo ClassifyConst(IROps Op) {
}
}
constexpr auto FlagInfos = std::invoke([] {
constexpr auto FlagInfos = [] {
std::array<FlagInfo, OP_LAST> ret = {};
for (unsigned i = 0; i < OP_LAST; ++i) {
ret[i] = ClassifyConst((IROps)i);
ret[i] = ClassifyConst(IROps(i));
}
return ret;
});
}();
FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
FlagInfo Info = FlagInfos[IROp->Op];
@@ -351,22 +352,22 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
switch (IROp->Op) {
case OP_NZCVSELECT:
case OP_NZCVSELECTINCREMENT: {
auto Op = IROp->CW<IR::IROp_NZCVSelect>();
auto Op = IROp->C<IR::IROp_NZCVSelect>();
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
}
case OP_NZCVSELECTV: {
auto Op = IROp->CW<IR::IROp_NZCVSelectV>();
auto Op = IROp->C<IR::IROp_NZCVSelectV>();
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
}
case OP_NEG: {
auto Op = IROp->CW<IR::IROp_Neg>();
auto Op = IROp->C<IR::IROp_Neg>();
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
}
case OP_CONDJUMP: {
auto Op = IROp->CW<IR::IROp_CondJump>();
auto Op = IROp->C<IR::IROp_CondJump>();
if (!Op->FromNZCV) {
return FlagInfo::Pack({});
}
@@ -376,7 +377,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
case OP_CONDSUBNZCV:
case OP_CONDADDNZCV: {
auto Op = IROp->CW<IR::IROp_CondAddNZCV>();
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
return FlagInfo::Pack({
.Read = FlagsForCondClassType(Op->Cond),
.Write = FLAG_NZCV,
@@ -385,7 +386,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
}
case OP_RMIFNZCV: {
auto Op = IROp->CW<IR::IROp_RmifNZCV>();
auto Op = IROp->C<IR::IROp_RmifNZCV>();
static_assert(FLAG_N == (1 << 3), "rmif mask lines up with our bits");
static_assert(FLAG_Z == (1 << 2), "rmif mask lines up with our bits");
@@ -399,7 +400,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
}
case OP_INVALIDATEFLAGS: {
auto Op = IROp->CW<IR::IROp_InvalidateFlags>();
auto Op = IROp->C<IR::IROp_InvalidateFlags>();
unsigned Flags = 0;
// TODO: Make this translation less silly
@@ -536,7 +537,7 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie
// Initialize the FlagsRead mask according to the exit instruction.
auto [ExitNode, ExitOp] = CodeLast();
if (ExitOp->Op == IR::OP_CONDJUMP) {
auto Op = ExitOp->CW<IR::IROp_CondJump>();
auto Op = ExitOp->C<IR::IROp_CondJump>();
FlagsRead = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags;
} else if (ExitOp->Op == IR::OP_JUMP) {
FlagsRead = CFG.Get(ExitOp->Args[0])->Flags;
@@ -643,7 +644,7 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
for (auto [CodeNode, IROp] : CurrentIR.GetCode(Block)) {
if (IROp->Op == OP_STOREPF) {
auto Op = IROp->CW<IR::IROp_StorePF>();
auto Op = IROp->C<IR::IROp_StorePF>();
auto Generator = CurrentIR.GetOp<IR::IROp_Header>(Op->Value);
// Determine if we only write 0/1 to the parity flag.
@@ -696,7 +697,7 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
--CodeLast;
auto [ExitNode, ExitOp] = CodeLast();
if (ExitOp->Op == IR::OP_CONDJUMP) {
auto Op = ExitOp->CW<IR::IROp_CondJump>();
auto Op = ExitOp->C<IR::IROp_CondJump>();
CFG.RecordEdge(Block->ID, Op->TrueBlock);
CFG.RecordEdge(Block->ID, Op->FalseBlock);
@@ -33,7 +33,7 @@ namespace {
Ref RegToSSA[32];
};
IR::RegClass GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
IR::RegClass GetRegClassFromNode(const IR::IROp_Header* IROp) {
const auto Class = IR::GetRegClass(IROp->Op);
if (Class != IR::RegClass::Complex) {
return Class;
@@ -49,9 +49,14 @@ namespace {
case IR::OP_FILLREGISTER: return IROp->C<IR::IROp_FillRegister>()->Class;
default: return IR::RegClass::Invalid;
}
};
}
} // Anonymous namespace
void RegisterAllocationPass::SetNumPairRegs(uint32_t NumRegs) {
LOGMAN_THROW_A_FMT((NumRegs % 2) == 0, "Number of pair regs must be even. (Given: {})", NumRegs);
PairRegs = NumRegs;
}
class ConstrainedRAPass final : public RegisterAllocationPass {
public:
explicit ConstrainedRAPass(const FEXCore::CPUIDEmu* CPUID)
@@ -85,27 +90,27 @@ private:
// SourcesNextUses is read backwards, this tracks the index
int64_t SourceIndex {};
bool Rematerializable(IROp_Header* IROp) {
static bool Rematerializable(const IROp_Header* IROp) {
return IROp->Op == OP_CONSTANT;
}
Ref InsertFill(Ref Node) {
IROp_Header* IROp = IR->GetOp<IROp_Header>(Node);
const auto* IROp = IR->GetOp<IROp_Header>(Node);
// Remat if we can
if (Rematerializable(IROp)) {
const auto Op = IROp->C<IR::IROp_Constant>();
uint64_t Const = Op->Constant;
const auto* Op = IROp->C<IR::IROp_Constant>();
const uint64_t Const = Op->Constant;
return IREmit->_Constant(Const, Op->Pad, Op->MaxBytes);
}
// Otherwise fill from stack
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value];
const uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value];
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Node must have been spilled");
const auto RegClass = GetRegClassFromNode(IR, IROp);
const auto RegClass = GetRegClassFromNode(IROp);
return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, SlotPlusOne - 1, RegClass);
};
}
// IP of next-use of each source. IPs are measured from the end of the
// block, so we don't need to size the block up-front.
@@ -113,32 +118,35 @@ private:
bool AnySpilled {};
bool IsValidArg(OrderedNodeWrapper Arg) {
bool IsValidArg(OrderedNodeWrapper Arg) const {
if (Arg.IsInvalid()) {
return false;
}
auto Op = IR->GetOp<IROp_Header>(Arg)->Op;
return Op != OP_INLINECONSTANT && Op != OP_INLINEENTRYPOINTOFFSET;
};
}
RegisterClassData* GetClass(PhysicalRegister Reg) {
return &Classes[Reg.Class];
};
}
const RegisterClassData* GetClass(PhysicalRegister Reg) const {
return &Classes[Reg.Class];
}
uint32_t GetRegBits(PhysicalRegister Reg) {
return 1 << Reg.Reg;
};
static uint32_t GetRegBits(PhysicalRegister Reg) {
return 1U << Reg.Reg;
}
bool IsInRegisterFile(Ref Node) {
bool IsInRegisterFile(Ref Node) const {
auto ID = IR->GetID(Node).Value;
LOGMAN_THROW_A_FMT(ID < SSAToReg.size(), "Only old nodes looked up");
PhysicalRegister Reg = SSAToReg[ID];
RegisterClassData* Class = GetClass(Reg);
const PhysicalRegister Reg = SSAToReg[ID];
const RegisterClassData* Class = GetClass(Reg);
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Node;
};
}
void FreeReg(PhysicalRegister Reg) {
RegisterClassData* Class = GetClass(Reg);
@@ -147,7 +155,7 @@ private:
LOGMAN_THROW_A_FMT(!(Class->Available & RegBits), "Register double-free");
Class->Available |= RegBits;
};
}
bool HasSource(IROp_Header* I, PhysicalRegister Reg) {
int NumArgs = IR::GetRAArgs(I->Op);
@@ -170,13 +178,13 @@ private:
}
return false;
};
}
Ref DecodeSRANode(const IROp_Header* IROp, Ref Node) {
if (IROp->Op == OP_LOADREGISTER || IROp->Op == OP_LOADPF || IROp->Op == OP_LOADAF) {
return Node;
} else if (IROp->Op == OP_STOREREGISTER) {
auto V = IROp->C<IR::IROp_StorePF>()->Value;
auto V = IROp->C<IR::IROp_StoreRegister>()->Value;
V.ClearKill();
return IR->GetNode(V);
} else if (IROp->Op == OP_STOREPF || IROp->Op == OP_STOREAF) {
@@ -186,9 +194,9 @@ private:
}
return nullptr;
};
}
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) {
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) const {
uint8_t FlagOffset = Classes[FEXCore::ToUnderlying(RegClass::GPRFixed)].Count - 2;
if (IROp->Op == OP_STOREREGISTER) {
@@ -207,9 +215,9 @@ private:
return PhysicalRegister {RegClass::GPRFixed, uint8_t(Op->Reg)};
}
}
};
}
bool IsTrivial(Ref Node, const IROp_Header* Header) {
bool IsTrivial(Ref Node, const IROp_Header* Header) const {
switch (Header->Op) {
case OP_ALLOCATEGPR: return true;
case OP_ALLOCATEGPRAFTER: return true;
@@ -320,7 +328,7 @@ private:
// If we already spilled the Candidate, we don't need to spill again.
// Similarly, if we can rematerialize the instruction, we don't spill it.
if (!Spilled && Header->Op != OP_CONSTANT) {
LOGMAN_THROW_A_FMT(Reg.AsRegClass() == GetRegClassFromNode(IR, Header), "Consistent");
LOGMAN_THROW_A_FMT(Reg.AsRegClass() == GetRegClassFromNode(Header), "Consistent");
// SpillSlots allocation is deferred.
if (SpillSlots.empty()) {
@@ -340,7 +348,7 @@ private:
// Now that we've spilled the value, take it out of the register file
FreeReg(Reg);
AnySpilled = true;
};
}
void RemapReg(Ref Node, PhysicalRegister Reg) {
RegisterClassData* Class = GetClass(Reg);
@@ -350,7 +358,7 @@ private:
if (Index < SSAToReg.size()) {
SSAToReg[Index] = Reg;
}
};
}
// Record a given assignment of register Reg to Node.
void SetReg(Ref Node, PhysicalRegister Reg) {
@@ -363,7 +371,7 @@ private:
RemapReg(Node, Reg);
Node->Reg = Reg.Raw;
};
}
// Assign a register for a given Node, spilling if necessary.
void AssignReg(IROp_Header* IROp, IROp_CodeBlock* Block, Ref CodeNode, IROp_Header* Pivot) {
@@ -419,7 +427,7 @@ private:
}
}
RegClass ClassType = GetRegClassFromNode(IR, IROp);
RegClass ClassType = GetRegClassFromNode(IROp);
RegisterClassData* Class = &Classes[FEXCore::ToUnderlying(ClassType)];
// Spill to make room in the register file.
@@ -432,7 +440,7 @@ private:
LOGMAN_THROW_A_FMT(Class->Available != 0, "Post-condition of spilling");
unsigned Reg = std::countr_zero(Class->Available);
SetReg(CodeNode, PhysicalRegister(ClassType, Reg));
};
}
};
void ConstrainedRAPass::AddRegisters(IR::RegClass Class, uint32_t RegisterCount) {
@@ -441,7 +449,7 @@ void ConstrainedRAPass::AddRegisters(IR::RegClass Class, uint32_t RegisterCount)
Classes[FEXCore::ToUnderlying(Class)].Count = RegisterCount;
}
inline bool KillMove(IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref CodeNode) {
static bool KillMove(const IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref CodeNode) {
// 32-bit moves in x86_64 are represented as a Bfe, detect them.
if (LastOp->Op == OP_BFE && LastOp->C<IR::IROp_Bfe>()->lsb == 0 && LastOp->C<IR::IROp_Bfe>()->Width == 32) {
auto Op = IROp->Op;
@@ -459,7 +467,7 @@ inline bool KillMove(IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref C
return LastOp->Op == OP_STOREREGISTER;
}
inline bool IsSignext(const IROp_Header* IROp, OrderedNodeWrapper Src, OpSize Size) {
static bool IsSignext(const IROp_Header* IROp, OrderedNodeWrapper Src, OpSize Size) {
if (IROp->Op == OP_SBFE) {
auto Sbfe = IROp->C<IR::IROp_Sbfe>();
return Sbfe->Width == 1 && Sbfe->lsb == (IR::OpSizeAsBits(Size) - 1) && Sbfe->Src == Src;
@@ -468,7 +476,7 @@ inline bool IsSignext(const IROp_Header* IROp, OrderedNodeWrapper Src, OpSize Si
}
}
inline bool IsZero(const IROp_Header* IROp) {
static bool IsZero(const IROp_Header* IROp) {
return IROp->Op == OP_CONSTANT && IROp->C<IROp_Constant>()->Constant == 0;
}
@@ -6,10 +6,11 @@ $end_info$
*/
#pragma once
#include "Interface/IR/PassManager.h"
#include <cstdint>
#include <memory>
#include <stdint.h>
namespace FEXCore::IR {
enum class RegClass : uint32_t;
@@ -18,6 +19,9 @@ class RegisterAllocationPass : public FEXCore::IR::Pass {
public:
virtual void AddRegisters(RegClass Class, uint32_t RegisterCount) = 0;
void SetNumPairRegs(uint32_t NumRegs);
protected:
// Number of GPRs usable for pairs at start of GPR set. Must be even.
uint32_t PairRegs {};
};
@@ -3,6 +3,7 @@
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/PassManager.h"
#include "FEXCore/IR/IR.h"
#include "FEXCore/Utils/Profiler.h"
@@ -32,14 +33,14 @@
namespace FEXCore::IR {
// FIXME(pmatos): copy from OpcodeDispatcher.h
inline uint32_t MMBaseOffset() {
static uint32_t MMBaseOffset() {
return static_cast<uint32_t>(offsetof(Core::CPUState, mm[0][0]));
}
// Similar helper to the one in OpcodeDispatcher.h except we do not
// need to handle flags, etc.
template<typename T>
void DeriveOp(Ref& RefV, IROps NewOp, IREmitter::IRPair<T> Expr) {
static void DeriveOp(Ref& RefV, IROps NewOp, IREmitter::IRPair<T> Expr) {
Expr.first->Header.Op = NewOp;
RefV = Expr;
}
@@ -52,8 +53,8 @@ template<typename T>
class FixedSizeStack {
public:
struct StackSlotEntry final {
StackSlot Type;
T Value;
StackSlot Type = StackSlot::UNUSED;
T Value = T::Invalid;
};
static constexpr uint8_t size = 8;
@@ -64,8 +65,7 @@ public:
// If SlowPath is true, then TopOffset is always zero.
int8_t TopOffset = 0;
FixedSizeStack()
: buffer(FixedSizeStack::size, {StackSlot::UNUSED, T::Invalid}) {}
FixedSizeStack() = default;
void push(const T& Value) {
rotate();
@@ -92,25 +92,23 @@ public:
return buffer[Offset];
}
void setTop(T Value, size_t Offset = 0) {
void setTop(const T& Value, size_t Offset = 0) {
buffer[Offset] = {StackSlot::VALID, Value};
}
bool isValid(size_t Offset) const {
return buffer[Offset].first;
return buffer[Offset].Type == StackSlot::VALID;
}
void clear() {
for (auto& Elem : buffer) {
Elem = {StackSlot::UNUSED, T::Invalid};
}
buffer.fill({StackSlot::UNUSED, T::Invalid});
TopOffset = 0;
}
void dump() const {
LogMan::Msg::DFmt("-- Stack");
for (size_t i = 0; i < 8; i++) {
for (size_t i = 0; i < buffer.size(); i++) {
const auto& [Valid, Element] = buffer[i];
if (Valid == StackSlot::VALID) {
LogMan::Msg::DFmt("| ST{}: 0x{:x}", i, (uintptr_t)(Element.StackDataNode));
@@ -126,7 +124,7 @@ public:
}
// Returns a mask to set in AbridgedTagWord
uint8_t getValidMask() {
uint8_t getValidMask() const {
uint8_t Mask = 0;
for (size_t i = 0; i < buffer.size(); i++) {
if (buffer[i].Type == StackSlot::VALID) {
@@ -137,7 +135,7 @@ public:
}
// Returns a mask to set in AbridgedTagWord
uint8_t getInvalidMask() {
uint8_t getInvalidMask() const {
uint8_t Mask = 0;
for (size_t i = 0; i < buffer.size(); i++) {
if (buffer[i].Type == StackSlot::INVALID) {
@@ -148,7 +146,7 @@ public:
}
private:
fextl::vector<StackSlotEntry> buffer;
std::array<StackSlotEntry, size> buffer {};
};
class X87StackOptimization final : public Pass {
@@ -201,11 +199,11 @@ private:
}
}
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
void StoreStackMem_Helper(const IRListView& IR, const IROp_StoreStackMem* Op, Ref StackNode) {
LOGMAN_THROW_A_FMT(!ReducedPrecisionMode, "Full precision mode expected.");
Ref AddrNode = IR->GetNode(Op->Addr);
Ref Offset = IR->GetNode(Op->Offset);
Ref AddrNode = IR.GetNode(Op->Addr);
Ref Offset = IR.GetNode(Op->Offset);
OpSize Align = Op->Align;
MemOffsetType OffsetType = Op->OffsetType;
uint8_t OffsetScale = Op->OffsetScale;
@@ -229,11 +227,11 @@ private:
// Performs a store to memory from a value the stack passed in as StackNode.
// This is the version dealing with the reduced precision case.
void StoreStackMem_Reduced_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
void StoreStackMem_Reduced_Helper(const IRListView& IR, const IROp_StoreStackMem* Op, Ref StackNode) {
LOGMAN_THROW_A_FMT(ReducedPrecisionMode, "Reduced precision mode expected.");
Ref AddrNode = IR->GetNode(Op->Addr);
Ref Offset = IR->GetNode(Op->Offset);
Ref AddrNode = IR.GetNode(Op->Addr);
Ref Offset = IR.GetNode(Op->Offset);
OpSize Align = Op->Align;
MemOffsetType OffsetType = Op->OffsetType;
uint8_t OffsetScale = Op->OffsetScale;
@@ -292,10 +290,10 @@ private:
void Reset();
struct StackMemberInfo {
StackMemberInfo() = delete;
StackMemberInfo(Ref Data)
constexpr StackMemberInfo() = default;
constexpr StackMemberInfo(Ref Data)
: StackDataNode(Data) {}
StackMemberInfo(Ref Data, Ref Source, OpSize Size)
constexpr StackMemberInfo(Ref Data, Ref Source, OpSize Size)
: StackDataNode(Data)
, Source({Size, Source}) {}
Ref StackDataNode {}; // Reference to the data in the Stack.
@@ -357,9 +355,9 @@ private:
// On the slow path TopCache is always the last obtained version of top.
// TopOffset is ignored
bool SlowPath = false;
// Keeping IREmitter not to pass arguments around
IREmitter* IREmit = nullptr;
IRListView* IR = nullptr;
};
inline const X87StackOptimization::StackMemberInfo X87StackOptimization::StackMemberInfo::Invalid {nullptr};
@@ -576,24 +574,34 @@ void X87StackOptimization::HandleBinopStack(IROps Op64, bool VFOp64, IROps Op80,
}
inline void X87StackOptimization::UpdateTopForPop_Slow() {
const auto PopContainer = [](auto& container) {
const auto begin = std::begin(container);
std::rotate(begin, std::next(begin), std::end(container));
};
// Pop the top of the x87 stack
GetOffsetTopWithCache_Slow(1);
std::rotate(TopOffsetCache.begin(), std::next(TopOffsetCache.begin()), TopOffsetCache.end());
std::rotate(TopOffsetAddressCache.begin(), std::next(TopOffsetAddressCache.begin()), TopOffsetAddressCache.end());
std::rotate(TopValueCache.begin(), std::next(TopValueCache.begin()), TopValueCache.end());
std::rotate(FlushValuesPending.begin(), std::next(FlushValuesPending.begin()), FlushValuesPending.end());
std::rotate(TopValidCache.begin(), std::next(TopValidCache.begin()), TopValidCache.end());
PopContainer(TopOffsetCache);
PopContainer(TopOffsetAddressCache);
PopContainer(TopValueCache);
PopContainer(FlushValuesPending);
PopContainer(TopValidCache);
FlushTopPending = true;
}
inline void X87StackOptimization::UpdateTopForPush_Slow() {
// Pop the top of the x87 stack
const auto PushContainer = [](auto& container) {
const auto end = std::end(container);
std::rotate(std::begin(container), std::prev(end), end);
};
// Push the top of the x87 stack
GetOffsetTopWithCache_Slow(1, true);
std::rotate(TopOffsetCache.begin(), std::prev(TopOffsetCache.end()), TopOffsetCache.end());
std::rotate(TopOffsetAddressCache.begin(), std::prev(TopOffsetAddressCache.end()), TopOffsetAddressCache.end());
std::rotate(TopValueCache.begin(), std::prev(TopValueCache.end()), TopValueCache.end());
std::rotate(FlushValuesPending.begin(), std::prev(FlushValuesPending.end()), FlushValuesPending.end());
std::rotate(TopValidCache.begin(), std::prev(TopValidCache.end()), TopValidCache.end());
PushContainer(TopOffsetCache);
PushContainer(TopOffsetAddressCache);
PushContainer(TopValueCache);
PushContainer(FlushValuesPending);
PushContainer(TopValidCache);
FlushTopPending = true;
}
@@ -724,7 +732,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
// Initialize IREmit member
IREmit = Emit;
IR = &CurrentIR;
// Run optimization proper
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
@@ -933,7 +940,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
UpdateTopForPush_Slow();
StoreStackValueAtOffset_Slow(SourceNode);
} else {
auto* SourceNode = CurrentIR.GetNode(Op->X80Src);
if (Op->OriginalValue.IsInvalid()) {
// No original value to track - just push the converted data
StackData.push(StackMemberInfo {SourceNode});
@@ -1019,11 +1025,11 @@ void X87StackOptimization::Run(IREmitter* Emit) {
}
if (ReducedPrecisionMode) {
StoreStackMem_Reduced_Helper(Op, StackNode);
StoreStackMem_Reduced_Helper(CurrentIR, Op, StackNode);
break;
}
StoreStackMem_Helper(Op, StackNode);
StoreStackMem_Helper(CurrentIR, Op, StackNode);
break;
}
@@ -1087,7 +1093,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
ResultNode = IREmit->_VFNeg(OpSize::i64Bit, OpSize::i64Bit, Value);
} else {
Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK);
ResultNode = IREmit->_VXor(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
ResultNode = IREmit->_VXor(OpSize::i128Bit, Value, HelperNode);
}
StoreStackValue(ResultNode);
break;
@@ -1102,7 +1108,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
} else {
// Intermediate insts
Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK);
ResultNode = IREmit->_VAndn(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode);
ResultNode = IREmit->_VAndn(OpSize::i128Bit, Value, HelperNode);
}
StoreStackValue(ResultNode);
break;
@@ -1227,8 +1233,6 @@ void X87StackOptimization::Run(IREmitter* Emit) {
SynchronizeStackValues();
FlushCachedRegs();
}
return;
}
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const HostFeatures& Features, OpSize GPROpSize) {
+10 -9
View File
@@ -38,15 +38,13 @@ namespace FEXCore::Allocator {
MMAP_Hook mmap {::mmap};
MUNMAP_Hook munmap {::munmap};
uint64_t HostVASize {};
using GLIBC_MALLOC_Hook = void* (*)(size_t, const void* caller);
using GLIBC_REALLOC_Hook = void* (*)(void*, size_t, const void* caller);
using GLIBC_FREE_Hook = void (*)(void*, const void* caller);
fextl::unique_ptr<Alloc::HostAllocator> Alloc64 {};
static fextl::unique_ptr<Alloc::HostAllocator> Alloc64 {};
void* FEX_mmap(void* addr, size_t length, int prot, int flags, int fd, off_t offset) {
static void* FEX_mmap(void* addr, size_t length, int prot, int flags, int fd, off_t offset) {
void* Result = Alloc64->Mmap(addr, length, prot, flags, fd, offset);
if (Result >= (void*)-4096) {
errno = -(uint64_t)Result;
@@ -70,7 +68,7 @@ void VirtualName(const char* Name, void* Ptr, size_t Size) {
}
}
int FEX_munmap(void* addr, size_t length) {
static int FEX_munmap(void* addr, size_t length) {
int Result = Alloc64->Munmap(addr, length);
if (Result != 0) {
@@ -104,9 +102,11 @@ void ClearHooks() {
}
#pragma GCC diagnostic pop
FEX_DEFAULT_VISIBILITY size_t DetermineVASize() {
if (HostVASize) {
return HostVASize;
FEX_DEFAULT_VISIBILITY size_t GetHostVABits() {
static uint64_t HostVABits = 0;
if (HostVABits) {
return HostVABits;
}
static constexpr std::array<uintptr_t, 7> TLBSizes = {
@@ -125,6 +125,7 @@ FEX_DEFAULT_VISIBILITY size_t DetermineVASize() {
::munmap(Ptr, FEXCore::Utils::FEX_PAGE_SIZE);
}
if (Ptr != (void*)~0ULL || errno == EEXIST) {
HostVABits = Bits;
return Bits;
}
}
@@ -273,7 +274,7 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
}
fextl::vector<MemoryRegion> Setup48BitAllocatorIfExists(size_t PageSize) {
size_t Bits = FEXCore::Allocator::DetermineVASize();
size_t Bits = FEXCore::Allocator::GetHostVABits();
if (Bits < 48) {
return {};
}
@@ -7,11 +7,9 @@
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/Utils/TypeDefines.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/sstream.h>
#include <FEXCore/fextl/vector.h>
#include <algorithm>
@@ -114,24 +112,24 @@ private:
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<FlexBitElementType>::SizeInBytes(NumElements);
}
static void InitializeVMARegionUsed(LiveVMARegion* Region, size_t AdditionalSize) {
size_t SizeOfLiveRegion =
static void InitializeVMARegionUsed(LiveVMARegion* Region) {
const size_t SizeOfLiveRegion =
FEXCore::AlignUp(LiveVMARegion::GetFEXManagedVMARegionSize(Region->SlabInfo->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
size_t SizePlusManagedData = SizeOfLiveRegion + AdditionalSize;
Region->FreeSpace = Region->SlabInfo->RegionSize - SizePlusManagedData;
Region->FreeSpace = Region->SlabInfo->RegionSize - SizeOfLiveRegion;
size_t NumManagedPages = SizePlusManagedData >> FEXCore::Utils::FEX_PAGE_SHIFT;
size_t NumManagedPages = SizeOfLiveRegion >> FEXCore::Utils::FEX_PAGE_SHIFT;
size_t ManagedSize = NumManagedPages << FEXCore::Utils::FEX_PAGE_SHIFT;
// Use madvise to set the full tracking region to zero.
// This ensures unused pages are zero, while not having the backing pages consuming memory.
::madvise(Region->UsedPages.Memory + ManagedSize, (Region->SlabInfo->RegionSize >> FEXCore::Utils::FEX_PAGE_SHIFT) - ManagedSize,
MADV_DONTNEED);
auto* MemoryAsBytes = reinterpret_cast<uint8_t*>(Region->UsedPages.Memory);
const auto TrackingRegionSize = Region->SlabInfo->RegionSize - ManagedSize;
::madvise(MemoryAsBytes + ManagedSize, TrackingRegionSize, MADV_DONTNEED);
// Use madvise to claim WILLNEED on the beginning pages for initial state tracking.
// Improves performance of the following MemClear by not doing a page level fault dance for data necessary to track >170TB of used pages.
::madvise(Region->UsedPages.Memory, ManagedSize, MADV_WILLNEED);
::madvise(MemoryAsBytes, ManagedSize, MADV_WILLNEED);
// Set our reserved pages
Region->UsedPages.MemSet(NumManagedPages);
@@ -154,28 +152,27 @@ private:
FEXCore::ForkableUniqueMutex AllocationMutex;
void DetermineVASize();
LiveVMARegion* MakeRegionActive(ReservedRegionListType::iterator ReservedIterator, uint64_t UsedSize) {
LiveVMARegion* MakeRegionActive(ReservedRegionListType::iterator ReservedIterator) {
ReservedVMARegion* ReservedRegion = *ReservedIterator;
ReservedRegions->erase(ReservedIterator);
// mprotect the new region we've allocated
size_t SizeOfLiveRegion =
const size_t SizeOfLiveRegion =
FEXCore::AlignUp(LiveVMARegion::GetFEXManagedVMARegionSize(ReservedRegion->RegionSize), FEXCore::Utils::FEX_PAGE_SIZE);
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
auto Res = mprotect(reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData, PROT_READ | PROT_WRITE);
auto Res = mprotect(reinterpret_cast<void*>(ReservedRegion->Base), SizeOfLiveRegion, PROT_READ | PROT_WRITE);
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno,
strerror(errno));
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData);
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(ReservedRegion->Base), SizeOfLiveRegion);
LiveVMARegion* LiveRange = new (reinterpret_cast<void*>(ReservedRegion->Base)) LiveVMARegion();
// Copy over the reserved data
LiveRange->SlabInfo = ReservedRegion;
// Initialize VMA
LiveVMARegion::InitializeVMARegionUsed(LiveRange, UsedSize);
LiveVMARegion::InitializeVMARegionUsed(LiveRange);
// Add to our active tracked ranges
auto LiveIter = LiveRegions->emplace_back(LiveRange);
@@ -187,7 +184,7 @@ private:
};
void OSAllocator_64Bit::DetermineVASize() {
size_t Bits = FEXCore::Allocator::DetermineVASize();
size_t Bits = FEXCore::Allocator::GetHostVABits();
uintptr_t Size = 1ULL << Bits;
UPPER_BOUND = Size;
@@ -224,7 +221,7 @@ OSAllocator_64Bit::LiveVMARegion* OSAllocator_64Bit::FindLiveRegionForAddress(ui
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
if (Addr >= ReservedRegion->Base && AddrEnd < RegionEnd) {
// Found one, let's make it active
LiveRegion = MakeRegionActive(it, 0);
LiveRegion = MakeRegionActive(it);
break;
}
}
@@ -394,7 +391,7 @@ again:
size_t lengthPlusManagedData = length + lengthOfLiveRegion;
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
if ((*it)->RegionSize >= lengthPlusManagedData) {
MakeRegionActive(it, 0);
MakeRegionActive(it);
goto again;
}
}
@@ -623,14 +620,14 @@ fextl::unique_ptr<T> make_alloc_unique(FEXCore::Allocator::MemoryRegion& Base, A
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocatorWithRegions(fextl::vector<FEXCore::Allocator::MemoryRegion>& Regions) {
// This is a bit tricky as we can't allocate memory safely except from the Regions provided. Otherwise we might overwrite memory pages we
// don't own. Scan the memory regions and find the smallest one.
FEXCore::Allocator::MemoryRegion& Smallest = Regions[0];
for (auto& it : Regions) {
if (it.Size <= Smallest.Size) {
Smallest = it;
FEXCore::Allocator::MemoryRegion* Smallest = &Regions[0];
for (auto& Region : Regions) {
if (Region.Size <= Smallest->Size) {
Smallest = &Region;
}
}
return make_alloc_unique<OSAllocator_64Bit>(Smallest, Regions);
return make_alloc_unique<OSAllocator_64Bit>(*Smallest, Regions);
}
} // namespace Alloc::OSAllocator
+2 -2
View File
@@ -39,10 +39,10 @@ struct FlexBitSet final {
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
}
void MemClear(size_t Elements) {
memset(Memory, 0, FEXCore::AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
memset(Memory, 0, SizeInBytes(Elements));
}
void MemSet(size_t Elements) {
memset(Memory, 0xFF, FEXCore::AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
memset(Memory, 0xFF, SizeInBytes(Elements));
}
// Range scanning results
+2
View File
@@ -1,4 +1,6 @@
// SPDX-License-Identifier: MIT
#include <FEXCore/Utils/CompilerDefs.h>
namespace FEXCore::Assert {
// This function can not be inlined
[[noreturn]]
+6 -3
View File
@@ -103,14 +103,17 @@ struct CodeMap {
static constexpr Entry LoadExternalLibrary = {0xffff'ffff'ffff'ffff, 0xffff'ffff};
struct FEX_PACKED SetExecutableFileId {
Entry Marker = {0xffff'ffff'ffff'ffff, 0xffff'fffe};
static constexpr Entry Marker32 = {0xffff'ffff'ffff'ffff, 0xffff'ffee};
static constexpr Entry Marker64 = {0xffff'ffff'ffff'ffff, 0xffff'ffef};
Entry Marker;
CodeMapFileId ExecutableFileId;
};
struct ParsedContents {
fextl::string Filename;
fextl::set<uint64_t> Blocks;
bool IsExecutable = false;
// 32/64 for executables, nullopt for non-executables (libraries)
std::optional<int> ExecutableBitness;
};
// Follows scheme fileid[-nomb]
@@ -152,7 +155,7 @@ public:
void AppendBlock(const FEXCore::ExecutableFileSectionInfo&, uint64_t Entry);
void AppendLibraryLoad(const FEXCore::ExecutableFileInfo&);
void AppendSetMainExecutable(const FEXCore::ExecutableFileInfo&);
void AppendSetMainExecutable(const FEXCore::ExecutableFileInfo&, bool Is64Bit);
// Thread-safely commit any pending data to disk
void Flush(size_t Offset);
+1 -1
View File
@@ -12,7 +12,7 @@ struct InternalThreadState;
}
namespace FEXCore::Allocator {
FEX_DEFAULT_VISIBILITY size_t DetermineVASize();
FEX_DEFAULT_VISIBILITY size_t GetHostVABits();
#ifdef GLIBC_ALLOCATOR_FAULT
// Glibc hooks should only fault once we are in main.
@@ -97,7 +97,8 @@ inline bool VirtualProtect(void* Ptr, size_t Size, ProtectOptions options) {
LOGMAN_MSG_A_FMT("Unknown VirtualProtect options combination");
}
return ::VirtualProtect(Ptr, Size, prot, nullptr) == 0;
DWORD OldProt {};
return ::VirtualProtect(Ptr, Size, prot, &OldProt) != 0;
}
FEX_DEFAULT_VISIBILITY extern VirtualNamePtr VirtualName;
+13 -5
View File
@@ -58,6 +58,8 @@ public:
}
IsValidHandle = Handle != INVALID_HANDLE_VALUE;
#endif
ShouldClose = IsValidHandle;
}
/**
@@ -202,12 +204,18 @@ private:
static constexpr int DEFAULT_USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
static uint32_t TranslateModes(FileModes Modes) {
const auto IsRead = (Modes & FileModes::READ) == FileModes::READ;
const auto IsWrite = (Modes & FileModes::WRITE) == FileModes::WRITE;
uint32_t Mode {};
if ((Modes & FileModes::READ) == FileModes::READ) {
Mode |= O_RDONLY;
}
if ((Modes & FileModes::WRITE) == FileModes::WRITE) {
Mode |= O_WRONLY;
if (IsRead && IsWrite) {
Mode |= O_RDWR;
} else {
if (IsRead) {
Mode |= O_RDONLY;
} else if (IsWrite) {
Mode |= O_WRONLY;
}
}
if ((Modes & FileModes::CREATE) == FileModes::CREATE) {
Mode |= O_CREAT;
+5 -2
View File
@@ -2,8 +2,7 @@
#pragma once
#ifndef _WIN32
#include <sys/mman.h>
#include <sys/user.h>
#include <sys/prctl.h>
#ifndef PR_SET_VMA
@@ -50,4 +49,8 @@
#define PR_SHADOW_STACK_ENABLE (1ULL << 0)
#endif
#ifndef PR_GET_MDWE
#define PR_GET_MDWE 66
#endif
#endif // ifndef _WIN32
@@ -144,6 +144,21 @@ static inline void WaitPred(T* Futex, T ComparisonValue) {
}
}
template<typename T, typename Pred>
static inline void WaitBitMaskPred(T* Futex, T BitMask, T ComparisonValue, Pred Predicate) {
auto AtomicFutex = std::atomic_ref<T>(*Futex);
T Result = AtomicFutex.load();
while (!Predicate(Result & BitMask, ComparisonValue)) {
Result = LoadExclusive(Futex);
if (Predicate(Result & BitMask, ComparisonValue)) {
return;
}
Result = WFELoadAtomic(Futex);
}
}
template<typename T, typename TT>
static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanoseconds& Timeout) {
auto AtomicFutex = std::atomic_ref<T>(*Futex);
@@ -213,6 +228,16 @@ static inline void WaitPred(T* Futex, T ComparisonValue) {
}
}
template<typename T, typename Pred>
static inline void WaitBitMaskPred(T* Futex, T BitMask, T ComparisonValue, Pred Predicate) {
auto AtomicFutex = std::atomic_ref<T>(*Futex);
T Result = AtomicFutex.load();
while (!Predicate(Result & BitMask, ComparisonValue)) {
Result = AtomicFutex.load();
}
}
template<typename T, typename TT>
static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanoseconds& Timeout) {
auto AtomicFutex = std::atomic_ref<T>(*Futex);
+8 -8
View File
@@ -5,21 +5,21 @@
namespace FEXCore::StringUtils {
// Trim the left side of the string of whitespace and new lines
inline fextl::string LeftTrim(fextl::string String, std::string_view TrimTokens = " \t\n\r\f\v") {
size_t pos = fextl::string::npos;
if ((pos = String.find_first_not_of(TrimTokens)) != fextl::string::npos) {
String.erase(0, pos);
const size_t pos = String.find_first_not_of(TrimTokens);
if (pos == fextl::string::npos) {
return "";
}
return String;
return String.erase(0, pos);
}
// Trim the right side of the string of whitespace and new lines
inline fextl::string RightTrim(fextl::string String, std::string_view TrimTokens = " \t\n\r\f\v") {
size_t pos = fextl::string::npos;
if ((pos = String.find_last_not_of(TrimTokens)) != fextl::string::npos) {
String.erase(String.begin() + pos + 1, String.end());
const size_t pos = String.find_last_not_of(TrimTokens);
if (pos == fextl::string::npos) {
return "";
}
String.erase(String.begin() + pos + 1, String.end());
return String;
}
+9 -8
View File
@@ -6,6 +6,7 @@
#include <fmt/format.h>
#include <fmt/ranges.h>
#include <fmt/std.h>
#include <unistd.h>
namespace fextl::fmt {
@@ -15,7 +16,7 @@ using memory_buffer = fextl::fmt::basic_memory_buffer<char>;
template<class OutputIt, class... Args>
OutputIt format_to(OutputIt out, ::fmt::format_string<Args...> fmt, Args&&... args) {
return ::fmt::vformat_to(out, fmt.str, ::fmt::make_format_args(args...));
return ::fmt::vformat_to(out, fmt.get(), ::fmt::make_format_args(args...));
}
template<typename Char, size_t SIZE>
@@ -35,44 +36,44 @@ FMT_INLINE fextl::string vformat(::fmt::string_view fmt, ::fmt::format_args args
template<typename... T>
FMT_NODISCARD FMT_INLINE auto format(::fmt::format_string<T...> fmt, T&&... args) -> fextl::string {
return fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
return fextl::fmt::vformat(fmt.get(), ::fmt::make_format_args(args...));
}
#ifndef _WIN32
template<typename... T>
FMT_INLINE auto print(::fmt::format_string<T...> fmt, T&&... args) -> void {
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
auto String = fextl::fmt::vformat(fmt.get(), ::fmt::make_format_args(args...));
write(STDOUT_FILENO, String.c_str(), String.size());
}
template<typename... T>
FMT_INLINE auto print(int FD, ::fmt::format_string<T...> fmt, T&&... args) -> void {
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
auto String = fextl::fmt::vformat(fmt.get(), ::fmt::make_format_args(args...));
write(FD, String.c_str(), String.size());
}
#else
template<typename... T>
FMT_INLINE auto print(::fmt::format_string<T...> fmt, T&&... args) -> void {
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
auto String = fextl::fmt::vformat(fmt.get(), ::fmt::make_format_args(args...));
auto f = FEXCore::File::File::GetStdOUT();
f.Write(String.c_str(), String.size());
}
template<typename... T>
FMT_INLINE auto print(HANDLE File, ::fmt::format_string<T...> fmt, T&&... args) -> void {
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
auto String = fextl::fmt::vformat(fmt.get(), ::fmt::make_format_args(args...));
WriteFile(File, String.c_str(), String.size(), nullptr, nullptr);
}
#endif
template<typename... T>
FMT_INLINE auto print(FEXCore::File::File& f, ::fmt::format_string<T...> fmt, T&&... args) -> void {
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
auto String = fextl::fmt::vformat(fmt.get(), ::fmt::make_format_args(args...));
f.Write(String.c_str(), String.size());
}
template<typename... T>
FMT_INLINE auto print(std::FILE* f, ::fmt::format_string<T...> fmt, T&&... args) -> void {
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
auto String = fextl::fmt::vformat(fmt.get(), ::fmt::make_format_args(args...));
write(fileno(f), String.c_str(), String.size());
}
} // namespace fextl::fmt
+2 -1
View File
@@ -391,7 +391,8 @@ inline char* Absolute(const char* Path, char Fill[PATH_MAX]) {
std::error_code ec;
const auto PathAbsolute = std::filesystem::absolute(Path, ec);
if (!ec) {
strncpy(Fill, PathAbsolute.string().c_str(), sizeof(*Fill));
auto end = PathAbsolute.string().copy(Fill, PATH_MAX - 1);
Fill[end] = '\0';
return Fill;
}
+4 -1
View File
@@ -71,8 +71,10 @@ private:
public:
~poll_reactor() {
if (AsyncStopRequest[0]) {
if (AsyncStopRequest[0] != -1) {
::close(AsyncStopRequest[0]);
}
if (AsyncStopRequest[1] != -1) {
::close(AsyncStopRequest[1]);
}
}
@@ -92,6 +94,7 @@ public:
}
// Wake up run() thread by closing this pipe endpoint
::close(AsyncStopRequest[1]);
AsyncStopRequest[1] = -1;
}
void cleanup() {
+2 -1
View File
@@ -14,7 +14,8 @@ if (NOT MINGW)
list(APPEND SRCS
FEXServerClient.cpp
FileFormatCheck.cpp
Linux/SBRKAllocations.cpp)
Linux/SBRKAllocations.cpp
Linux/LinuxVersion.cpp)
endif()
add_library(${NAME} STATIC ${SRCS})
+14 -8
View File
@@ -186,6 +186,7 @@ public:
explicit MainLoader(FEXCore::Config::LayerType Type, std::optional<fextl::string> AppName = std::nullopt);
explicit MainLoader(fextl::string ConfigFile, std::optional<fextl::string> AppName = std::nullopt);
explicit MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigFile);
explicit MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigFile, std::optional<fextl::string> AppName = std::nullopt);
void Load() override;
@@ -197,7 +198,7 @@ private:
class AppLoader final : public OptionMapper {
public:
explicit AppLoader(const fextl::string& AppName, FEXCore::Config::LayerType Type);
void Load();
void Load() override;
private:
const fextl::string AppName;
@@ -241,12 +242,12 @@ void OptionMapper::MapNameToOption(const char* ConfigName, const char* ConfigStr
MainLoader::MainLoader(FEXCore::Config::LayerType Type, std::optional<fextl::string> AppName)
: OptionMapper(Type)
, AppName {AppName}
, AppName {std::move(AppName)}
, Config {FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {}
MainLoader::MainLoader(fextl::string ConfigFile, std::optional<fextl::string> AppName)
: OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
, AppName {AppName}
, AppName {std::move(AppName)}
, Config {std::move(ConfigFile)} {}
@@ -254,6 +255,11 @@ MainLoader::MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigF
: OptionMapper(Type)
, Config {ConfigFile} {}
MainLoader::MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigFile, std::optional<fextl::string> AppName)
: OptionMapper(Type)
, AppName {std::move(AppName)}
, Config {ConfigFile} {}
void MainLoader::Load() {
SetCurrentConfigFile(Config);
JSON::LoadJSonConfig(Config, AppName, [this](const char* Name, const char* ConfigString) { MapNameToOption(Name, ConfigString); });
@@ -349,8 +355,8 @@ fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(const fextl::string* F
}
}
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(std::string_view AppConfig) {
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_USER_OVERRIDE, AppConfig);
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(std::string_view AppConfig, std::optional<fextl::string> AppName) {
return fextl::make_unique<MainLoader>(FEXCore::Config::LayerType::LAYER_USER_OVERRIDE, AppConfig, std::move(AppName));
}
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type) {
@@ -361,7 +367,7 @@ fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char* const _en
return fextl::make_unique<EnvLoader>(_envp);
}
fextl::string RecoverGuestProgramFilename(fextl::string Program, bool ExecFDInterp, int ProgramFDFromEnv) {
static fextl::string RecoverGuestProgramFilename(fextl::string Program, bool ExecFDInterp, int ProgramFDFromEnv) {
// If executed with a FEX FD then the Program argument might be empty.
// In this case we need to scan the FD node to recover the application binary that exists on disk.
// Only do this if the Program argument is empty, since we would prefer the application's expectation
@@ -510,7 +516,7 @@ void LoadConfig(fextl::string ProgramName, char** const envp, const PortableInfo
}
if (FHU::Filesystem::Exists(AppConfigStr)) {
FEXCore::Config::AddLayer(CreateUserOverrideLayer(AppConfigStr));
FEXCore::Config::AddLayer(CreateUserOverrideLayer(AppConfigStr, ProgramName.empty() ? std::nullopt : std::optional {ProgramName}));
}
}
@@ -519,7 +525,7 @@ void LoadConfig(fextl::string ProgramName, char** const envp, const PortableInfo
}
#ifndef _WIN32
fextl::string FindUserHomeThroughUID() {
static fextl::string FindUserHomeThroughUID() {
// `getpwuid` allocates memory, parse `/etc/passwd` manually.
// Format is trivial: `<name>:<password hash>:<uid>:<gid>:<comment>:<home>:<shell>`
+2 -2
View File
@@ -59,7 +59,7 @@ void LoadConfig(fextl::string ProgramName = {}, char** const envp = nullptr, con
fextl::string GetHomeDirectory();
fextl::string GetDataDirectory(const PortableInformation& PortableInfo);
fextl::string GetDataDirectory(bool Global, const PortableInformation& PortableInfo);
fextl::string GetConfigDirectory(bool Global, const PortableInformation& PortableInfo);
fextl::string GetConfigFileLocation(bool Global, const PortableInformation& PortableInfo);
fextl::string GetCacheDirectory();
@@ -82,7 +82,7 @@ fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer();
* @return unique_ptr for that layer
*/
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(const fextl::string* File = nullptr, std::optional<fextl::string> AppName = std::nullopt);
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(std::string_view AppConfig);
fextl::unique_ptr<FEXCore::Config::Layer> CreateUserOverrideLayer(std::string_view AppConfig, std::optional<fextl::string> AppName);
/**
* @brief Create an application configuration loader
+100 -231
View File
@@ -1,6 +1,9 @@
// SPDX-License-Identifier: MIT
#include "Common/CPUInfo.h"
#include "Common/HostFeatures.h"
#ifndef _WIN32
#include "Common/Linux/LinuxVersion.h"
#endif
#include <FEXCore/Config/Config.h>
#include <FEXCore/Core/HostFeatures.h>
@@ -69,7 +72,8 @@ GetSysReg(MIDR_EL1, MIDR_EL1);
GetSysReg(ISAR1_EL1, ID_AA64ISAR1_EL1);
GetSysReg(MMFR0_EL1, ID_AA64MMFR0_EL1);
GetSysReg(MMFR2_EL1, ID_AA64MMFR2_EL1);
GetSysReg(ZFR0_EL1, s3_0_c0_c4_4); // Can't request by name
GetSysReg(MMFR3_EL1, s3_0_c0_c7_3); // Can't request by name
GetSysReg(ZFR0_EL1, s3_0_c0_c4_4); // Can't request by name
GetSysReg(MMFR1_EL1, ID_AA64MMFR1_EL1);
GetSysReg(ISAR2_EL1, ID_AA64ISAR2_EL1);
GetSysReg(DCZID_EL0, DCZID_EL0);
@@ -84,6 +88,12 @@ public:
ISAR1.SetReg(Get_ISAR1_EL1());
MMFR0.SetReg(Get_MMFR0_EL1());
MMFR2.SetReg(Get_MMFR2_EL1());
#ifndef _WIN32
if (FEX::LinuxVersion::CalculateHostKernelVersion() >= FEX::LinuxVersion::KernelVersion(6, 5)) {
// Only exists in kernel 6.5 and newer.
MMFR3.SetReg(Get_MMFR3_EL1());
}
#endif
MMFR1.SetReg(Get_MMFR1_EL1());
ISAR2.SetReg(Get_ISAR2_EL1());
DCZID.SetReg(Get_DCZID_EL0());
@@ -158,6 +168,8 @@ public:
MMFR1.SetReg(ValueHex);
} else if (Key == "mmfr2") {
MMFR2.SetReg(ValueHex);
} else if (Key == "mmfr3") {
MMFR3.SetReg(ValueHex);
} else if (Key == "zfr0") {
ZFR0.SetReg(ValueHex);
} else if (Key == "dczid") {
@@ -191,63 +203,65 @@ public:
};
void FEX::CPUFeatures::FillFeatureFlags() {
// ISAR0
if (ISAR0.SupportsAES()) {
SetFeature(Feature::AES);
}
if (ISAR0.SupportsPMULL()) {
SetFeature(Feature::PMULL);
}
if (ISAR0.SupportsSHA1()) {
SetFeature(Feature::SHA1);
}
if (ISAR0.SupportsSHA2()) {
SetFeature(Feature::SHA2);
}
if (ISAR0.SupportsSHA512()) {
SetFeature(Feature::SHA512);
}
if (ISAR0.SupportsCRC32()) {
SetFeature(Feature::CRC32);
}
if (ISAR0.SupportsLSE()) {
SetFeature(Feature::LSE);
}
if (ISAR0.SupportsLSE128()) {
SetFeature(Feature::LSE128);
}
if (ISAR0.SupportsTME()) {
SetFeature(Feature::TME);
}
if (ISAR0.SupportsRDM()) {
SetFeature(Feature::RDM);
}
if (ISAR0.SupportsSHA3()) {
SetFeature(Feature::SHA3);
}
if (ISAR0.SupportsSM3()) {
SetFeature(Feature::SM3);
}
if (ISAR0.SupportsSM4()) {
SetFeature(Feature::SM4);
}
if (ISAR0.SupportsDotProd()) {
SetFeature(Feature::DotProd);
}
if (ISAR0.SupportsFlagM()) {
SetFeature(Feature::FlagM);
}
if (ISAR0.SupportsFlagM2()) {
SetFeature(Feature::FlagM2);
}
if (ISAR0.SupportsRNDR()) {
SetFeature(Feature::RNDR);
#define ENABLE_FEATURE_IF(Reg, FeatureName) \
if ((Reg).Supports##FeatureName()) { \
SetFeature(Feature::FeatureName); \
}
// ISAR0
ENABLE_FEATURE_IF(ISAR0, AES);
ENABLE_FEATURE_IF(ISAR0, PMULL);
ENABLE_FEATURE_IF(ISAR0, SHA1);
ENABLE_FEATURE_IF(ISAR0, SHA2);
ENABLE_FEATURE_IF(ISAR0, SHA512);
ENABLE_FEATURE_IF(ISAR0, CRC32);
ENABLE_FEATURE_IF(ISAR0, LSE);
ENABLE_FEATURE_IF(ISAR0, LSE128);
ENABLE_FEATURE_IF(ISAR0, TME);
ENABLE_FEATURE_IF(ISAR0, RDM);
ENABLE_FEATURE_IF(ISAR0, SHA3);
ENABLE_FEATURE_IF(ISAR0, SM3);
ENABLE_FEATURE_IF(ISAR0, SM4);
ENABLE_FEATURE_IF(ISAR0, DotProd);
ENABLE_FEATURE_IF(ISAR0, FlagM);
ENABLE_FEATURE_IF(ISAR0, FlagM2);
ENABLE_FEATURE_IF(ISAR0, RNDR);
// ISAR1
ENABLE_FEATURE_IF(ISAR1, DPB);
ENABLE_FEATURE_IF(ISAR1, DPB2);
ENABLE_FEATURE_IF(ISAR1, JSCVT);
ENABLE_FEATURE_IF(ISAR1, FCMA);
ENABLE_FEATURE_IF(ISAR1, LRCPC);
ENABLE_FEATURE_IF(ISAR1, LRCPC2);
ENABLE_FEATURE_IF(ISAR1, LRCPC3);
ENABLE_FEATURE_IF(ISAR1, FRINTTS);
ENABLE_FEATURE_IF(ISAR1, SB);
ENABLE_FEATURE_IF(ISAR1, SPECRES);
ENABLE_FEATURE_IF(ISAR1, SPECRES2);
ENABLE_FEATURE_IF(ISAR1, BF16);
ENABLE_FEATURE_IF(ISAR1, SME_F64F64);
ENABLE_FEATURE_IF(ISAR1, I8MM);
ENABLE_FEATURE_IF(ISAR1, XS);
ENABLE_FEATURE_IF(ISAR1, LS64);
ENABLE_FEATURE_IF(ISAR1, LS64_V);
ENABLE_FEATURE_IF(ISAR1, LS64_ACCDATA);
// ISAR2
ENABLE_FEATURE_IF(ISAR2, WFxt);
ENABLE_FEATURE_IF(ISAR2, RPRES);
ENABLE_FEATURE_IF(ISAR2, PACQARMA3);
ENABLE_FEATURE_IF(ISAR2, MOPS);
ENABLE_FEATURE_IF(ISAR2, HBC);
ENABLE_FEATURE_IF(ISAR2, CLRBHB);
ENABLE_FEATURE_IF(ISAR2, SYSREG128);
ENABLE_FEATURE_IF(ISAR2, SYSINSTR128);
ENABLE_FEATURE_IF(ISAR2, PRFMSLC);
ENABLE_FEATURE_IF(ISAR2, RPRFM);
ENABLE_FEATURE_IF(ISAR2, CSSC);
// PFR0
if (PFR0.SupportsFP()) {
SetFeature(Feature::FP);
}
ENABLE_FEATURE_IF(PFR0, FP);
if (PFR0.SupportsHP()) {
SetFeature(Feature::FP16);
}
@@ -257,193 +271,48 @@ void FEX::CPUFeatures::FillFeatureFlags() {
if (PFR0.SupportsASIMDHP()) {
SetFeature(Feature::ASIMD16);
}
if (PFR0.SupportsRAS()) {
SetFeature(Feature::RAS);
}
if (PFR0.SupportsSVE()) {
SetFeature(Feature::SVE);
}
if (PFR0.SupportsDIT()) {
SetFeature(Feature::DIT);
}
if (PFR0.SupportsCSV2()) {
SetFeature(Feature::CSV2);
}
if (PFR0.SupportsCSV3()) {
SetFeature(Feature::CSV3);
}
ENABLE_FEATURE_IF(PFR0, RAS);
ENABLE_FEATURE_IF(PFR0, SVE);
ENABLE_FEATURE_IF(PFR0, DIT);
ENABLE_FEATURE_IF(PFR0, CSV2);
ENABLE_FEATURE_IF(PFR0, CSV3);
// PFR1
if (PFR1.SupportsBTI()) {
SetFeature(Feature::BTI);
}
if (PFR1.SupportsSSBS()) {
SetFeature(Feature::SSBS);
}
if (PFR1.SupportsSSBS()) {
SetFeature(Feature::SSBS2);
}
if (PFR1.SupportsMTE()) {
SetFeature(Feature::MTE);
}
if (PFR1.SupportsMTE2()) {
SetFeature(Feature::MTE2);
}
if (PFR1.SupportsMTE3()) {
SetFeature(Feature::MTE3);
}
if (PFR1.SupportsSME()) {
SetFeature(Feature::SME);
}
if (PFR1.SupportsSME2()) {
SetFeature(Feature::SME2);
}
// ISAR1
if (ISAR1.SupportsDPB()) {
SetFeature(Feature::DPB);
}
if (ISAR1.SupportsDPB2()) {
SetFeature(Feature::DPB2);
}
if (ISAR1.SupportsJSCVT()) {
SetFeature(Feature::JSCVT);
}
if (ISAR1.SupportsFCMA()) {
SetFeature(Feature::FCMA);
}
if (ISAR1.SupportsLRCPC()) {
SetFeature(Feature::LRCPC);
}
if (ISAR1.SupportsLRCPC2()) {
SetFeature(Feature::LRCPC2);
}
if (ISAR1.SupportsLRCPC3()) {
SetFeature(Feature::LRCPC3);
}
if (ISAR1.SupportsFRINTTS()) {
SetFeature(Feature::FRINTTS);
}
if (ISAR1.SupportsSB()) {
SetFeature(Feature::SB);
}
if (ISAR1.SupportsSPECRES()) {
SetFeature(Feature::SPECRES);
}
if (ISAR1.SupportsSPECRES2()) {
SetFeature(Feature::SPECRES2);
}
if (ISAR1.SupportsBF16()) {
SetFeature(Feature::BF16);
}
if (ISAR1.SupportsSME_F64F64()) {
SetFeature(Feature::SME_F64F64);
}
if (ISAR1.SupportsI8MM()) {
SetFeature(Feature::I8MM);
}
if (ISAR1.SupportsXS()) {
SetFeature(Feature::XS);
}
if (ISAR1.SupportsLS64()) {
SetFeature(Feature::LS64);
}
if (ISAR1.SupportsLS64_V()) {
SetFeature(Feature::LS64_V);
}
if (ISAR1.SupportsLS64_ACCDATA()) {
SetFeature(Feature::LS64_ACCDATA);
}
ENABLE_FEATURE_IF(PFR1, BTI);
ENABLE_FEATURE_IF(PFR1, SSBS);
ENABLE_FEATURE_IF(PFR1, SSBS2);
ENABLE_FEATURE_IF(PFR1, MTE);
ENABLE_FEATURE_IF(PFR1, MTE2);
ENABLE_FEATURE_IF(PFR1, MTE3);
ENABLE_FEATURE_IF(PFR1, SME);
ENABLE_FEATURE_IF(PFR1, SME2);
// MMFR0
if (MMFR0.SupportsECV()) {
SetFeature(Feature::ECV);
}
ENABLE_FEATURE_IF(MMFR0, ECV);
// MMFR1
ENABLE_FEATURE_IF(MMFR1, AFP);
// MMFR2
if (MMFR2.SupportsLSE2()) {
SetFeature(Feature::LSE2);
}
ENABLE_FEATURE_IF(MMFR2, LSE2);
// ZFR0
if (Supports(Feature::SVE)) {
if (ZFR0.SupportsSVE2()) {
SetFeature(Feature::SVE2);
}
if (ZFR0.SupportsSVE2_1()) {
SetFeature(Feature::SVE2_1);
}
if (ZFR0.SupportsSVE_AES()) {
SetFeature(Feature::SVE_AES);
}
if (ZFR0.SupportsSVE_PMULL128()) {
SetFeature(Feature::SVE_PMULL128);
}
if (ZFR0.SupportsSVE_BitPerm()) {
SetFeature(Feature::SVE_BitPerm);
}
if (ZFR0.SupportsSVE_BF16()) {
SetFeature(Feature::SVE_BF16);
}
if (ZFR0.SupportsSVE_B16B16()) {
SetFeature(Feature::SVE_B16B16);
}
if (ZFR0.SupportsSVE_SHA3()) {
SetFeature(Feature::SVE_SHA3);
}
if (ZFR0.SupportsSVE_SM4()) {
SetFeature(Feature::SVE_SM4);
}
if (ZFR0.SupportsSVE_I8MM()) {
SetFeature(Feature::SVE_I8MM);
}
if (ZFR0.SupportsSVE_F32MM()) {
SetFeature(Feature::SVE_F32MM);
}
if (ZFR0.SupportsSVE_F64MM()) {
SetFeature(Feature::SVE_F64MM);
}
ENABLE_FEATURE_IF(ZFR0, SVE2);
ENABLE_FEATURE_IF(ZFR0, SVE2_1);
ENABLE_FEATURE_IF(ZFR0, SVE_AES);
ENABLE_FEATURE_IF(ZFR0, SVE_PMULL128);
ENABLE_FEATURE_IF(ZFR0, SVE_BitPerm);
ENABLE_FEATURE_IF(ZFR0, SVE_BF16);
ENABLE_FEATURE_IF(ZFR0, SVE_B16B16);
ENABLE_FEATURE_IF(ZFR0, SVE_SHA3);
ENABLE_FEATURE_IF(ZFR0, SVE_SM4);
ENABLE_FEATURE_IF(ZFR0, SVE_I8MM);
ENABLE_FEATURE_IF(ZFR0, SVE_F32MM);
ENABLE_FEATURE_IF(ZFR0, SVE_F64MM);
}
// MMFR1
if (MMFR1.SupportsAFP()) {
SetFeature(Feature::AFP);
}
// ISAR2
if (ISAR2.SupportsWFxt()) {
SetFeature(Feature::WFxt);
}
if (ISAR2.SupportsRPRES()) {
SetFeature(Feature::RPRES);
}
if (ISAR2.SupportsPACQARMA3()) {
SetFeature(Feature::PACQARMA3);
}
if (ISAR2.SupportsMOPS()) {
SetFeature(Feature::MOPS);
}
if (ISAR2.SupportsHBC()) {
SetFeature(Feature::HBC);
}
if (ISAR2.SupportsCLRBHB()) {
SetFeature(Feature::CLRBHB);
}
if (ISAR2.SupportsSYSREG128()) {
SetFeature(Feature::SYSREG128);
}
if (ISAR2.SupportsSYSINSTR128()) {
SetFeature(Feature::SYSINSTR128);
}
if (ISAR2.SupportsPRFMSLC()) {
SetFeature(Feature::PRFMSLC);
}
if (ISAR2.SupportsRPRFM()) {
SetFeature(Feature::RPRFM);
}
if (ISAR2.SupportsCSSC()) {
SetFeature(Feature::CSSC);
}
#undef ENABLE_FEATURE_IF
}
#ifdef ARCHITECTURE_arm64
+28
View File
@@ -101,6 +101,8 @@ public:
SVE_F64MM,
// MMFR1
AFP,
// MMFR3
S1POE,
// ISAR2
WFxt,
RPRES,
@@ -148,6 +150,7 @@ public:
MMFR2_EL1,
ZFR0_EL1,
MMFR1_EL1,
MMFR3_EL1,
ISAR2_EL1,
};
@@ -559,6 +562,30 @@ public:
};
};
class MMFR3Reg final : public FeatureReg {
public:
FIELD_FETCHER(S1POE, S1POE, 0b0001);
private:
enum Field {
TCRX = 0 * 4,
SCTLRX = 1 * 4,
S1PIE = 2 * 4,
S2PIE = 3 * 4,
S1POE = 4 * 4,
S2POE = 5 * 4,
AIE = 6 * 4,
MEC = 7 * 4,
D128 = 8 * 4,
D128_2 = 9 * 4,
SNERR = 10 * 4,
ANERR = 11 * 4,
SDERR = 13 * 4,
ADERR = 14 * 4,
SPEC_FPACC = 15 * 4,
};
};
class ISAR2Reg final : public FeatureReg {
public:
FIELD_FETCHER(WFxt, WFxt, 0b0010);
@@ -621,6 +648,7 @@ public:
ZFR0Reg ZFR0;
MMFR2Reg MMFR2;
MMFR1Reg MMFR1;
MMFR3Reg MMFR3;
ISAR2Reg ISAR2;
DCZIDReg DCZID;
SVEVLReg SVEVL;
+25
View File
@@ -0,0 +1,25 @@
// SPDX-License-Identifier: MIT
#include <cstdint>
#include <charconv>
#include <sys/utsname.h>
namespace FEX::LinuxVersion {
uint32_t CalculateHostKernelVersion() {
struct utsname buf {};
if (uname(&buf) == -1) {
return 0;
}
uint32_t Major {};
uint32_t Minor {};
uint32_t Patch {};
// Parse kernel version in the form of `<Major>.<Minor>.<Patch>[Optional Data]`
const auto End = buf.release + sizeof(buf.release);
auto Results = std::from_chars(buf.release, End, Major, 10);
Results = std::from_chars(Results.ptr + 1, End, Minor, 10);
Results = std::from_chars(Results.ptr + 1, End, Patch, 10);
return (Major << 24) | (Minor << 16) | Patch;
}
} // namespace FEX::LinuxVersion
+22
View File
@@ -0,0 +1,22 @@
// SPDX-License-Identifier: MIT
#pragma once
#include <cstdint>
namespace FEX::LinuxVersion {
inline uint32_t KernelVersion(uint32_t Major, uint32_t Minor = 0, uint32_t Patch = 0) {
return (Major << 24) | (Minor << 16) | Patch;
}
inline uint32_t KernelMajor(uint32_t Version) {
return Version >> 24;
}
inline uint32_t KernelMinor(uint32_t Version) {
return (Version >> 16) & 0xFF;
}
inline uint32_t KernelPatch(uint32_t Version) {
return Version & 0xFFFF;
}
uint32_t CalculateHostKernelVersion();
} // namespace FEX::LinuxVersion
@@ -611,13 +611,13 @@ void ELFContainer::CalculateSymbols() {
void ELFContainer::GetDynamicLibs() {
if (Mode == MODE_32BIT) {
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
const Elf32_Shdr* hdr = SectionHeaders.at(i)._32;
const Elf32_Shdr* hdr = SectionHeaders[i]._32;
if (hdr->sh_type == SHT_DYNAMIC) {
const Elf32_Shdr* StrHeader = SectionHeaders.at(hdr->sh_link)._32;
const char* SHStrings = &RawFile.at(StrHeader->sh_offset);
size_t Entries = hdr->sh_size / hdr->sh_entsize;
for (size_t j = 0; i < Entries; ++j) {
for (size_t j = 0; j < Entries; ++j) {
const Elf32_Dyn* Dynamic = reinterpret_cast<const Elf32_Dyn*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize));
if (Dynamic->d_tag == DT_NULL) {
break;
@@ -630,13 +630,13 @@ void ELFContainer::GetDynamicLibs() {
}
} else {
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
const Elf64_Shdr* hdr = SectionHeaders.at(i)._64;
const Elf64_Shdr* hdr = SectionHeaders[i]._64;
if (hdr->sh_type == SHT_DYNAMIC) {
const Elf64_Shdr* StrHeader = SectionHeaders.at(hdr->sh_link)._64;
const char* SHStrings = &RawFile.at(StrHeader->sh_offset);
size_t Entries = hdr->sh_size / hdr->sh_entsize;
for (size_t j = 0; i < Entries; ++j) {
for (size_t j = 0; j < Entries; ++j) {
const Elf64_Dyn* Dynamic = reinterpret_cast<const Elf64_Dyn*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize));
if (Dynamic->d_tag == DT_NULL) {
break;
@@ -813,10 +813,10 @@ void ELFContainer::GetInitLocations(uint64_t GuestELFBase, fextl::vector<uint64_
if (Mode == MODE_32BIT) {
// If INIT exists then add that first
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
const Elf32_Shdr* hdr = SectionHeaders.at(i)._32;
const Elf32_Shdr* hdr = SectionHeaders[i]._32;
if (hdr->sh_type == SHT_DYNAMIC) {
size_t Entries = hdr->sh_size / hdr->sh_entsize;
for (size_t j = 0; i < Entries; ++j) {
for (size_t j = 0; j < Entries; ++j) {
const Elf32_Dyn* Dynamic = reinterpret_cast<const Elf32_Dyn*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize));
if (Dynamic->d_tag == DT_NULL) {
break;
@@ -830,7 +830,7 @@ void ELFContainer::GetInitLocations(uint64_t GuestELFBase, fextl::vector<uint64_
// Fill init_array
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
const Elf32_Shdr* hdr = SectionHeaders.at(i)._32;
const Elf32_Shdr* hdr = SectionHeaders[i]._32;
if (hdr->sh_type == SHT_INIT_ARRAY) {
size_t Entries = hdr->sh_size / hdr->sh_entsize;
for (size_t j = 0; j < Entries; ++j) {
@@ -841,10 +841,10 @@ void ELFContainer::GetInitLocations(uint64_t GuestELFBase, fextl::vector<uint64_
} else {
// If INIT exists then add that first
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
const Elf64_Shdr* hdr = SectionHeaders.at(i)._64;
const Elf64_Shdr* hdr = SectionHeaders[i]._64;
if (hdr->sh_type == SHT_DYNAMIC) {
size_t Entries = hdr->sh_size / hdr->sh_entsize;
for (size_t j = 0; i < Entries; ++j) {
for (size_t j = 0; j < Entries; ++j) {
const Elf64_Dyn* Dynamic = reinterpret_cast<const Elf64_Dyn*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize));
if (Dynamic->d_tag == DT_NULL) {
break;
@@ -858,7 +858,7 @@ void ELFContainer::GetInitLocations(uint64_t GuestELFBase, fextl::vector<uint64_
// Fill init_array
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
const Elf64_Shdr* hdr = SectionHeaders.at(i)._64;
const Elf64_Shdr* hdr = SectionHeaders[i]._64;
if (hdr->sh_type == SHT_INIT_ARRAY) {
size_t Entries = hdr->sh_size / hdr->sh_entsize;
for (size_t j = 0; j < Entries; ++j) {
+140 -130
View File
@@ -94,7 +94,7 @@ TSOEmulationFacts GetTSOEmulationFacts() {
namespace SIGBUSTest {
static bool* FaultArray {};
__attribute__((naked)) void atomic_load_u16(std::byte* Data) {
__attribute__((naked)) static void atomic_load_u16(std::byte* Data) {
asm volatile(R"(
ldarh w1, [x0];
ret;
@@ -102,7 +102,7 @@ __attribute__((naked)) void atomic_load_u16(std::byte* Data) {
: "x1", "memory");
}
__attribute__((naked)) void atomic_load_u32(std::byte* Data) {
__attribute__((naked)) static void atomic_load_u32(std::byte* Data) {
asm volatile(R"(
ldar w1, [x0];
ret;
@@ -110,14 +110,14 @@ __attribute__((naked)) void atomic_load_u32(std::byte* Data) {
: "x1", "memory");
}
__attribute__((naked)) void atomic_load_u64(std::byte* Data) {
__attribute__((naked)) static void atomic_load_u64(std::byte* Data) {
asm volatile(R"(
ldar x1, [x0];
ret;
)" ::
: "x1", "memory");
}
__attribute__((naked)) void atomic_load_u128(std::byte* Data) {
__attribute__((naked)) static void atomic_load_u128(std::byte* Data) {
asm volatile(R"(
ldaxp x1, x2, [x0];
ret;
@@ -125,7 +125,7 @@ __attribute__((naked)) void atomic_load_u128(std::byte* Data) {
: "x1", "x2", "x3", "memory");
}
__attribute__((naked)) void atomic_set_u16(std::byte* Data, uint16_t value) {
__attribute__((naked)) static void atomic_set_u16(std::byte* Data, uint16_t value) {
asm volatile(R"(
.word 0x78e13002; // ldsetalh w1, w2, [x0];
ret;
@@ -133,7 +133,7 @@ __attribute__((naked)) void atomic_set_u16(std::byte* Data, uint16_t value) {
: "memory");
}
__attribute__((naked)) void atomic_set_u32(std::byte* Data, uint32_t value) {
__attribute__((naked)) static void atomic_set_u32(std::byte* Data, uint32_t value) {
asm volatile(R"(
.word 0xb8e13002; // ldsetal w1, w2, [x0];
ret;
@@ -141,14 +141,14 @@ __attribute__((naked)) void atomic_set_u32(std::byte* Data, uint32_t value) {
: "memory");
}
__attribute__((naked)) void atomic_set_u64(std::byte* Data, uint64_t value) {
__attribute__((naked)) static void atomic_set_u64(std::byte* Data, uint64_t value) {
asm volatile(R"(
.word 0xf8e13002; // ldsetal x1, x2, [x0];
ret;
)" ::
: "memory");
}
__attribute__((naked)) void atomic_set_u128_impl(__uint128_t expected, __uint128_t desired, std::byte* Data) {
__attribute__((naked)) static void atomic_set_u128_impl(__uint128_t expected, __uint128_t desired, std::byte* Data) {
asm volatile(R"(
.word 0x4860fc82; // caspal x0, x1, x2, x3, [x4];
ret;
@@ -180,7 +180,7 @@ static bool FaultOffset_RMW_32bit[64] {};
static bool FaultOffset_RMW_64bit[64] {};
static bool FaultOffset_RMW_128bit[64] {};
void RunFaultTests() {
static void RunFaultTests() {
if (CalculatedFaultOffsets) {
return;
}
@@ -225,7 +225,7 @@ void RunFaultTests() {
CalculatedFaultOffsets = true;
}
void PrintSIGBUSInfo() {
static void PrintSIGBUSInfo() {
RunFaultTests();
auto print_granule = [](const char* size, bool* FaultArray) {
@@ -275,7 +275,7 @@ struct FirstFaultInformation {
int32_t RMWFaultAlignment {};
};
FirstFaultInformation CalculateFirstFaultInformation() {
static FirstFaultInformation CalculateFirstFaultInformation() {
RunFaultTests();
FirstFaultInformation Info {};
auto FindFirstFaultOffset = [](bool FaultOffsets[64]) -> int32_t {
@@ -294,6 +294,123 @@ FirstFaultInformation CalculateFirstFaultInformation() {
}
} // namespace SIGBUSTest
static void PrintTSOInfo() {
auto TSOFacts = GetTSOEmulationFacts();
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
const char* GPRMemoryTSOEmulation {};
const char* MemcpyMemoryTSOEmulation {};
const char* VectorMemoryTSOEmulation {};
const char* UnalignedMemoryLoadStoreTSOEmulation {};
const char* SplitLock16BEmulationType {};
const char* SplitLock16BConfigurationType {};
std::string UnalignedMemoryLoadStoreAlignmentGranularity {};
std::string UnalignedRMWAlignmentGranularity {};
if (TSOFacts.HardwareTSO) {
GPRMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
} else if (TSOFacts.LRCPC3) {
GPRMemoryTSOEmulation = "\e[32mLRCPC3\e[0m";
} else if (TSOFacts.LRCPC2) {
GPRMemoryTSOEmulation = "\e[32mLRCPC2\e[0m";
} else if (TSOFacts.LRCPC1) {
GPRMemoryTSOEmulation = "\e[32mLRCPC\e[0m";
} else {
GPRMemoryTSOEmulation = "\e[31mAtomics\e[0m";
}
// Memcpy only uses Hardware TSO, LRCPC, and Atomics.
if (TSOFacts.HardwareTSO) {
MemcpyMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
} else if (TSOFacts.LRCPC1) {
MemcpyMemoryTSOEmulation = "\e[32mLRCPC\e[0m";
} else {
MemcpyMemoryTSOEmulation = "\e[31mAtomics\e[0m";
}
if (TSOFacts.HardwareTSO) {
VectorMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
} else if (TSOFacts.LRCPC3) {
VectorMemoryTSOEmulation = "\e[32mLRCPC3\e[0m";
} else {
VectorMemoryTSOEmulation = "\e[31mHalf-Barriers\e[0m";
}
if (TSOFacts.HardwareTSO) {
UnalignedMemoryLoadStoreTSOEmulation = "\e[32mHardware TSO\e[0m";
} else {
UnalignedMemoryLoadStoreTSOEmulation = "\e[31mHalf-Barriers\e[0m";
}
const auto FFInfo = SIGBUSTest::CalculateFirstFaultInformation();
if (FFInfo.RMWFaultAlignment >= 64) {
SplitLock16BEmulationType = "\e[32mHardware cacheline unaligned atomics\e[0m";
SplitLock16BConfigurationType = "\e[32mTear-free\e[0m";
} else {
SplitLock16BEmulationType = TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m";
SplitLock16BConfigurationType = StrictInProcessSplitLocks() ? "In-process mutex" : "Tearing";
}
if (FFInfo.LoadStoreFaultAlignment != 1) {
UnalignedMemoryLoadStoreAlignmentGranularity = fmt::format("\e[32m{}-byte\e[0m", FFInfo.LoadStoreFaultAlignment);
} else {
UnalignedMemoryLoadStoreAlignmentGranularity = TSOFacts.LSE2 ? "\e[32m16-byte\e[0m" : "\e[31mNatural alignment\e[0m";
}
if (FFInfo.LoadStoreFaultAlignment != FFInfo.RMWFaultAlignment) {
if (FFInfo.LoadStoreFaultAlignment != 1) {
UnalignedRMWAlignmentGranularity = fmt::format("\e[32m{}-byte\e[0m", FFInfo.RMWFaultAlignment);
} else {
UnalignedRMWAlignmentGranularity = TSOFacts.LSE2 ? "\e[32m16-byte\e[0m" : "\e[31mNatural alignment\e[0m";
}
}
fprintf(stdout, "Hardware Features:\n");
fprintf(stdout, "\tMemory atomics emulation method: %s\n", TSOFacts.LSE ? "\e[32mLSE\e[0m" : "\e[31mLL/SC\e[0m");
fprintf(stdout, "\tUnaligned atomic memory granularity: %s\n", UnalignedMemoryLoadStoreAlignmentGranularity.c_str());
if (FFInfo.LoadStoreFaultAlignment != FFInfo.RMWFaultAlignment) {
fprintf(stdout, "\tUnaligned atomic RMW granularity: %s\n", UnalignedRMWAlignmentGranularity.c_str());
}
fprintf(stdout, "\tUnaligned memory loadstore emulation: %s\n", UnalignedMemoryLoadStoreTSOEmulation);
fprintf(stdout, "\t16-Byte split-lock atomic emulation: %s\n", SplitLock16BEmulationType);
fprintf(stdout, "\t64-Byte split-lock atomic emulation: %s\n", TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m");
fprintf(stdout, "\tGPR memory model emulation: %s\n", GPRMemoryTSOEmulation);
fprintf(stdout, "\tMemcpy memory model emulation: %s\n", MemcpyMemoryTSOEmulation);
fprintf(stdout, "\tVector memory model emulation: %s\n", VectorMemoryTSOEmulation);
fprintf(stdout, "\nConfiguration:\n");
fprintf(stdout, "\tTSO Emulation: %s\n", TSOEnabled() ? "Enabled" : "Disabled");
fprintf(stdout, "\tMemcpy TSO Emulation: %s\n", TSOEnabled() && MemcpySetTSOEnabled() ? "Enabled" : "Disabled");
fprintf(stdout, "\tVector TSO Emulation: %s\n", TSOEnabled() && VectorTSOEnabled() ? "Enabled" : "Disabled");
fprintf(stdout, "\tHalf-barrier unaligned TSO emulation: %s\n", TSOEnabled() && HalfBarrierTSOEnabled() ? "Enabled" : "Disabled");
fprintf(stdout, "\t16-Byte strict split-lock emulation: %s\n", SplitLock16BConfigurationType);
fprintf(stdout, "\t64-Byte strict split-lock emulation: %s\n", StrictInProcessSplitLocks() ? "In-process mutex" : "Tearing");
}
static void PrintIDRegInfo() {
auto Features = FEX::GetCPUFeaturesFromIDRegisters();
fextl::string features {};
features += fmt::format("isar0=0x{:x},", Features.ISAR0.Get());
features += fmt::format("isar1=0x{:x},", Features.ISAR1.Get());
features += fmt::format("isar2=0x{:x},", Features.ISAR2.Get());
features += fmt::format("pfr0=0x{:x},", Features.PFR0.Get());
features += fmt::format("pfr1=0x{:x},", Features.PFR1.Get());
features += fmt::format("midr=0x{:x},", Features.MIDR.Get());
features += fmt::format("mmfr0=0x{:x},", Features.MMFR0.Get());
features += fmt::format("mmfr1=0x{:x},", Features.MMFR1.Get());
features += fmt::format("mmfr2=0x{:x},", Features.MMFR2.Get());
features += fmt::format("mmfr3=0x{:x},", Features.MMFR3.Get());
features += fmt::format("zfr0=0x{:x},", Features.ZFR0.Get());
features += fmt::format("dczid=0x{:x},", Features.DCZID.Get());
features += fmt::format("svevl=0x{:x}", Features.SVEVL.Get());
fprintf(stderr, "Features: '%s'\n", features.c_str());
}
#endif
int main(int argc, char** argv, char** envp) {
@@ -317,6 +434,7 @@ int main(int argc, char** argv, char** envp) {
Parser.add_option("--tso-emulation-info").action("store_true").help("Print how FEX is emulating the x86-TSO memory model.");
Parser.add_option("--test-fault-granularity").action("store_true").help("Show SIGBUS fault granularity");
Parser.add_option("--identification-reg-info").action("store_true").help("Print identification registers");
Parser.add_option("-e", "--all-emu-info").action("store_true").help("Prints all relevant emulation related information");
#endif
Parser.add_option("--version").action("store_true").help("Print the installed FEX-Emu version");
@@ -344,16 +462,12 @@ int main(int argc, char** argv, char** envp) {
// Reload the meta layer
FEXCore::Config::ReloadMetaLayer();
if (Options.is_set_by_user("version")) {
const bool IsAllEmuInfo = Options.is_set_by_user("all_emu_info");
if (IsAllEmuInfo || Options.is_set_by_user("version")) {
fprintf(stdout, GIT_DESCRIBE_STRING "\n");
}
#ifdef ARCHITECTURE_arm64
if (Options.is_set_by_user("test_fault_granularity")) {
SIGBUSTest::PrintSIGBUSInfo();
}
#endif
if (Options.is_set_by_user("install_prefix")) {
char SelfPath[PATH_MAX];
auto Result = readlink("/proc/self/exe", SelfPath, PATH_MAX);
@@ -375,120 +489,16 @@ int main(int argc, char** argv, char** envp) {
}
#ifdef ARCHITECTURE_arm64
if (Options.is_set_by_user("tso_emulation_info")) {
auto TSOFacts = GetTSOEmulationFacts();
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
const char* GPRMemoryTSOEmulation {};
const char* MemcpyMemoryTSOEmulation {};
const char* VectorMemoryTSOEmulation {};
const char* UnalignedMemoryLoadStoreTSOEmulation {};
const char* SplitLock16BEmulationType {};
const char* SplitLock16BConfigurationType {};
std::string UnalignedMemoryLoadStoreAlignmentGranularity {};
std::string UnalignedRMWAlignmentGranularity {};
if (TSOFacts.HardwareTSO) {
GPRMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
} else if (TSOFacts.LRCPC3) {
GPRMemoryTSOEmulation = "\e[32mLRCPC3\e[0m";
} else if (TSOFacts.LRCPC2) {
GPRMemoryTSOEmulation = "\e[32mLRCPC2\e[0m";
} else if (TSOFacts.LRCPC1) {
GPRMemoryTSOEmulation = "\e[32mLRCPC\e[0m";
} else {
GPRMemoryTSOEmulation = "\e[31mAtomics\e[0m";
}
// Memcpy only uses Hardware TSO, LRCPC, and Atomics.
if (TSOFacts.HardwareTSO) {
MemcpyMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
} else if (TSOFacts.LRCPC1) {
MemcpyMemoryTSOEmulation = "\e[32mLRCPC\e[0m";
} else {
MemcpyMemoryTSOEmulation = "\e[31mAtomics\e[0m";
}
if (TSOFacts.HardwareTSO) {
VectorMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
} else if (TSOFacts.LRCPC3) {
VectorMemoryTSOEmulation = "\e[32mLRCPC3\e[0m";
} else {
VectorMemoryTSOEmulation = "\e[31mHalf-Barriers\e[0m";
}
if (TSOFacts.HardwareTSO) {
UnalignedMemoryLoadStoreTSOEmulation = "\e[32mHardware TSO\e[0m";
} else {
UnalignedMemoryLoadStoreTSOEmulation = "\e[31mHalf-Barriers\e[0m";
}
const auto FFInfo = SIGBUSTest::CalculateFirstFaultInformation();
if (FFInfo.RMWFaultAlignment >= 64) {
SplitLock16BEmulationType = "\e[32mHardware cacheline unaligned atomics\e[0m";
SplitLock16BConfigurationType = "\e[32mTear-free\e[0m";
} else {
SplitLock16BEmulationType = TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m";
SplitLock16BConfigurationType = StrictInProcessSplitLocks() ? "In-process mutex" : "Tearing";
}
if (FFInfo.LoadStoreFaultAlignment != 1) {
UnalignedMemoryLoadStoreAlignmentGranularity = fmt::format("\e[32m{}-byte\e[0m", FFInfo.LoadStoreFaultAlignment);
} else {
UnalignedMemoryLoadStoreAlignmentGranularity = TSOFacts.LSE2 ? "\e[32m16-byte\e[0m" : "\e[31mNatural alignment\e[0m";
}
if (FFInfo.LoadStoreFaultAlignment != FFInfo.RMWFaultAlignment) {
if (FFInfo.LoadStoreFaultAlignment != 1) {
UnalignedRMWAlignmentGranularity = fmt::format("\e[32m{}-byte\e[0m", FFInfo.RMWFaultAlignment);
} else {
UnalignedRMWAlignmentGranularity = TSOFacts.LSE2 ? "\e[32m16-byte\e[0m" : "\e[31mNatural alignment\e[0m";
}
}
fprintf(stdout, "Hardware Features:\n");
fprintf(stdout, "\tMemory atomics emulation method: %s\n", TSOFacts.LSE ? "\e[32mLSE\e[0m" : "\e[31mLL/SC\e[0m");
fprintf(stdout, "\tUnaligned atomic memory granularity: %s\n", UnalignedMemoryLoadStoreAlignmentGranularity.c_str());
if (FFInfo.LoadStoreFaultAlignment != FFInfo.RMWFaultAlignment) {
fprintf(stdout, "\tUnaligned atomic RMW granularity: %s\n", UnalignedRMWAlignmentGranularity.c_str());
}
fprintf(stdout, "\tUnaligned memory loadstore emulation: %s\n", UnalignedMemoryLoadStoreTSOEmulation);
fprintf(stdout, "\t16-Byte split-lock atomic emulation: %s\n", SplitLock16BEmulationType);
fprintf(stdout, "\t64-Byte split-lock atomic emulation: %s\n", TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m");
fprintf(stdout, "\tGPR memory model emulation: %s\n", GPRMemoryTSOEmulation);
fprintf(stdout, "\tMemcpy memory model emulation: %s\n", MemcpyMemoryTSOEmulation);
fprintf(stdout, "\tVector memory model emulation: %s\n", VectorMemoryTSOEmulation);
fprintf(stdout, "\nConfiguration:\n");
fprintf(stdout, "\tTSO Emulation: %s\n", TSOEnabled() ? "Enabled" : "Disabled");
fprintf(stdout, "\tMemcpy TSO Emulation: %s\n", TSOEnabled() && MemcpySetTSOEnabled() ? "Enabled" : "Disabled");
fprintf(stdout, "\tVector TSO Emulation: %s\n", TSOEnabled() && VectorTSOEnabled() ? "Enabled" : "Disabled");
fprintf(stdout, "\tHalf-barrier unaligned TSO emulation: %s\n", TSOEnabled() && HalfBarrierTSOEnabled() ? "Enabled" : "Disabled");
fprintf(stdout, "\t16-Byte strict split-lock emulation: %s\n", SplitLock16BConfigurationType);
fprintf(stdout, "\t64-Byte strict split-lock emulation: %s\n", StrictInProcessSplitLocks() ? "In-process mutex" : "Tearing");
if (IsAllEmuInfo || Options.is_set_by_user("tso_emulation_info")) {
PrintTSOInfo();
}
if (Options.is_set_by_user("identification_reg_info")) {
auto Features = FEX::GetCPUFeaturesFromIDRegisters();
fextl::string features {};
features += fmt::format("isar0=0x{:x},", Features.ISAR0.Get());
features += fmt::format("isar1=0x{:x},", Features.ISAR1.Get());
features += fmt::format("isar2=0x{:x},", Features.ISAR2.Get());
features += fmt::format("pfr0=0x{:x},", Features.PFR0.Get());
features += fmt::format("pfr1=0x{:x},", Features.PFR1.Get());
features += fmt::format("midr=0x{:x},", Features.MIDR.Get());
features += fmt::format("mmfr0=0x{:x},", Features.MMFR0.Get());
features += fmt::format("mmfr1=0x{:x},", Features.MMFR1.Get());
features += fmt::format("mmfr2=0x{:x},", Features.MMFR2.Get());
features += fmt::format("zfr0=0x{:x},", Features.ZFR0.Get());
features += fmt::format("dczid=0x{:x},", Features.DCZID.Get());
features += fmt::format("svevl=0x{:x}", Features.SVEVL.Get());
fprintf(stderr, "Features: '%s'\n", features.c_str());
if (IsAllEmuInfo || Options.is_set_by_user("identification_reg_info")) {
PrintIDRegInfo();
}
if (IsAllEmuInfo || Options.is_set_by_user("test_fault_granularity")) {
SIGBUSTest::PrintSIGBUSInfo();
}
#endif
@@ -31,12 +31,6 @@ install(TARGETS FEX RUNTIME
DESTINATION bin
COMPONENT Runtime)
# Create a copy of FEX with legacy names until phased out.
install(PROGRAMS ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/FEX
RENAME FEXInterpreter
DESTINATION bin
COMPONENT LegacyRuntime)
if (ARCHITECTURE_arm64)
if (NOT USE_LEGACY_BINFMTMISC)
# Just restart the systemd service
+8 -8
View File
@@ -297,8 +297,8 @@ public:
#ifndef AT_FLAGS_PRESERVE_ARGV0
#define AT_FLAGS_PRESERVE_ARGV0 1
#endif
uint32_t HostKernel = FEX::HLE::SyscallHandler::CalculateHostKernelVersion();
if ((HostKernel >= FEX::HLE::SyscallHandler::KernelVersion(5, 12, 0) && (AtFlags & AT_FLAGS_PRESERVE_ARGV0)) || LoadedWithFD) {
uint32_t HostKernel = FEX::LinuxVersion::CalculateHostKernelVersion();
if ((HostKernel >= FEX::LinuxVersion::KernelVersion(5, 12, 0) && (AtFlags & AT_FLAGS_PRESERVE_ARGV0)) || LoadedWithFD) {
// Erase the initial argument from the list in this case
ApplicationArgs.erase(ApplicationArgs.begin());
@@ -454,16 +454,16 @@ public:
// On the upside, this more accurately emulates how the kernel allocates stack space for the application when hinting at the location.
//
void* StackPointerBase {};
auto VASize = FEXCore::Allocator::DetermineVASize();
auto VABits = FEXCore::Allocator::GetHostVABits();
uint64_t StackHint {};
if (Is64BitMode()) {
if (VASize > 47) {
if (VABits > 47) {
// If VA size is at least as large as minimum x86 specification, then set to max.
VASize = 47;
VABits = 47;
}
// Calculate the highest point the stack could go.
StackHint = (1ULL << VASize) - FULL_STACK_SIZE;
StackHint = (1ULL << VABits) - FULL_STACK_SIZE;
} else {
// Needs to be under the 4GB VA space.
StackHint = 0x1'0000'0000ULL - FULL_STACK_SIZE;
@@ -557,8 +557,8 @@ public:
if (Is64BitMode()) {
// Ensure that if we are running on a 36-bit VA system, we don't try hinting that an ELF should
// live way outside the VA space.
uint64_t HostVASize = 1ULL << FEXCore::Allocator::DetermineVASize();
ELFLoadHint = std::min(HostVASize, TASK_SIZE_64) / 3 * 2;
uint64_t HostVABits = 1ULL << FEXCore::Allocator::GetHostVABits();
ELFLoadHint = std::min(HostVABits, TASK_SIZE_64) / 3 * 2;
} else {
ELFLoadHint = TASK_SIZE_32 / 3 * 2;
}
@@ -10,6 +10,7 @@ $end_info$
#include "Common/FEXServerClient.h"
#include "Common/Config.h"
#include "Common/HostFeatures.h"
#include "Common/Linux/LinuxVersion.h"
#include "Common/Linux/SBRKAllocations.h"
#include "PortabilityInfo.h"
#include "ELFCodeLoader.h"
@@ -463,8 +464,8 @@ int main(int argc, char** argv, char** const envp) {
return -ENOEXEC;
}
uint32_t KernelVersion = FEX::HLE::SyscallHandler::CalculateHostKernelVersion();
if (KernelVersion < FEX::HLE::SyscallHandler::KernelVersion(5, 15)) {
uint32_t KernelVersion = FEX::LinuxVersion::CalculateHostKernelVersion();
if (KernelVersion < FEX::LinuxVersion::KernelVersion(5, 15)) {
LogMan::Msg::EFmt("FEX requires kernel 5.15 minimum. Expect problems.");
}
@@ -524,16 +525,18 @@ int main(int argc, char** argv, char** const envp) {
FEXCore::Profiler::Init(Program.ProgramName, Program.ProgramPath);
bool SupportsAVX {};
bool SupportsSVE256 {};
fextl::unique_ptr<FEXCore::Context::Context> CTX;
{
auto HostFeatures = FEX::FetchHostFeatures();
CTX = FEXCore::Context::Context::CreateNewContext(HostFeatures);
SupportsAVX = HostFeatures.SupportsAVX;
SupportsSVE256 = HostFeatures.SupportsSVE256;
}
FEX::Kernel::Init(Loader.Is64BitMode(), CTX.get());
auto SignalDelegation = FEX::HLE::CreateSignalDelegator(CTX.get(), Program.ProgramName, SupportsAVX);
auto SignalDelegation = FEX::HLE::CreateSignalDelegator(CTX.get(), Program.ProgramName, SupportsAVX, SupportsSVE256);
auto ThunkHandler = FEX::HLE::CreateThunkHandler();
auto SyscallHandler = Loader.Is64BitMode() ?
@@ -9,6 +9,11 @@ target_link_libraries(FEXOfflineCompiler PRIVATE
fmt::fmt)
if (MINGW)
if (ARCHITECTURE_arm64ec)
set_target_properties(FEXOfflineCompiler PROPERTIES OUTPUT_NAME "FEXOfflineCompiler64")
else()
set_target_properties(FEXOfflineCompiler PROPERTIES OUTPUT_NAME "FEXOfflineCompiler32")
endif()
patch_library_wine(FEXOfflineCompiler)
target_include_directories(FEXOfflineCompiler PRIVATE
"${CMAKE_SOURCE_DIR}/Source/Windows/include/"
+41 -16
View File
@@ -51,6 +51,14 @@ static std::unique_ptr<FEX::Windows::OvercommitTracker> OvercommitTracker;
#endif
static FEXCore::Core::InternalThreadState* Thread = nullptr;
#ifdef _WIN32
#ifdef _M_ARM64EC
const bool Is64BitCompiler = true;
#else
const bool Is64BitCompiler = false;
#endif
#endif
#ifdef _WIN32
class AOTSyscallHandler : public FEXCore::HLE::SyscallHandler {
#else
@@ -396,10 +404,8 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
return std::nullopt;
}
const bool Is64Bit = Loader.Is64BitMode();
#elif defined(_M_ARM64EC)
const bool Is64Bit = true;
#else
const bool Is64Bit = false;
const bool Is64Bit = Is64BitCompiler;
#endif
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS64BIT_MODE, Is64Bit ? "1" : "0");
@@ -407,7 +413,7 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
#ifndef _WIN32
auto HostFeatures = FEX::FetchHostFeatures();
#else
const auto NtDll = GetModuleHandle("ntdll.dll");
const auto NtDll = GetModuleHandleW(L"ntdll.dll");
const bool IsWine = !!GetProcAddress(NtDll, "wine_get_version");
auto HostFeatures = FEX::Windows::CPUFeatures::FetchHostFeatures(
IsWine, Is64Bit ? FEXCore::HostFeatures::HostTypeEnum::Arm64ec : FEXCore::HostFeatures::HostTypeEnum::Wow64);
@@ -419,7 +425,7 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
#ifdef _WIN32
OvercommitTracker = std::make_unique<FEX::Windows::OvercommitTracker>(IsWine);
auto SyscallOSABI = Is64Bit ? FEXCore::HLE::SyscallOSABI::OS_LINUX64 : FEXCore::HLE::SyscallOSABI::OS_LINUX32;
auto SyscallOSABI = FEXCore::HLE::SyscallOSABI::OS_GENERIC;
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(*CTX, SyscallOSABI);
SyscallHandler->VAFileStart =
@@ -572,7 +578,7 @@ static int GenerateCache(int argc, const char** argv) {
}
for (auto& [FileId, Contents] : Parsed) {
if (!ExplicitFileId && (Contents.IsExecutable || Parsed.size() == 1)) {
if (!ExplicitFileId && (Contents.ExecutableBitness || Parsed.size() == 1)) {
ProgramName.FileId = FileId;
ProgramName.Filename = Contents.Filename;
}
@@ -621,7 +627,7 @@ static int GenerateCache(int argc, const char** argv) {
* Writes aggregated code map data into a single code map file that is ready to be used for cache generation
*/
static void WriteNewCodeMap(const FEXCore::ExecutableFileInfo& File, const std::string& OutputName, const fextl::set<uintptr_t>& Blocks,
bool IsExecutable, const std::set<FEXCore::ExecutableFileInfo>& Dependencies) {
std::optional<int> ExecutableBitness, const std::set<FEXCore::ExecutableFileInfo>& Dependencies) {
fmt::print("Writing {} blocks to {}\n", Blocks.size(), OutputName);
struct CodeMapOpener : FEXCore::CodeMapOpener {
@@ -638,9 +644,9 @@ static void WriteNewCodeMap(const FEXCore::ExecutableFileInfo& File, const std::
CodeMapOpener CodeMapOpener(OutputName);
FEXCore::CodeMapWriter OutputCodeMap(CodeMapOpener, true);
if (IsExecutable) {
if (ExecutableBitness) {
// List the main executable and all used libraries
OutputCodeMap.AppendSetMainExecutable(File);
OutputCodeMap.AppendSetMainExecutable(File, ExecutableBitness == 64);
for (auto& Dependency : Dependencies) {
OutputCodeMap.AppendLibraryLoad(Dependency);
@@ -658,7 +664,7 @@ static void WriteNewCodeMap(const FEXCore::ExecutableFileInfo& File, const std::
struct ParsedContentsAndDependencies {
fextl::string Filename;
fextl::set<uint64_t> Blocks;
bool IsExecutable = false;
std::optional<int> ExecutableBitness;
std::set<FEXCore::CodeMapFileId> Dependencies;
};
@@ -688,13 +694,13 @@ static std::map<FEXCore::CodeMapFileId, ParsedContentsAndDependencies> ImportPen
std::set<FEXCore::CodeMapFileId> Dependencies;
std::optional<FEXCore::CodeMapFileId> ExecutableFileId;
for (auto& [FileId, Contents] : FEXCore::CodeMap::ParseCodeMap(Incoming)) {
auto& [Filename, Blocks, IsExecutable, _] =
auto& [Filename, Blocks, ExecutableBitness, _] =
Result.emplace(std::piecewise_construct, std::forward_as_tuple(FileId), std::tuple {}).first->second;
Filename = std::move(Contents.Filename);
Blocks.merge(std::move(Contents.Blocks));
IsExecutable = Contents.IsExecutable;
if (IsExecutable) {
LOGMAN_THROW_A_FMT(!ExecutableFileId, "Expected a unique executable identifiers per code map");
ExecutableBitness = Contents.ExecutableBitness;
if (ExecutableBitness) {
LOGMAN_THROW_A_FMT(!ExecutableFileId, "Expected a unique executable identifier per code map");
ExecutableFileId = FileId;
} else {
Dependencies.insert(FileId);
@@ -745,7 +751,7 @@ static void AggregateCodeMaps(const std::string& NewCodeMapDirectory, const std:
for (auto& Dependency : Contents.Dependencies) {
Dependencies.emplace(nullptr, Dependency, IncomingCodeMap.at(Dependency).Filename);
}
WriteNewCodeMap(File, OutputName, Contents.Blocks, Contents.IsExecutable, Dependencies);
WriteNewCodeMap(File, OutputName, Contents.Blocks, Contents.ExecutableBitness, Dependencies);
}
}
@@ -767,12 +773,22 @@ static int ProcessAll() {
for (auto& Entry : std::filesystem::directory_iterator(ReadyCodeMapDirectory)) {
std::ifstream CodeMap(Entry.path(), std::ios_base::binary);
auto Parsed = FEXCore::CodeMap::ParseCodeMap(CodeMap);
auto ExecutableIt = std::ranges::find_if(Parsed, [](const auto& Entry) { return Entry.second.IsExecutable; });
auto ExecutableIt = std::ranges::find_if(Parsed, [](const auto& Entry) { return Entry.second.ExecutableBitness.has_value(); });
if (ExecutableIt == Parsed.end()) {
// Skip libraries; they're only processed as dependencies of a main executable
continue;
}
#ifdef _WIN32
// For WoA, spawn a subprocess for each cache generation run.
// This ensures robustness and allows for switching FEXOfflineCompiler between WoW64 and ARM64EC.
char SelfPathRaw[PATH_MAX];
GetModuleFileNameA(nullptr, SelfPathRaw, sizeof(SelfPathRaw));
std::string SelfPath = SelfPathRaw;
auto NewExecName = fmt::format("FEXOfflineCompiler{}.exe", ExecutableIt->second.ExecutableBitness.value());
SelfPath.replace(SelfPath.size() - NewExecName.size(), NewExecName.size(), NewExecName);
#endif
fmt::println("\nChecking caches for executable {}", ExecutableIt->second.Filename);
// TODO: Compute the cache config id from the active FEX configuration
@@ -808,9 +824,18 @@ static int ProcessAll() {
std::vector<const char*> GenerateArgs {
"generate", "--fileid", FileIdArg.c_str(), "--outdir", OutDir.c_str(), MergedCodeMapFilename.c_str(),
};
#ifndef _WIN32
if (GenerateCache(GenerateArgs.size(), GenerateArgs.data()) != 0) {
fmt::println("ERROR: Cache generation failed for {}", BinaryName);
}
#else
GenerateArgs.insert(GenerateArgs.begin(), SelfPath.c_str());
GenerateArgs.push_back(nullptr);
auto Status = _spawnv(_P_WAIT, SelfPath.c_str(), GenerateArgs.data());
if (Status) {
fmt::println("ERROR: Cache generation failed for {}", BinaryName);
}
#endif
}
}
+2 -2
View File
@@ -9,11 +9,11 @@
#include <fmt/format.h>
namespace FEXServer::Config {
static fextl::string Version = "FEX-Emu (" GIT_DESCRIBE_STRING ") ";
constexpr std::string_view Version = "FEX-Emu (" GIT_DESCRIBE_STRING ") ";
FEXServerOptions Load(int argc, char** argv) {
FEXServerOptions FEXOptions {};
optparse::OptionParser Parser = optparse::OptionParser().version(Version);
optparse::OptionParser Parser = optparse::OptionParser().version(fextl::string(Version));
Parser.add_option("-k", "--kill").action("store_true").set_default(false).help("Shutdown an already active FEXServer");
+6 -4
View File
@@ -1,4 +1,6 @@
// SPDX-License-Identifier: MIT
#include "Logger.h"
#include <Common/Async.h>
#include <Common/FEXServerClient.h>
@@ -10,10 +12,10 @@ void ClientMsgHandler(int FD, FEXServerClient::Logging::PacketMsg* const Msg, co
}
namespace Logger {
int LogClientQueuePipe[2];
std::thread LogThread;
static int LogClientQueuePipe[2];
static std::thread LogThread;
void HandleLogData(int Socket) {
static void HandleLogData(int Socket) {
std::vector<uint8_t> Data(1500);
size_t CurrentRead {};
while (true) {
@@ -54,7 +56,7 @@ void HandleLogData(int Socket) {
}
}
void LogThreadFunc() {
static void LogThreadFunc() {
fasio::poll_reactor Reactor;
auto Pipe = fasio::posix_descriptor {Reactor, LogClientQueuePipe[0]};
+5 -5
View File
@@ -37,12 +37,12 @@ static timespec StartTime {};
static std::optional<fmt::text_style> DisableColors = isatty(STDOUT_FILENO) ? std::nullopt : std::optional {fmt::text_style {}};
namespace Logging {
void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
static void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
const auto Output = fmt::format("{} {}\n", fmt::styled(LogMan::DebugLevelStr(Level), DisableColors.value_or(DebugLevelStyle(Level))), Message);
write(STDOUT_FILENO, Output.c_str(), Output.size());
}
void AssertHandler(const char* Message) {
static void AssertHandler(const char* Message) {
return MsgHandler(LogMan::ASSERT, Message);
}
@@ -77,9 +77,9 @@ void ActionHandler(int sig, siginfo_t* info, void* context) {
_exit(1);
}
void ActionIgnore(int sig, siginfo_t* info, void* context) {}
static void ActionIgnore(int sig, siginfo_t* info, void* context) {}
void SetupSignals() {
static void SetupSignals() {
// Setup our signal handlers now so we can capture some events
struct sigaction act {};
act.sa_sigaction = ActionHandler;
@@ -104,7 +104,7 @@ void SetupSignals() {
/**
* @brief Deparents itself by forking and terminating the parent process.
*/
void DeparentSelf() {
static void DeparentSelf() {
auto SystemdEnv = getenv("INVOCATION_ID");
if (SystemdEnv) {
// If FEXServer was launched through systemd then don't deparent, otherwise systemd kills the entire server.
+5 -2
View File
@@ -1,13 +1,16 @@
// SPDX-License-Identifier: MIT
#include "PipeScanner.h"
#include <dirent.h>
#include <fcntl.h>
#include <sys/types.h>
#include <sys/stat.h>
#include <unistd.h>
#include <vector>
#include <fcntl.h>
namespace PipeScanner {
std::vector<int> IncomingPipes {};
static std::vector<int> IncomingPipes {};
void SetWaitPipe(int FD) {
int flags = fcntl(FD, F_GETFD);
flags |= FD_CLOEXEC;
+36 -28
View File
@@ -1,6 +1,7 @@
// SPDX-License-Identifier: MIT
#include "FEXHeaderUtils/Syscalls.h"
#include "Logger.h"
#include "ProcessPipe.h"
#include "SquashFS.h"
#include <Common/AsyncNet.h>
@@ -31,7 +32,7 @@
#include <xxhash.h>
namespace FEXCore {
inline bool operator<(const FEXCore::ExecutableFileInfo& a, const FEXCore::ExecutableFileInfo& b) noexcept {
static bool operator<(const FEXCore::ExecutableFileInfo& a, const FEXCore::ExecutableFileInfo& b) noexcept {
return a.FileId < b.FileId;
}
} // namespace FEXCore
@@ -45,19 +46,19 @@ struct std::hash<FEXCore::ExecutableFileInfo> {
namespace ProcessPipe {
constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
int ServerLockFD {-1};
int WatchFD {-1};
std::optional<fasio::tcp_acceptor> ServerAcceptor;
std::optional<fasio::tcp_acceptor> ServerFSAcceptor;
int NumClients = 0;
time_t RequestTimeout {10};
bool Foreground {false};
std::vector<struct pollfd> PollFDs {};
static int ServerLockFD {-1};
static int WatchFD {-1};
static std::optional<fasio::tcp_acceptor> ServerAcceptor;
static std::optional<fasio::tcp_acceptor> ServerFSAcceptor;
static int NumClients = 0;
static time_t RequestTimeout {10};
static bool Foreground {false};
static std::vector<struct pollfd> PollFDs {};
// FD count watching
constexpr size_t static MAX_FD_DISTANCE = 32;
rlimit MaxFDs {};
std::atomic<size_t> NumFilesOpened {};
constexpr size_t MAX_FD_DISTANCE = 32;
static rlimit MaxFDs {};
static std::atomic<size_t> NumFilesOpened {};
// Path to directory for unprocessed code maps dumped by FEX
static std::string NewCodeMapDirectory;
@@ -72,14 +73,14 @@ void SetWatchFD(int FD) {
WatchFD = FD;
}
size_t GetNumFilesOpen() {
static size_t GetNumFilesOpen() {
// Walk /proc/self/fd/ to see how many open files we currently have
const std::filesystem::path self {"/proc/self/fd/"};
return std::distance(std::filesystem::directory_iterator {self}, std::filesystem::directory_iterator {});
}
void GetMaxFDs() {
static void GetMaxFDs() {
// Get our kernel limit for the number of open files
if (getrlimit(RLIMIT_NOFILE, &MaxFDs) != 0) {
fprintf(stderr, "[FEXMountDaemon] getrlimit(RLIMIT_NOFILE) returned error %d %s\n", errno, strerror(errno));
@@ -89,7 +90,7 @@ void GetMaxFDs() {
NumFilesOpened = GetNumFilesOpen();
}
void CheckRaiseFDLimit() {
static void CheckRaiseFDLimit() {
if (NumFilesOpened < (MaxFDs.rlim_cur - MAX_FD_DISTANCE)) {
// No need to raise the limit.
return;
@@ -221,7 +222,7 @@ bool InitializeServerPipe() {
static fasio::poll_reactor Reactor;
void HandleSocketData(fasio::tcp_socket&);
static void HandleSocketData(fasio::tcp_socket&);
bool InitializeServerSocket(bool abstract) {
fextl::string ServerSocketName;
@@ -281,7 +282,7 @@ bool InitializeServerSocket(bool abstract) {
return true;
}
void SendEmptyErrorPacket(fasio::tcp_socket& Socket) {
static void SendEmptyErrorPacket(fasio::tcp_socket& Socket) {
FEXServerClient::FEXServerResultPacket Res {
.Header {
.Type = FEXServerClient::PacketType::TYPE_ERROR,
@@ -293,7 +294,7 @@ void SendEmptyErrorPacket(fasio::tcp_socket& Socket) {
write(Socket, Data, ec);
}
void SendFDSuccessPacket(fasio::tcp_socket& Socket, int FD) {
static void SendFDSuccessPacket(fasio::tcp_socket& Socket, int FD) {
FEXServerClient::FEXServerResultPacket Res {
.Header {
.Type = FEXServerClient::PacketType::TYPE_SUCCESS,
@@ -307,7 +308,7 @@ void SendFDSuccessPacket(fasio::tcp_socket& Socket, int FD) {
// Discovers any pending code maps, parses their contents into a runtime data structure, and deletes them
static std::map<FEXCore::ExecutableFileInfo, fextl::set<uintptr_t>>
ImportPendingCodeMaps(const FEXCore::ExecutableFileInfo& MainFileId, bool HasMultiblock) {
ImportPendingCodeMaps(const FEXCore::ExecutableFileInfo& MainFileId, bool HasMultiblock, std::optional<int>& ExecutableBitness) {
// Detect code maps by checking file name suffixes by counting up an index.
// Code maps that are ready for reading must be non-empty and flock(FLOCK_EX) must succeed:
// - If empty, we tried generating the cache before the client could even lock it
@@ -334,7 +335,7 @@ ImportPendingCodeMaps(const FEXCore::ExecutableFileInfo& MainFileId, bool HasMul
fmt::print("Found code map {}, queuing for merge\n", CodeMap);
}
close(FD);
CodeMaps.push_back(CodeMap);
CodeMaps.push_back(std::move(CodeMap));
}
// Update merged code map
@@ -348,6 +349,11 @@ ImportPendingCodeMaps(const FEXCore::ExecutableFileInfo& MainFileId, bool HasMul
for (auto& [FileId, Contents] : NewBlocks) {
ImportedCodeMaps.emplace(std::piecewise_construct, std::forward_as_tuple(nullptr, FileId, std::move(Contents.Filename)),
std::forward_as_tuple(std::move(Contents.Blocks)));
if (FileId == MainFileId.FileId) {
LOGMAN_THROW_A_FMT(Contents.ExecutableBitness && (!ExecutableBitness.has_value() || ExecutableBitness == Contents.ExecutableBitness),
"Inconsistent executable bitness");
ExecutableBitness = Contents.ExecutableBitness;
}
}
}
}
@@ -365,7 +371,7 @@ ImportPendingCodeMaps(const FEXCore::ExecutableFileInfo& MainFileId, bool HasMul
* Writes aggregated code map data into a single code map file that is ready to be used for cache generation
*/
static void WriteNewCodeMap(const FEXCore::ExecutableFileInfo& File, const std::string& OutputName, const fextl::set<uintptr_t>& Blocks,
bool IsMainFile, const auto& Dependencies) {
std::optional<int> ExecutableBitness, const auto& Dependencies) {
fmt::print("Writing {} blocks to {}\n", Blocks.size(), OutputName);
struct CodeMapOpener : FEXCore::CodeMapOpener {
@@ -382,9 +388,9 @@ static void WriteNewCodeMap(const FEXCore::ExecutableFileInfo& File, const std::
CodeMapOpener CodeMapOpener(OutputName);
FEXCore::CodeMapWriter OutputCodeMap(CodeMapOpener, true);
if (IsMainFile) {
if (ExecutableBitness) {
// List the main executable and all used libraries
OutputCodeMap.AppendSetMainExecutable(File);
OutputCodeMap.AppendSetMainExecutable(File, ExecutableBitness == 64);
for (auto& [Dependency, _] : Dependencies) {
OutputCodeMap.AppendLibraryLoad(Dependency);
@@ -426,11 +432,13 @@ static std::map<FEXCore::ExecutableFileInfo, NeedsCacheRefresh> AggregateCodeMap
}
// Accumulate information from new code maps
auto IncomingCodeMap = ImportPendingCodeMaps(MainFileId, HasMultiblock);
std::optional<int> ExecutableBitness;
auto IncomingCodeMap = ImportPendingCodeMaps(MainFileId, HasMultiblock, ExecutableBitness);
for (auto& [File, _] : IncomingCodeMap) {
Result.emplace(std::piecewise_construct, std::forward_as_tuple(nullptr, File.FileId, File.Filename),
std::forward_as_tuple(NeedsCacheRefresh::No));
}
LOGMAN_THROW_A_FMT(ExecutableBitness, "New code maps did not contain executable marker");
// For each referenced library, add referenced offsets to that library's reference code map
for (auto& [File, Blocks] : IncomingCodeMap) {
@@ -452,14 +460,14 @@ static std::map<FEXCore::ExecutableFileInfo, NeedsCacheRefresh> AggregateCodeMap
// Update code map and queue for cache generation
std::map<FEXCore::ExecutableFileInfo, NeedsCacheRefresh> Empty;
WriteNewCodeMap(File, OutputName, Blocks, true, File.FileId == MainFileId.FileId ? Result : Empty);
WriteNewCodeMap(File, OutputName, Blocks, ExecutableBitness, File.FileId == MainFileId.FileId ? Result : Empty);
Result.at(File) = NeedsCacheRefresh::Yes;
}
return Result;
}
int32_t EmbedSubprocess(const char* path, char* const* args) {
static int32_t EmbedSubprocess(const char* path, char* const* args) {
pid_t pid = fork();
if (pid == 0) {
execvp(path, args);
@@ -484,7 +492,7 @@ static int RunOfflineCompiler(const char* CodeMap) {
return EmbedSubprocess(OfflineCompilerPath.c_str(), const_cast<char* const*>(&ExecveArgs[0]));
};
void HandleSocketData(fasio::tcp_socket& Socket) {
static void HandleSocketData(fasio::tcp_socket& Socket) {
std::vector<uint8_t> Data(1500);
// Get the current number of FDs of the process before we start handling sockets.
@@ -719,7 +727,7 @@ void HandleSocketData(fasio::tcp_socket& Socket) {
}
}
void CloseConnections() {
static void CloseConnections() {
// Close the server pipe so new processes will know to spin up a new FEXServer.
// This one is closing
close(ServerLockFD);
+9 -7
View File
@@ -1,4 +1,6 @@
// SPDX-License-Identifier: MIT
#include "SquashFS.h"
#include "Common/FEXServerClient.h"
#include "Common/FileFormatCheck.h"
@@ -17,17 +19,17 @@
namespace SquashFS {
constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
int ServerRootFSLockFD {-1};
int FuseMountPID {};
fextl::string MountFolder {};
static int ServerRootFSLockFD {-1};
static int FuseMountPID {};
static fextl::string MountFolder {};
void ShutdownImagePID() {
static void ShutdownImagePID() {
if (FuseMountPID) {
FHU::Syscalls::tgkill(FuseMountPID, FuseMountPID, SIGINT);
}
}
bool InitializeSquashFSPipe() {
static bool InitializeSquashFSPipe() {
auto RootFSLockFile = FEXServerClient::GetServerRootFSLockFile();
int Ret = open(RootFSLockFile.c_str(), O_CREAT | O_RDWR | O_TRUNC | O_EXCL | O_CLOEXEC, USER_PERMS);
@@ -84,7 +86,7 @@ bool InitializeSquashFSPipe() {
return true;
}
bool DowngradeRootFSPipeToReadLock() {
static bool DowngradeRootFSPipeToReadLock() {
flock lk {
.l_type = F_RDLCK,
.l_whence = SEEK_SET,
@@ -104,7 +106,7 @@ bool DowngradeRootFSPipeToReadLock() {
return true;
}
bool MountRootFSImagePath(const fextl::string& SquashFS, bool EroFS) {
static bool MountRootFSImagePath(const fextl::string& SquashFS, bool EroFS) {
pid_t ParentTID = ::getpid();
MountFolder = fmt::format("{}/.FEXMount{}-XXXXXX", FEXServerClient::GetServerMountFolder(), ParentTID);
char* MountFolderStr = MountFolder.data();
@@ -67,7 +67,7 @@ static void SealTmpFD(int fd) {
}
}
fextl::string GenerateCPUInfo(FEXCore::Context::Context* ctx, uint32_t CPUCores) {
static fextl::string GenerateCPUInfo(FEXCore::Context::Context* ctx, uint32_t CPUCores) {
fextl::ostringstream cpu_stream {};
auto res_0 = ctx->RunCPUIDFunction(0, 0);
auto res_1 = ctx->RunCPUIDFunction(1, 0);
@@ -532,8 +532,8 @@ EmulatedFDManager::EmulatedFDManager(FEXCore::Context::Context* ctx)
int FD = GenTmpFD(pathname, flags);
uint32_t GuestVersion = FEX::HLE::_SyscallHandler->GetGuestKernelVersion();
char Tmp[64] {};
snprintf(Tmp, sizeof(Tmp), "%d.%d.%d\n", FEX::HLE::SyscallHandler::KernelMajor(GuestVersion),
FEX::HLE::SyscallHandler::KernelMinor(GuestVersion), FEX::HLE::SyscallHandler::KernelPatch(GuestVersion));
snprintf(Tmp, sizeof(Tmp), "%d.%d.%d\n", FEX::LinuxVersion::KernelMajor(GuestVersion), FEX::LinuxVersion::KernelMinor(GuestVersion),
FEX::LinuxVersion::KernelPatch(GuestVersion));
// + 1 to ensure null at the end
write(FD, Tmp, strlen(Tmp) + 1);
lseek(FD, 0, SEEK_SET);
@@ -548,8 +548,8 @@ EmulatedFDManager::EmulatedFDManager(FEXCore::Context::Context* ctx)
const char kernel_version[] = "Linux version %d.%d.%d (FEX@FEX) (clang) #" GIT_DESCRIBE_STRING " SMP " __DATE__ " " __TIME__ "\n";
uint32_t GuestVersion = FEX::HLE::_SyscallHandler->GetGuestKernelVersion();
char Tmp[sizeof(kernel_version) + 64] {};
snprintf(Tmp, sizeof(Tmp), kernel_version, FEX::HLE::SyscallHandler::KernelMajor(GuestVersion),
FEX::HLE::SyscallHandler::KernelMinor(GuestVersion), FEX::HLE::SyscallHandler::KernelPatch(GuestVersion));
snprintf(Tmp, sizeof(Tmp), kernel_version, FEX::LinuxVersion::KernelMajor(GuestVersion), FEX::LinuxVersion::KernelMinor(GuestVersion),
FEX::LinuxVersion::KernelPatch(GuestVersion));
// + 1 to ensure null at the end
write(FD, Tmp, strlen(Tmp) + 1);
lseek(FD, 0, SEEK_SET);
@@ -48,7 +48,7 @@ extern "C" uint64_t CopyToUser_FaultInst;
void* const CopyToUser_FaultLocation = &CopyToUser_FaultInst;
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED && defined(ARCHITECTURE_arm64)
__attribute__((naked)) bool VerifyIsReadableImpl(const void* Src, size_t Size) {
__attribute__((naked)) static bool VerifyIsReadableImpl(const void* Src, size_t Size) {
__asm volatile(R"(
// Early exit if size is zero.
cbz x1, 2f;
@@ -67,7 +67,7 @@ __attribute__((naked)) bool VerifyIsReadableImpl(const void* Src, size_t Size) {
: "memory");
}
__attribute__((naked)) bool VerifyIsOnlyWritable(void* Src, size_t Size) {
__attribute__((naked)) static bool VerifyIsOnlyWritable(void* Src, size_t Size) {
__asm volatile(R"(
// Early exit if size is zero.
cbz x1, 2f;
@@ -88,7 +88,7 @@ __attribute__((naked)) bool VerifyIsOnlyWritable(void* Src, size_t Size) {
: "memory");
}
__attribute__((naked)) bool VerifyIsStringReadableMaxSizeImpl(const char* Src, size_t MaxSize) {
__attribute__((naked)) static bool VerifyIsStringReadableMaxSizeImpl(const char* Src, size_t MaxSize) {
__asm volatile(R"(
1:
cbz x1, 2f;
@@ -1441,9 +1441,11 @@ void GdbServer::GdbServerLoop() {
Acceptor->async_accept([this](fasio::error ec, std::optional<fasio::tcp_socket> Socket) {
if (ec != fasio::error::success) {
// Listen socket error or shutting down
LogMan::Msg::EFmt("[GdbServer] gdbserver shutting down: {}");
close(CommsSocket->FD);
CommsSocket.reset();
if (CommsSocket) {
LogMan::Msg::EFmt("[GdbServer] gdbserver shutting down: {}", CommsSocket->FD);
close(CommsSocket->FD);
CommsSocket.reset();
}
// Repeat to wait for another connection
return fasio::post_callback::repeat;
}
@@ -23,8 +23,8 @@ namespace FEX::HLE {
class MemAllocator32Bit final : public FEX::HLE::MemAllocator {
private:
static constexpr uint64_t BASE_KEY = 16;
const uint64_t TOP_KEY = 0xFFFF'F000ULL >> FEXCore::Utils::FEX_PAGE_SHIFT;
const uint64_t TOP_KEY32BIT = 0x7FFF'F000ULL >> FEXCore::Utils::FEX_PAGE_SHIFT;
static constexpr uint64_t TOP_KEY = 0xFFFF'F000ULL >> FEXCore::Utils::FEX_PAGE_SHIFT;
static constexpr uint64_t TOP_KEY32BIT = 0x7FFF'F000ULL >> FEXCore::Utils::FEX_PAGE_SHIFT;
public:
MemAllocator32Bit() {
@@ -110,7 +110,7 @@ uint64_t BPFEmitter::HandleStore(uint32_t BPFIP, const sock_filter* Inst) {
[[maybe_unused]] size_t OpSize {};
const auto SrcReg = BPF_CLASS(Inst->code) == BPF_LD ? REG_A : REG_X;
const auto SrcReg = BPF_CLASS(Inst->code) == BPF_ST ? REG_A : REG_X;
// Must be smaller than scratch space size.
VALIDATE(Inst->k < 16);
@@ -64,18 +64,12 @@ static void SignalHandlerThunk(int Signal, siginfo_t* Info, void* UContext) {
ThreadObject->SignalInfo.Delegator->HandleSignal(ThreadObject, Signal, Info, UContext);
}
uint64_t SigIsMember(GuestSAMask* Set, int Signal) {
static uint64_t SigIsMember(GuestSAMask* Set, int Signal) {
// Signal 0 isn't real, so everything is offset by one inside the set
Signal -= 1;
return (Set->Val >> Signal) & 1;
}
uint64_t SetSignal(GuestSAMask* Set, int Signal) {
// Signal 0 isn't real, so everything is offset by one inside the set
Signal -= 1;
return Set->Val | (1ULL << Signal);
}
/**
* @name Signal frame setup
* @{ */
@@ -105,7 +99,7 @@ void SignalDelegator::HandleSignal(FEX::HLE::ThreadStateObject* Thread, int Sign
}
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
SetHostSignalHandler(Signal, std::move(Func), Required);
SetHostSignalHandler(Signal, std::move(Func));
FrontendRegisterHostSignalHandler(Signal, Required);
}
@@ -122,7 +116,7 @@ void SignalDelegator::SpillSRA(FEXCore::Core::InternalThreadState* Thread, void*
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRAIdxMap);
}
if (SupportsAVX) {
if (SupportsAVX && SupportsSVE256) {
// TODO: This doesn't save the upper 128-bits of the 256-bit registers.
// This needs to be implemented still.
for (size_t i = 0; i < Config.SRAFPRCount; i++) {
@@ -582,7 +576,8 @@ bool SignalDelegator::HandleFrontendSIGSEGV(FEXCore::Core::InternalThreadState*
}
#ifdef ARCHITECTURE_arm64
if (Signal == SIGSEGV && SigInfo.si_code == SEGV_ACCERR && SigInfo.si_addr >= reinterpret_cast<void*>(Thread->JITGuardPage) &&
if (Signal == SIGSEGV && SigInfo.si_code == SEGV_ACCERR && Thread->JITGuardPage &&
SigInfo.si_addr >= reinterpret_cast<void*>(Thread->JITGuardPage) &&
SigInfo.si_addr < reinterpret_cast<void*>(Thread->JITGuardPage + FEXCore::Utils::FEX_PAGE_SIZE)) {
FEXCore::UncheckedLongJump::ManuallyLoadJumpBuf(Thread->RestartJump, Thread->JITGuardOverflowArgument,
ArchHelpers::Context::GetArmGPRs(UContext), ArchHelpers::Context::GetArmFPRs(UContext),
@@ -877,10 +872,11 @@ void SignalDelegator::QueueSignal(pid_t tgid, pid_t tid, int Signal, siginfo_t*
}
}
SignalDelegator::SignalDelegator(FEXCore::Context::Context* _CTX, const std::string_view ApplicationName, bool SupportsAVX)
SignalDelegator::SignalDelegator(FEXCore::Context::Context* _CTX, const std::string_view ApplicationName, bool SupportsAVX, bool SupportsSVE256)
: CTX {_CTX}
, ApplicationName {ApplicationName}
, SupportsAVX {SupportsAVX} {
, SupportsAVX {SupportsAVX}
, SupportsSVE256 {SupportsSVE256} {
// Signal zero isn't real
HostHandlers[0].Installed = true;
@@ -1082,7 +1078,7 @@ void SignalDelegator::RegisterHostSignalHandlerForGuest(int Signal, FEX::HLE::Ho
}
void SignalDelegator::RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
SetFrontendHostSignalHandler(Signal, std::move(Func), Required);
SetFrontendHostSignalHandler(Signal, std::move(Func));
FrontendRegisterFrontendHostSignalHandler(Signal, Required);
}
@@ -1219,7 +1215,7 @@ uint64_t SignalDelegator::GuestSigProcMask(FEX::HLE::ThreadStateObject* Thread,
// 3) Give old mask back
auto OldSet = Thread->SignalInfo.CurrentSignalMask.Val;
if (!!set) {
if (set) {
uint64_t IgnoredSignalsMask = ~((1ULL << (SIGKILL - 1)) | (1ULL << (SIGSTOP - 1)));
if (how == SIG_BLOCK) {
Thread->SignalInfo.CurrentSignalMask.Val |= *set & IgnoredSignalsMask;
@@ -1244,7 +1240,7 @@ uint64_t SignalDelegator::GuestSigProcMask(FEX::HLE::ThreadStateObject* Thread,
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &HostMask, nullptr, 8);
}
if (!!oldset) {
if (oldset) {
*oldset = OldSet;
}
@@ -1317,11 +1313,7 @@ uint64_t SignalDelegator::GuestSigSuspend(FEX::HLE::ThreadStateObject* Thread, u
}
uint64_t SignalDelegator::GuestSigTimedWait(uint64_t* set, siginfo_t* info, const struct timespec* timeout, size_t sigsetsize) {
if (sigsetsize > sizeof(uint64_t)) {
return -EINVAL;
}
uint64_t Result = ::syscall(SYS_rt_sigtimedwait, set, info, timeout);
uint64_t Result = ::syscall(SYS_rt_sigtimedwait, set, info, timeout, sigsetsize);
return Result == -1 ? -errno : Result;
}
@@ -1354,7 +1346,7 @@ uint64_t SignalDelegator::GuestSignalFD(int fd, const uint64_t* set, size_t sigs
}
fextl::unique_ptr<FEX::HLE::SignalDelegator>
CreateSignalDelegator(FEXCore::Context::Context* CTX, const std::string_view ApplicationName, bool SupportsAVX) {
return fextl::make_unique<FEX::HLE::SignalDelegator>(CTX, ApplicationName, SupportsAVX);
CreateSignalDelegator(FEXCore::Context::Context* CTX, const std::string_view ApplicationName, bool SupportsAVX, bool SupportsSVE256) {
return fextl::make_unique<FEX::HLE::SignalDelegator>(CTX, ApplicationName, SupportsAVX, SupportsSVE256);
}
} // namespace FEX::HLE
@@ -54,7 +54,7 @@ public:
// Returns true if the host handled the signal
// Arguments are the same as sigaction handler
SignalDelegator(FEXCore::Context::Context* _CTX, const std::string_view ApplicationName, bool SupportsAVX);
SignalDelegator(FEXCore::Context::Context* _CTX, const std::string_view ApplicationName, bool SupportsAVX, bool SupportsSVE256);
~SignalDelegator() override;
// Called from the signal trampoline function.
@@ -106,8 +106,6 @@ public:
void UninstallHostHandler(int Signal);
void QueueSignal(pid_t tgid, pid_t tid, int Signal, siginfo_t* info, bool IgnoreMask);
FEXCore::Context::Context* CTX;
void SetVDSOSymbols() {
// Get symbols from VDSO.
VDSOPointers = FEX::VDSO::GetVDSOSymbols();
@@ -149,31 +147,6 @@ public:
void SpillSRA(FEXCore::Core::InternalThreadState* Thread, void* ucontext, uint32_t IgnoreMask);
private:
// Called from the thunk handler to handle the signal
void HandleGuestSignal(FEX::HLE::ThreadStateObject* ThreadObject, int Signal, void* Info, void* UContext);
bool HandleFrontendSIGSEGV(FEXCore::Core::InternalThreadState* Thread, int Signal, void* Info, void* UContext);
/**
* @brief Registers a signal handler for the host to handle a signal
*
* It's a process level signal handler so one must be careful
*/
void FrontendRegisterHostSignalHandler(int Signal, bool Required);
void FrontendRegisterFrontendHostSignalHandler(int Signal, bool Required);
void SetHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
HostHandlers[Signal].Handlers.push_back(std::move(Func));
}
void SetFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
HostHandlers[Signal].FrontendHandler = std::move(Func);
}
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
const fextl::string ApplicationName;
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
FEXCore::ArchHelpers::Arm64::UnalignedHandlerType UnalignedHandlerType {FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::HalfBarrier};
enum DefaultBehaviourType {
DEFAULT_TERM,
// Core dump based signals are supposed to have a coredump appear
@@ -208,12 +181,45 @@ private:
HostSignalDelegatorFunction FrontendHandler {};
};
FEXCore::Context::Context* CTX;
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
const fextl::string ApplicationName;
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
FEXCore::ArchHelpers::Arm64::UnalignedHandlerType UnalignedHandlerType {FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::HalfBarrier};
std::array<SignalHandler, MAX_SIGNALS + 1> HostHandlers {};
bool InstallHostThunk(int Signal);
bool UpdateHostThunk(int Signal);
FEX::VDSO::VDSOEntrypoints VDSOPointers {};
std::mutex HostDelegatorMutex;
std::mutex GuestDelegatorMutex;
bool SupportsAVX;
bool SupportsSVE256;
// Called from the thunk handler to handle the signal
void HandleGuestSignal(FEX::HLE::ThreadStateObject* ThreadObject, int Signal, void* Info, void* UContext);
bool HandleFrontendSIGSEGV(FEXCore::Core::InternalThreadState* Thread, int Signal, void* Info, void* UContext);
/**
* @brief Registers a signal handler for the host to handle a signal
*
* It's a process level signal handler so one must be careful
*/
void FrontendRegisterHostSignalHandler(int Signal, bool Required);
void FrontendRegisterFrontendHostSignalHandler(int Signal, bool Required);
void SetHostSignalHandler(int Signal, HostSignalDelegatorFunction Func) {
HostHandlers[Signal].Handlers.push_back(std::move(Func));
}
void SetFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func) {
HostHandlers[Signal].FrontendHandler = std::move(Func);
}
bool InstallHostThunk(int Signal);
bool UpdateHostThunk(int Signal);
bool IsAddressInDispatcher(uint64_t Address) const {
return Address >= Config.DispatcherBegin && Address < Config.DispatcherEnd;
}
@@ -286,12 +292,8 @@ private:
bool HandleSIGILL(FEXCore::Core::InternalThreadState* Thread, int Signal, void* info, void* ucontext);
uint64_t GetNewSigMask(int Signal) const;
std::mutex HostDelegatorMutex;
std::mutex GuestDelegatorMutex;
bool SupportsAVX;
};
fextl::unique_ptr<FEX::HLE::SignalDelegator>
CreateSignalDelegator(FEXCore::Context::Context* CTX, const std::string_view ApplicationName, bool SupportsAVX);
CreateSignalDelegator(FEXCore::Context::Context* CTX, const std::string_view ApplicationName, bool SupportsAVX, bool SupportsSVE256);
} // namespace FEX::HLE
@@ -790,7 +790,7 @@ SyscallHandler::SyscallHandler(FEXCore::Context::Context* _CTX, FEX::HLE::Signal
, SignalDelegation {_SignalDelegation}
, ThunkHandler {ThunkHandler} {
FEX::HLE::_SyscallHandler = this;
HostKernelVersion = CalculateHostKernelVersion();
HostKernelVersion = LinuxVersion::CalculateHostKernelVersion();
GuestKernelVersion = CalculateGuestKernelVersion();
Alloc32Handler = FEX::HLE::Create32BitAllocator();
@@ -803,28 +803,9 @@ SyscallHandler::~SyscallHandler() {
FEXCore::Allocator::munmap(reinterpret_cast<void*>(DataSpace), DataSpaceMappedSize);
}
uint32_t SyscallHandler::CalculateHostKernelVersion() {
struct utsname buf {};
if (uname(&buf) == -1) {
return 0;
}
uint32_t Major {};
uint32_t Minor {};
uint32_t Patch {};
// Parse kernel version in the form of `<Major>.<Minor>.<Patch>[Optional Data]`
const auto End = buf.release + sizeof(buf.release);
auto Results = std::from_chars(buf.release, End, Major, 10);
Results = std::from_chars(Results.ptr + 1, End, Minor, 10);
Results = std::from_chars(Results.ptr + 1, End, Patch, 10);
return (Major << 24) | (Minor << 16) | Patch;
}
uint32_t SyscallHandler::CalculateGuestKernelVersion() {
// We currently only emulate a kernel between the ranges of Kernel 5.15.0 and 6.11.0
return std::max(KernelVersion(5, 15), std::min(KernelVersion(6, 11), GetHostKernelVersion()));
return std::max(LinuxVersion::KernelVersion(5, 15), std::min(LinuxVersion::KernelVersion(6, 11), GetHostKernelVersion()));
}
uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) {
@@ -8,6 +8,7 @@ $end_info$
#pragma once
#include "Common/Linux/LinuxVersion.h"
#include "Common/VolatileMetadata.h"
#include "LinuxSyscalls/FileManagement.h"
#include "LinuxSyscalls/LinuxAllocator.h"
@@ -210,26 +211,11 @@ public:
}
bool IsHostKernelVersionAtLeast(uint32_t Major, uint32_t Minor = 0, uint32_t Patch = 0) const {
return GetHostKernelVersion() >= KernelVersion(Major, Minor, Patch);
return GetHostKernelVersion() >= LinuxVersion::KernelVersion(Major, Minor, Patch);
}
static uint32_t CalculateHostKernelVersion();
uint32_t CalculateGuestKernelVersion();
static uint32_t KernelVersion(uint32_t Major, uint32_t Minor = 0, uint32_t Patch = 0) {
return (Major << 24) | (Minor << 16) | Patch;
}
static uint32_t KernelMajor(uint32_t Version) {
return Version >> 24;
}
static uint32_t KernelMinor(uint32_t Version) {
return (Version >> 16) & 0xFF;
}
static uint32_t KernelPatch(uint32_t Version) {
return Version & 0xFFFF;
}
virtual FEX::HLE::MemAllocator* Get32BitAllocator() {
return Alloc32Handler.get();
}
Loaded 100 of 200 files, more files were not shown because too many files have changed in this diff. Show more