Compare commits

...
124 Commits
Author SHA1 Message Date
Ryan Houdek a408749eef Docs: Update for release FEX-2203 2022-03-06 04:49:31 -08:00
Ryan Houdek d8a3687ac3 Merge pull request #1604 from Sonicadvance1/fix_cmpxchg_66h
OpcodeDispatcher: Fixes CMPXCHG8B/16B with 66h/72h/73h prefix
2022-03-06 04:10:58 -08:00
Ryan Houdek a32c7f6ce2 unittests: Adds cmpxchg unit tests for prefixes 2022-03-06 03:54:45 -08:00
Ryan Houdek 540feb857b OpcodeDispatcher: Fixes CMPXCHG8B/16B with 66h/72h/73h prefix
The documentation is incorrect about this instruction. It claims that
you use 66h prefix to choose between operating at 8B or 16B.
This is incorrect, real hardware only responds to REX.W for choosing the
operating size. These other prefixes are ignored but is still accepted as
an instruction decoding.
2022-03-06 03:54:45 -08:00
Ryan Houdek a3902a0d2d FEXCore: Fixes usage of GPRPair in operations
These were working around the previous quirks by accident
2022-03-06 03:54:45 -08:00
Ryan Houdek 5fbd01536f IR: Fixes some GPRPair IR op definitions
These were always wrong but how it the operations were handled meant
that it happened to work even though the IR representation was broken
2022-03-06 03:54:45 -08:00
Ryan Houdek bebcab0277 Merge pull request #1603 from Sonicadvance1/rng_support
FEXCore: Adds support for RDRAND/RDSEED
2022-03-06 03:54:20 -08:00
Ryan Houdek d0f17d400e unittests: Adds RDRAND/RDSEED unit tests 2022-03-06 03:40:25 -08:00
Ryan Houdek 11a07eb3f2 FEXCore: Adds support for RDRAND/RDSEED
This matches the AArch64 implementation fairly well.
Bundles RDRAND and RDSEED together for simplification, both instructions
are a single flag on AArch64.
2022-03-06 03:40:25 -08:00
Ryan Houdek cb13e1bdb9 X86Tables: Fixes secondary group decoding
If we're hitting these group tables then it needs the ignore overlay extension
since the prefixes are used to select ops inside this table.
2022-03-06 03:40:25 -08:00
Ryan Houdek 3e8c6d0be0 Merge pull request #1601 from Sonicadvance1/new_ir_json
IR: New IR JSON format
2022-03-06 03:40:05 -08:00
Ryan Houdek 847b6e5026 LinuxSyscalls: Fixes struct verifier on Ubuntu 20.04
The `linux/types.h` header needs to be included before the rest
otherwise we are missing types.
2022-03-04 17:44:26 -08:00
Ryan Houdek 7f6e9d3dae unittests/IR: Resolves fallout from recent JSON changes 2022-03-04 16:39:11 -08:00
Ryan Houdek e244e8f66b x86Jit: Resolves fallout from recent JSON changes 2022-03-04 16:39:11 -08:00
Ryan Houdek 0adaa85d97 ArmJit: Resolves fallout from recent JSON changes 2022-03-04 16:39:11 -08:00
Ryan Houdek f1d34c7407 Interpreter: Resolves fallout from recent JSON changes 2022-03-04 16:39:11 -08:00
Ryan Houdek 7a9492ceca IRPasses: Resolves fallout from recent JSON changes 2022-03-04 16:39:11 -08:00
Ryan Houdek 39192cd092 OpcodeDispatcher: Resolves fallout from recent JSON changes
Also fixes a bug in FCVTIntTo where it was getting passed an FPR when it
wants a GPR. Would cause RA problems on the JITs
2022-03-04 05:10:45 -08:00
Ryan Houdek 73abf9b9dd IRParser: Resolves fallout from recent JSON changes 2022-03-04 04:07:25 -08:00
Ryan Houdek 0387c24e21 IREmitter: Resolves the fallout from recent JSON changes 2022-03-04 04:07:25 -08:00
Ryan Houdek 9d14bc7846 IR: Updates generators and json to new format
This greatly simplifies the IR format by using string parsing for
gathering the information.
Tons of redundant information removed.
Significantly more difficult to mess up adding a new IR op.

Significantly improves the generator functions in IREmitter
2022-03-04 03:51:44 -08:00
Ryan Houdek ac32ecadbe Merge pull request #1597 from Sonicadvance1/3dnow_and_back_again
OpcodeDispatcher: Implements all the 3DNow! instructions
2022-03-04 03:36:40 -08:00
Mai M 08938ecb11 Merge pull request #1596 from Sonicadvance1/fix_old_kernel_bug
LinuxAllocator: Fixes bug with old kernels and hint allocation
2022-03-02 00:26:51 -05:00
Ryan Houdek 4a3cbf1b1f Merge pull request #1598 from Sonicadvance1/hypervisor_cpuid
CPUID: Implements leaf 4000_0000
2022-03-01 21:02:32 -08:00
Ryan Houdek 0a0bc21c88 CPUID: Implements leaf 4000_0000
This region is reserved for hypervisor uses. Let's follow other examples
and return a hypervisor vendor id signature as another way for software
to find if it is running under FEX-Emu.
2022-03-01 20:47:03 -08:00
Mai M a7ad7f4456 Merge pull request #1593 from Sonicadvance1/add_robin_map
Adds tsl::robin_map
2022-03-01 23:24:34 -05:00
Mai M 46ac05e3cf Merge pull request #1599 from Sonicadvance1/deprecated_distutils
Scripts: Stop using deprecated Distutils
2022-03-01 23:23:55 -05:00
Ryan Houdek 9911fe68d4 Scripts: Stop using deprecated Distutils
According to PEP 386: https://www.python.org/dev/peps/pep-0386/

distutils is deprecated and will be removed in an upcoming python
version.

Switch over to pkg_resources for version parsing and comparison
2022-03-01 07:10:41 -08:00
Ryan Houdek 57a56545b2 Merge pull request #1589 from Sonicadvance1/fix_vixl_assert
Update vixl to fix assert
2022-03-01 05:35:31 -08:00
Ryan Houdek b4e0565907 Merge pull request #1590 from Sonicadvance1/termux_fixes
Termux fixes
2022-03-01 04:08:48 -08:00
Ryan Houdek 5137af5bae Fixes epoxy include in FEXConfig and FEXLogServer 2022-03-01 03:55:19 -08:00
Ryan Houdek 385ed2c2ec Update imgui to fix autodetect 2022-03-01 03:55:16 -08:00
Ryan Houdek 3418cc8054 LinuxSyscalls: More type fixes 2022-03-01 03:55:15 -08:00
Ryan Houdek 1f835ea1f4 LinuxSyscalls: Fixes semid and ipc types
Newer headers redefine semid_ds and ipc_perm as semid64_ds and
ipc64_perm silently.
Use the new types directly since we are a 64-bit only application.
2022-03-01 03:55:12 -08:00
Ryan Houdek 0a40753624 More missing include fixes 2022-03-01 03:55:10 -08:00
Ryan Houdek 609538e758 Utils/Allocator: Use a namespace alias for pmr
This still lives under experimental in Termux environment
2022-03-01 03:55:08 -08:00
Ryan Houdek 09010e1292 LinuxSyscalls: Remove unused headers now
These don't even exist on termux.
2022-03-01 03:54:01 -08:00
Ryan Houdek bddf3871ba LinuxSyscalls/x32/Types: Fixes stat type definitions
The time argument definitions are defines in termux.
Rename our definition name of these so we don't get caught by define.
2022-03-01 03:50:18 -08:00
Ryan Houdek 1ca9e56502 LinuxSyscalls/x64/Types: Fix guest_stat definition
kernel types don't exist in termux. Use uint64_t and int64_t directly.

Also using reserved `__` causes compile failure.
2022-03-01 03:50:17 -08:00
Ryan Houdek fe58a9ae2c LinuxSyscalls/Thread: Don't use set_robust_list on Termux
Would get caught by seccomp and crash FEX
2022-03-01 03:50:15 -08:00
Ryan Houdek bc7c4dfa74 LinuxSyscalls/x32/Types: Don't redefine SIGEV defines 2022-03-01 03:50:12 -08:00
Ryan Houdek 7af7b1309d Msg: Switch msqd_t to FEX defined type 2022-03-01 03:50:10 -08:00
Ryan Houdek dd4630750f LinuxSyscalls/Types: Adds missing types for Termux 2022-03-01 03:50:07 -08:00
Ryan Houdek 302da029eb Work around Termux not supporting hardlinks
The Android filesystem they are on just doesn't support them
Instead of hardlinking FEXLoader to FEXInterpter, just build the
executable twice and eat the filesystem cost.
2022-03-01 03:50:05 -08:00
Ryan Houdek 3cc59bc68e LinuxSyscalls: Switches to a bunch of raw syscalls
For older and Termux build environments these helper libc functions
don't exist.
2022-03-01 03:49:31 -08:00
Ryan Houdek 30c27851ef FEXRootFSFetcher: Termux build environments 2022-03-01 03:30:47 -08:00
Ryan Houdek 9b29ff61e9 Adds some missing headers 2022-03-01 03:30:45 -08:00
Ryan Houdek 91f780223b Stop self-defining PAGE_SIZE
We only work on targets with 4096 byte page sizes.
Adds a cmake compile test to ensure this is adhered to.
2022-03-01 03:30:43 -08:00
Ryan Houdek 2933a00b12 Merge pull request #1579 from Sonicadvance1/optimize_syscalls_with_flags
Allow classifying syscalls with flags
2022-03-01 03:23:43 -08:00
Ryan Houdek a0efd2b01f Resolve comments. 2022-02-28 21:04:03 -08:00
Ryan Houdek 3e9dbda146 Classify syscalls 2022-02-28 21:04:03 -08:00
Ryan Houdek 6501715a3c Allow classifying syscalls with flags
In some cases we can generate more optimal code if we have more
information about a syscall which number gets const-propagated.

In particular optimizing through syscalls, not synchronizing state, and
never returning.

- Noreturn is used by a syscall that never returns, like exit.

This means that it never needs to try and synchronize state coming back

- Not synchronizing state and optimizing through syscalls

Useful for syscalls that don't read the state past arguments and only
returns a value.
2022-02-28 21:03:54 -08:00
Ryan Houdek 2af23d9bec Merge pull request #1591 from Sonicadvance1/new_cpus_in_native_fit
Scripts: Updates CPU fitting script for latest CPUs
2022-02-28 07:05:45 -08:00
Ryan Houdek 3ba2d6cfb4 Merge pull request #1586 from Sonicadvance1/testharness_env
TestHarnessRunner: Wire up environment variable option setting
2022-02-28 07:05:28 -08:00
Ryan Houdek abd266441c unittests: Implements 3DNow! unit tests
Covers the full space, of which there aren't many.

3DNow! unit tests are disabled on the CI runner since the x86 CPU in CI
doesn't support it.
2022-02-28 04:06:14 -08:00
Ryan Houdek b1b6078518 CPUID: Enables 3DNow! + Extensions
Now that we support these
2022-02-28 04:05:18 -08:00
Ryan Houdek a9a89bea9f OpcodeDispatcher: Implements all the 3DNow! instructions
This picks up all the instruction implementations, including 3DNow!
Extended and the Geode specific instructions that were added.

Most of these match preexisting SSE instructions except that they
operate at 64-bit and in the MMX registers.
2022-02-28 04:03:55 -08:00
Ryan Houdek 9b1b2e6496 X86Tables: Fills out 3DNow tables
Fully decoded the same way and adds the Geode specific instruction
decodings as well.
2022-02-28 04:02:57 -08:00
Ryan Houdek 638da92f45 Frontend: Fixes minor bug decoding 3DNow!
We already decoded the modrm `rm` bits, check while decoding modrm to
ensure we don't try decoding it again.
Was causing double decoding of SIB and displacement bytes, breaking
things
2022-02-28 04:01:32 -08:00
Ryan Houdek 0f3a169cc4 Opdispatcher: Minor bug fix with unimplemented op
If multiblock isn't enabled then on Unimplemented op we shouldn't create
a new block.

Was causing IR validation to get angry
2022-02-28 04:00:41 -08:00
Ryan Houdek c7dd176799 IR: Adds a VRev64 op
This directly matches the AArch64 instruction and will be used shortly
2022-02-28 04:00:13 -08:00
Ryan Houdek 23daf4cb72 LinuxAllocator: Fixes bug with old kernels and hint allocation
In the face of an application using MAP_FIXED_NOREPLACE AND the host
linux kernel doesn't understand this flag. Then we were falling down the
hint allocation path which would allocate a pointer in 64-bit space,
returning this pointer to a 32-bit userspace and breaking things.

Now when the hint fails with this flag, we know that it intersecting a
range and can early exit.
2022-02-26 21:12:44 -08:00
Ryan Houdek bb7fa84fb8 Scripts: Updates CPU fitting script for latest CPUs
Clang-13 doesn't yet understand the latest ARM CPUs so just document them
and set to the closest thing.
2022-02-26 02:14:52 -08:00
Ryan Houdek 3325ba52b9 Adds tsl::robin_map
This will be used with the code serialization service soon
2022-02-26 00:43:43 -08:00
Ryan Houdek 7f47fe6d73 Update vixl to fix assert
Any hardware using MTE will assert without this
2022-02-24 13:46:56 -08:00
Ryan Houdek ee165379c5 TestHarnessRunner: Wire up environment variable option setting
Wire up the environment variable option setting so asm files can set
these and it works
2022-02-24 13:37:39 -08:00
Ryan Houdek fa554d3096 Merge pull request #1588 from Azkali/main
Improve compatibility with older uapi kernel headers
2022-02-24 01:34:24 -08:00
The Great Wizard Azkali d60710d3f1 Define proper statx syscall depending on CPU architecture 2022-02-24 10:20:28 +01:00
Azkali 75988b2ae5 Improve compatibility with older uapi kernel headers
Following up the work previously done in 2079f6b3c7.
Adding more defines for older Linux uapi headers missing some defines.
2022-02-24 09:49:56 +01:00
Mai M 30803c66f7 Merge pull request #1587 from Sonicadvance1/fix_missing_telemetry_names
Telemetry: Fix missing telemetry names
2022-02-22 21:46:00 -05:00
Ryan Houdek a9d838fa27 Telemetry: Fix missing telemetry names
Didn't have names for tearing
2022-02-22 18:30:40 -08:00
Ryan Houdek 2b8f60c108 TestHarness: Support for asm files having the option to set config options
Allows some something like the following:
"Env": {
  "FEX_MAXINST": "500"
}

Not that I would recommend overriding MAXINST in the asm tests, as
command line overrides that
2022-02-21 14:53:27 -08:00
Mai M 5ec6ee5b69 Merge pull request #1578 from Sonicadvance1/update_vixl
Updates vixl for new cursor updating methods
2022-02-17 16:18:26 -05:00
Mai M 0db7205f61 Merge pull request #1582 from Sonicadvance1/ccache_option
Adds option to disable ccache
2022-02-16 22:49:48 -05:00
Ryan Houdek 6027494d69 Adds option to disable ccache
Can be useful when running static analysis tools
2022-02-16 19:24:24 -08:00
Mai M 252dcfe26f Merge pull request #1581 from Sonicadvance1/add_required_growsdown
FEXLoader: Adds back required MAP_GROWSDOWN
2022-02-16 19:07:40 -05:00
Ryan Houdek 944c93c10c FEXLoader: Adds back required MAP_GROWSDOWN
I was overzealous with my removal of MAP_GROWSDOWN.
We still require the primary thread to have this flag set.
2022-02-16 15:41:44 -08:00
Ryan Houdek a5fb7e7313 Merge pull request #1580 from Sonicadvance1/fix_hostthunks_install
Fixes Host and guest thunks install path
2022-02-15 15:46:10 -08:00
Ryan Houdek 364b3380fc Fixes Host and guest thunks install path
Hosts were using the cmake install path with $DESTDIR which duplicates
paths.

GuestThunks were doing some magic that wasn't actually necessary
2022-02-15 15:36:31 -08:00
Ryan Houdek a0edab8040 Updates vixl for new cursor updating methods
These will be required for code cache
2022-02-14 16:33:51 -08:00
Ryan Houdek b65194f433 Merge pull request #1576 from Sonicadvance1/move_x87_constant_helpers
JIT: Implements x87 fallback helpers as lookups in to state
2022-02-14 14:13:33 -08:00
Ryan Houdek 0d1c9cd7df JIT: Implements x87 fallback helpers as lookups in to state
This allows x87 fallbacks to be loaded from the upcoming code cache
without relocations.

Only 40 pointers necessary to store and means x87 code won't hit
relocations heavily.

Probably improves performance slightly on the x86 host side but should
be neglible.

Needs #1574 and #1575 merged first.
2022-02-14 14:00:10 -08:00
Ryan Houdek 99dcda7f8c Merge pull request #1575 from Sonicadvance1/move_constant_functions_x86
JITx86: Switches over to loading pointers from state
2022-02-14 13:59:07 -08:00
Ryan Houdek 2aa4d33de5 JITx86: Switches over to loading pointers from state
Just like the previous AArch64 JIT.
These pointers are process or thread specific depending on the pointer
and should be loaded from the State object.

Performance here might slightly increase.

This is required for code cache on x86

Needs #1574 merged first.
2022-02-14 13:49:54 -08:00
Ryan Houdek 1ef78d0a00 Merge pull request #1574 from Sonicadvance1/move_constant_functions
ARMJIT: Switches over to loading pointers from state
2022-02-14 13:39:28 -08:00
Stefanos Kornilios Mitsis Poiitidis 64d1840f1e Merge pull request #1571 from Sonicadvance1/fix_syscall_strace
Linux: Fix missing types for syscall strace
2022-02-14 19:05:48 +02:00
Stefanos Kornilios Mitsis Poiitidis 34530236f9 Merge pull request #1573 from Sonicadvance1/disable_int_tests_with_no_int
unittests: Disables Interpreter tests when its disabled
2022-02-14 19:00:13 +02:00
Ryan Houdek f5a9da082f ARMJIT: Switches over to loading pointers from state
These pointers are process or thread specific depending on which pointer
it is.
All of these pointers end up getting used inside of the JIT blocks
themselves and with code caching would result in a ton of relocations
occuring inside the code.

The pointers used within the dispatcher don't currently matter since I'm
not expecting to cache the dispatcher itself. It's only a page per
thread after all. This does move us significantly closer towards using a
single dispatcher for all threads though.

The performance impact of this change is unlikely to be felt at all,
some locations have less code generation which could improve perf
slightly. Some locations move from a 1-3 cycle constant calculation to a
4 cycle load, hard to be felt since it gets hidden by other
instructions.

x86-64 JIT will be added soon after this
2022-02-13 23:06:26 -08:00
Ryan Houdek 9b49e8cb59 FEXCore: Adds utility class for class member function casting
Adds validation for our class member casting to ensure we don't try
casting a virtual member.
2022-02-13 23:06:25 -08:00
Ryan Houdek 1b6d20b731 unittests: Disables Interpreter tests when its disabled
Would result in failures if you weren't expecting it.
2022-02-11 12:55:10 -08:00
Mai M e9c7c76174 Merge pull request #1572 from Sonicadvance1/LoadConstant_no_opt
Arm64Emitter: Allow non-optimizing LoadConstant
2022-02-11 00:50:56 -05:00
Ryan Houdek cbce06d012 Arm64Emitter: Allow non-optimizing LoadConstant
This is pulled from the code cache PR. Will be necessary for supporting
relocations.

Not currently being used but will be once we have code caching in place.
2022-02-10 20:04:02 -08:00
Ryan Houdek d95326b23b Linux: Fix missing types for syscall strace 2022-02-10 19:46:43 -08:00
Mai M 43fada7555 Merge pull request #1568 from Sonicadvance1/fix_musl_load
ELFCodeLoader: Fixes typo in AT_BASE calculation
2022-02-10 21:25:48 -05:00
Mai M afa7172cb1 Merge pull request #1570 from Sonicadvance1/remove_growsdown
Removes MAP_GROWSDOWN usage
2022-02-10 21:25:23 -05:00
Ryan Houdek b88d8a7cc4 Removes MAP_GROWSDOWN usage
This is just a memory leak waiting to happen.
Only the primary thread in an application really should have this set
since the kernel cleans it up.

We only ever allocate the primary thread of the guest application then
every host thread's stack on top of that. It's up to the guest when it
is cloning to set up new stack pointers, we don't manage that.

We are already allocating the first thread's size at the soft stack
limit with RLIMIT_STACK anyway.

Fixes #1556
2022-02-10 18:03:18 -08:00
Ryan Houdek 6add09b78b ELFCodeLoader: Fixes typo in AT_BASE calculation
Fixes executing musl applications with the dynamic linker.

It was using the main executable's p_offset instead of the
interpreter's.
Wasn't a problem with glibc since it uses a different symbol to find the
base (Don't ask me why it does this).

musl dynamic linker on the other hand just uses AT_BASE directly and
since it was calculated incorrectly it was crashing.

Testing application was `ls` which had a p_offset of 0x40, so it would
try and read some values from AT_BASE, starting at an offset below where
it was mapped.
2022-02-10 17:43:58 -08:00
Mai M 4bb3a54ccf Merge pull request #1566 from neobrain/refactor_thunk_misc
Miscellaneous thunk cleanups
2022-02-10 15:33:34 -05:00
Mai M 76f86e51a6 Merge pull request #1567 from Sonicadvance1/fix_fexgetconfig_rootfs
FEXGetConfig: Fix --current-rootfs option
2022-02-10 15:15:26 -05:00
Ryan Houdek 2a1b27df58 FEXGetConfig: Fix --current-rootfs option
If the configured rootfs wasn't a squashfs then it was failing to return
the directory.

Now it works for both squashfs and directory rootfs again.
2022-02-10 11:45:29 -08:00
Tony Wasserka 68426735a5 Thunks: Clean up ASTMatcher-based testing helpers
The run_thunkgen* helpers now parse generated source code and return its AST
representation, so HasASTMatching helper calls don't each need to redundantly
compile it themselves. This also ensures the generator output actually compiles
in tests where we didn't explicitly check that before.

This also allows printing the full AST of the generator output on test
failures. This must be enabled manually by changing a variable in the ostream
output operator for SourceWithAST.
2022-02-10 12:11:52 +01:00
Tony Wasserka ee6b558000 Thunks: Fix warning about unused field 2022-02-10 12:11:52 +01:00
Tony Wasserka 3c7872c6d5 Thunks: Rename FrontendAction to GenerateThunkLibsAction 2022-02-10 12:11:52 +01:00
Tony Wasserka 4aed6fc56c Thunks/gen: Remove now unneeded code 2022-02-10 12:11:52 +01:00
Tony Wasserka 5d555a10a2 Thunks: Explicitly put thunks into the text library section
Previously, defining zero-initialized variables right before LOAD_LIB
could cause the compiler to put thunk definitions into bss, hence triggering
errors during assembly ("attempt to store non-zero value in section `.bss'").
2022-02-10 12:11:52 +01:00
Ryan Houdek 5854d4ad1c Merge pull request #1565 from neobrain/refactor_thunk_ide_integration
Enable proper IDE integration of thunk libraries
2022-02-10 02:44:29 -08:00
Tony Wasserka bd6999eb87 CMake: Clean up build architecture for ThunkLibs
Host thunk libraries are always built as part of the main project now.
Guest thunk libraries are still cross-compiled in a CMake ExternalProject,
but *additionally* there are CMake targets in the main project to make
sure IDE engines can properly handle guest source files.
2022-02-10 11:23:43 +01:00
Tony Wasserka 3ebb2eaf0d Thunks: Fix guest libs build on clang 2022-02-10 11:23:41 +01:00
Mai M defd3be30c Merge pull request #1563 from Sonicadvance1/fix_auto_script
Updates Readme to fix install script
2022-02-09 18:18:39 -05:00
Mai M a8e5a0a68e Merge pull request #1562 from Sonicadvance1/remove_debug_memory_mapping
FEXLoader: Removes memory mapping check on startup
2022-02-09 18:18:18 -05:00
Ryan Houdek f27c43ff8a Updates Readme to fix install script
Fixes an issue where the FEXRootFSFetcher wouldn't get a any user input
and just fail out.
Save it to the tmp folder and execute from there instead.

Fixes #1557
2022-02-09 14:01:48 -08:00
Ryan Houdek 8840fc818c FEXLoader: Removes memory mapping check on startup
FEX always builds with PIE and we don't hit this issue anymore anyway.
If some application wants to inject a page in to the lower 32-bits then
we have no reason to complain about it anymore. Just let it go and
hopefully they know what they are doing.

Fixes #1559
2022-02-09 13:52:05 -08:00
Mai M ffcaf294e3 Merge pull request #1555 from Sonicadvance1/weirdo_edge_case
OpcodeDispatcher: Fixes weirdo edge case in segment moving
2022-02-08 00:26:55 -05:00
Ryan Houdek 6b7a84bef2 OpcodeDispatcher: Fixes weirdo edge case in segment moving
Just noticed this while casually reading the x86 architecture manuals.
The move segment registers instructions ignore the REX.R prefix on the
segment register.

Previously this was expected to create an invalid register selection.
A little bit silly but sure, support it.
2022-02-07 21:12:53 -08:00
Mai M 287c65dc64 Merge pull request #1554 from Sonicadvance1/fix_tricky_stat
Linux: x32: Fixes tricky stat64 defines
2022-02-06 19:52:31 -05:00
Mai M 62397bb19c Merge pull request #1553 from Sonicadvance1/fix_sigevent
Linux: Make sure to use correct accessors for sigevent
2022-02-06 19:52:14 -05:00
Ryan Houdek 0216bcf27f Linux: x32: Fixes tricky stat64 defines
Some build environments use a define to change stat64 and statfs64 to be
the same definition as stat and statfs.

Check if the define exists and if it does then remove the 64bit
constructors.
2022-02-06 16:05:27 -08:00
Ryan Houdek b981fcfe42 Linux: Make sure to use correct accessors for sigevent
Some of these are defined differently depending on environment
2022-02-06 15:33:07 -08:00
Mai M cb491a8acb Merge pull request #1552 from Sonicadvance1/fix_ucontext_copy
UContext: Fixes 32-bit siginfo_t copying definition
2022-02-06 18:23:04 -05:00
Mai M 1b99495b4b Merge pull request #1551 from Sonicadvance1/fix_older_env
Some fixes for older environments
2022-02-06 18:22:26 -05:00
Ryan Houdek 3c5a2cec90 UContext: Fixes 32-bit siginfo_t copying definition
The host provided siginfo_t definition can vary depending on the build
environment.
What doesn't change however is how the data is laid out.
It's always 128bytes, The first three 32-bit words are always known.
The 64-bit host side always has an additional 32-bit pad member.
Then the remaining bytes is the sifields.
2022-02-06 15:03:56 -08:00
Ryan Houdek 9a64e7f100 Linux: Renamed some 64-bit syscall names
Some build environments use defines to rename these. Which breaks our
naming
2022-02-06 14:19:30 -08:00
Ryan Houdek 8e8baec47a Linux: Use raw syscalls for pkey syscalls
For older libc environments
2022-02-06 14:19:30 -08:00
Ryan Houdek 59ca60e39f Fixes a bunch of header includes
Necessary for older build environments
2022-02-06 14:19:30 -08:00
243 changed files with 8304 additions and 6628 deletions

No files matched your search

+4
View File
@@ -45,3 +45,7 @@
[submodule "External/Catch2"]
path = External/Catch2
url = https://github.com/catchorg/Catch2.git
[submodule "External/robin-map"]
shallow = true
path = External/robin-map
url = https://github.com/Tessil/robin-map.git
+76 -28
View File
@@ -20,6 +20,7 @@ option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
@@ -79,10 +80,12 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
add_definitions(-D_M_ARM_64=1)
endif()
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
message(STATUS "CCache enabled")
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
if (ENABLE_CCACHE)
find_program(CCACHE_PROGRAM ccache)
if(CCACHE_PROGRAM)
message(STATUS "CCache enabled")
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
endif()
endif()
if (ENABLE_XRAY)
@@ -112,6 +115,65 @@ if (NOT ENABLE_OFFLINE_TELEMETRY)
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
endif()
# Check if the build target page size is 4096
include(CheckCSourceRuns)
check_c_source_runs(
"#include <unistd.h>
int main(int argc, char* argv[])
{
return getpagesize() == 4096 ? 0 : 1;
}"
PAGEFILE_RESULT
)
if (NOT ${PAGEFILE_RESULT})
message(FATAL_ERROR "Host PAGE_SIZE is not 4096. Can't build on this target")
endif()
include(CheckCXXSourceCompiles)
check_cxx_source_compiles(
"#include <sys/user.h>
int main() {
return PAGE_SIZE;
}
"
HAS_PAGESIZE)
check_cxx_source_compiles(
"#include <sys/user.h>
int main() {
return PAGE_SHIFT;
}
"
HAS_PAGESHIFT)
check_cxx_source_compiles(
"#include <sys/user.h>
int main() {
return PAGE_MASK;
}
"
HAS_PAGEMASK)
if (NOT HAS_PAGESIZE)
add_definitions(-DPAGE_SIZE=4096)
endif()
if (NOT HAS_PAGESHIFT)
add_definitions(-DPAGE_SHIFT=12)
endif()
if (NOT HAS_PAGEMASK)
add_definitions("-DPAGE_MASK=(~(PAGE_SIZE-1))")
endif()
if(DEFINED ENV{TERMUX_VERSION})
add_definitions(-DTERMUX_BUILD=1)
set(TERMUX_BUILD 1)
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
set(ENABLE_JEMALLOC FALSE)
endif()
if (ENABLE_STATIC_PIE)
if (_M_ARM_64 AND ENABLE_LLD)
message (FATAL_ERROR "Static linking does not currently work with AArch64+LLD. Use GNU ld for now.")
@@ -265,6 +327,8 @@ set (CMAKE_LINKER_FLAGS_RELWITHDEBINFO "${CMAKE_LINKER_FLAGS_RELWITHDEBINFO} -fn
set (CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fomit-frame-pointer")
set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-pointer")
include_directories(External/robin-map/include/)
add_subdirectory(External/vixl/)
include_directories(External/vixl/src/)
@@ -480,30 +544,14 @@ endif()
if (BUILD_THUNKS)
add_subdirectory(ThunkLibs/Generator)
# Thunk targets for both host libraries and IDE integration
add_subdirectory(ThunkLibs/HostLibs)
# Thunk targets for IDE integration of guest code, only
add_subdirectory(ThunkLibs/GuestLibs)
# Thunk targets for guest libraries
include(ExternalProject)
ExternalProject_Add(host-libs
PREFIX host-libs
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/HostLibs"
BINARY_DIR "Host"
CMAKE_ARGS
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
INSTALL_COMMAND ""
BUILD_ALWAYS ON
DEPENDS thunkgen
)
install(
CODE "MESSAGE(\"-- Installing: host-libs\")"
CODE "
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkHostsInstall
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Host
)"
DEPENDS host-libs
)
ExternalProject_Add(guest-libs
PREFIX guest-libs
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
@@ -523,7 +571,7 @@ if (BUILD_THUNKS)
install(
CODE "MESSAGE(\"-- Installing: guest-libs\")"
CODE "
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target ThunkGuestsInstall
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
)"
DEPENDS guest-libs
+6 -43
View File
@@ -7,17 +7,12 @@ OpClasses = collections.OrderedDict()
def get_ir_classes(ops, defines):
global OpClasses
for op_key, op_vals in ops.items():
if not ("Last" in op_vals):
OpClass = "#Unknown"
for op_class, opslist in ops.items():
if not (op_class in OpClasses):
OpClasses[op_class] = []
if ("OpClass" in op_vals):
OpClass = op_vals["OpClass"]
if not (OpClass in OpClasses):
OpClasses[OpClass] = []
OpClasses[OpClass].append([op_key, op_vals])
for op, op_val in opslist.items():
OpClasses[op_class].append([op, op_val])
# Sort the dictionary after we are done parsing it
OpClasses = collections.OrderedDict(sorted(OpClasses.items()))
@@ -38,41 +33,9 @@ def print_ir_ops():
op_key = op[0]
op_vals = op[1]
output_file.write("## %s\n" % (op_key))
HasDest = ("HasDest" in op_vals and op_vals["HasDest"] == True)
HasSSAArgs = ("SSAArgs" in op_vals and len(op_vals["SSAArgs"]) > 0)
HasSSAArgNames = "SSANames" in op_vals
HasArgs = "Args" in op_vals
SSAArgsCount = 0
ArgCount = 0
if (HasSSAArgs):
SSAArgsCount = int(op_vals["SSAArgs"])
if (HasArgs):
ArgCount = len(op_vals["Args"])
TotalArgsCount = SSAArgsCount + (ArgCount / 2)
output_file.write(">")
if (HasDest):
output_file.write("%dest = ")
output_file.write("%s " % op_key)
ArgComma = (", ", "")
if (HasSSAArgs):
for i in range(0, SSAArgsCount):
FinalArg = (i + 1) == TotalArgsCount
if (HasSSAArgNames):
output_file.write("%%%s%s" % (op_vals["SSANames"][i], ArgComma[FinalArg]))
else:
output_file.write("%%ssa%d%s" % (i, ArgComma[FinalArg]))
if (HasArgs):
Args = op_vals["Args"]
for i in range(0, ArgCount, 2):
FinalArg = ((i / 2) + SSAArgsCount + 1) == TotalArgsCount
data_type = Args[i]
data_name = Args[i + 1]
output_file.write("\<%s %s\>%s" % (data_type, data_name, ArgComma[FinalArg]))
output_file.write(op_key)
output_file.write("\n\n")
Vendored Regular → Executable
+469 -382
View File
File diff suppressed because it is too large. Load diff
@@ -28,20 +28,32 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size) : vixl::
SetCPUFeatures(Features);
}
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) {
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad) {
bool Is64Bit = Reg.IsX();
int Segments = Is64Bit ? 4 : 2;
if (Is64Bit && ((~Constant)>> 16) == 0) {
movn(Reg, (~Constant) & 0xFFFF);
if (NOPPad) {
nop(); nop(); nop();
}
return;
}
int NumMoves = 1;
movz(Reg, (Constant) & 0xFFFF, 0);
for (int i = 1; i < Segments; ++i) {
uint16_t Part = (Constant >> (i * 16)) & 0xFFFF;
if (Part) {
movk(Reg, Part, i * 16);
++NumMoves;
}
}
if (NOPPad) {
for (int i = NumMoves; i < Segments; ++i) {
nop();
}
}
}
@@ -128,51 +140,75 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
}
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t SpillMask) {
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
if (StaticRegisterAllocation()) {
for (size_t i = 0; i < SRA64.size(); i+=2) {
auto Reg1 = SRA64[i];
auto Reg2 = SRA64[i+1];
if (((1U << Reg1.GetCode()) & SpillMask) &&
((1U << Reg2.GetCode()) & SpillMask)) {
if (((1U << Reg1.GetCode()) & GPRSpillMask) &&
((1U << Reg2.GetCode()) & GPRSpillMask)) {
stp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg1.GetCode()) & SpillMask)) {
else if (((1U << Reg1.GetCode()) & GPRSpillMask)) {
str(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg2.GetCode()) & SpillMask)) {
else if (((1U << Reg2.GetCode()) & GPRSpillMask)) {
str(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
}
}
if (FPRs) {
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
stp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
auto Reg1 = SRAFPR[i];
auto Reg2 = SRAFPR[i+1];
if (((1U << Reg1.GetCode()) & FPRSpillMask) &&
((1U << Reg2.GetCode()) & FPRSpillMask)) {
stp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
else if (((1U << Reg1.GetCode()) & FPRSpillMask)) {
str(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
else if (((1U << Reg2.GetCode()) & FPRSpillMask)) {
str(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i+1][0])));
}
}
}
}
}
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t FillMask) {
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
if (StaticRegisterAllocation()) {
for (size_t i = 0; i < SRA64.size(); i+=2) {
auto Reg1 = SRA64[i];
auto Reg2 = SRA64[i+1];
if (((1U << Reg1.GetCode()) & FillMask) &&
((1U << Reg2.GetCode()) & FillMask)) {
if (((1U << Reg1.GetCode()) & GPRFillMask) &&
((1U << Reg2.GetCode()) & GPRFillMask)) {
ldp(Reg1, Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg1.GetCode()) & FillMask)) {
else if (((1U << Reg1.GetCode()) & GPRFillMask)) {
ldr(Reg1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i])));
}
else if (((1U << Reg2.GetCode()) & FillMask)) {
else if (((1U << Reg2.GetCode()) & GPRFillMask)) {
ldr(Reg2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i+1])));
}
}
if (FPRs) {
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
ldp(SRAFPR[i].Q(), SRAFPR[i+1].Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
auto Reg1 = SRAFPR[i];
auto Reg2 = SRAFPR[i+1];
if (((1U << Reg1.GetCode()) & FPRFillMask) &&
((1U << Reg2.GetCode()) & FPRFillMask)) {
ldp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
else if (((1U << Reg1.GetCode()) & FPRFillMask)) {
ldr(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
}
else if (((1U << Reg2.GetCode()) & FPRFillMask)) {
ldr(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i+1][0])));
}
}
}
}
@@ -61,9 +61,15 @@ protected:
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
vixl::aarch64::CPU CPU;
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant);
void SpillStaticRegs(bool FPRs = true, uint32_t SpillMask = ~0U);
void FillStaticRegs(bool FPRs = true, uint32_t FillMask = ~0U);
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
static constexpr uint32_t CALLER_GPR_MASK = 0b0011'1111'1111'1111'1111;
// This isn't technically true because the lower 64-bits of v8..v15 are callee saved
// We can't guarantee only the lower 64bits are used so flush everything
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
void PushDynamicRegsAndLR();
void PopDynamicRegsAndLR();
+29 -4
View File
@@ -445,7 +445,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
(0 << 27) | // OSXSAVE
(SUPPORTS_AVX << 28) | // AVX
(0 << 29) | // F16C
(0 << 30) | // RDRAND
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
(0 << 31); // Hypervisor always returns zero
Res.edx =
@@ -647,7 +647,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
(0 << 15) | // Intel Resource Directory Technology Allocation
(0 << 16) | // Reserved
(0 << 17) | // Reserved
(0 << 18) | // RDSEED
(CTX->HostFeatures.SupportsRAND << 18) | // RDSEED
(1 << 19) | // ADCX and ADOX instructions
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
(0 << 21) | // Reserved
@@ -812,6 +812,28 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_1Ah(uint32_t Leaf) {
return Res;
}
// Hypervisor CPUID information leaf
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_4000_0000h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults Res{};
// Maximum supported hypervisor leafs
// We only expose the information leaf
//
// Common courtesy to follow VMWare's "Hypervisor CPUID Interface proposal"
// 4000_0000h - Information leaf. Advertising to the software which hypervisor this is
// 4000_0001h - 4000_000Fh - Hypervisor specific leafs. FEX can use these for anything
// 4000_0010h - 4000_00FFh - "Generic Leafs" - Try not to overwrite, other hypervisors might expect information in these
//
// CPUID documentation information:
// 4000_0000h - 4FFF_FFFFh - No existing or future CPU will return information in this range
// Reserved entirely for VMs to do whatever they want.
Res.eax = 0x40000000;
// EBX, EDX, ECX become the hypervisor ID signature
constexpr static char HypervisorID[12] = "FEXIFEXIEMU";
memcpy(&Res.ebx, HypervisorID, sizeof(HypervisorID));
return Res;
}
// Highest extended function implemented
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
FEXCore::CPUID::FunctionResults Res{};
@@ -902,8 +924,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
(1 << 27) | // RDTSCP
(0 << 28) | // Reserved
(1 << 29) | // Long Mode
(0 << 30) | // 3DNow! Extensions
(0 << 31); // 3DNow!
(1 << 30) | // 3DNow! Extensions
(1 << 31); // 3DNow!
return Res;
}
@@ -1204,6 +1226,9 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
#ifndef CPUID_AMD
RegisterFunction(0x1A, &CPUIDEmu::Function_1Ah);
#endif
// Hypervisor CPUID information leaf
RegisterFunction(0x4000'0000, &CPUIDEmu::Function_4000_0000h);
// Largest extended function number
RegisterFunction(0x8000'0000, &CPUIDEmu::Function_8000_0000h);
// Processor vendor
+2
View File
@@ -3,6 +3,7 @@
#include <cstdint>
#include <unordered_map>
#include <utility>
#include <vector>
#include <FEXCore/Core/CPUID.h>
#include <FEXCore/Config/Config.h>
@@ -78,6 +79,7 @@ private:
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_4000_0000h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
@@ -53,10 +53,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
// Ptr();
// }
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
Literal l_VirtualMemory {VirtualMemorySize};
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
Literal l_L1Ptr {Thread->LookupCache->GetL1Pointer()};
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
Literal l_CompileBlock {GetCompileBlockPtr()};
@@ -99,7 +96,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
auto RipReg = x2;
// L1 Cache
ldr(x0, &l_L1Ptr);
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.L1Pointer)));
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x3, Shift::LSL, 4));
@@ -121,11 +118,12 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
ldr(x0, &l_PagePtr);
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
if (std::popcount(VirtualMemorySize) == 1) {
and_(x3, RipReg, Thread->LookupCache->GetVirtualMemorySize() - 1);
and_(x3, RipReg, VirtualMemorySize - 1);
}
else {
ldr(x3, &l_VirtualMemory);
LoadConstant(x3, VirtualMemorySize);
and_(x3, RipReg, x3);
}
@@ -159,7 +157,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldr(x0, &l_L1Ptr);
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.L1Pointer)));
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x1, Shift::LSL, 4));
@@ -438,9 +436,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
b(&LoopTop);
}
place(&l_VirtualMemory);
place(&l_PagePtr);
place(&l_L1Ptr);
place(&l_CTX);
place(&l_Sleep);
place(&l_CompileBlock);
@@ -461,6 +457,20 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
}
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto &Pointers = ThreadState->CurrentFrame->Pointers.AArch64;
Pointers.DispatcherLoopTop = AbsoluteLoopTopAddress;
Pointers.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Pointers.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
Pointers.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
Pointers.UnimplementedInstructionHandler = UnimplementedInstructionAddress;
Pointers.OverflowExceptionHandler = OverflowExceptionInstructionAddress;
Pointers.SignalReturnHandler = SignalHandlerReturnAddress;
Pointers.L1Pointer = Thread->LookupCache->GetL1Pointer();
}
}
void Arm64Dispatcher::SpillSRA(void *ucontext, uint32_t IgnoreMask) {
@@ -16,9 +16,9 @@
#include <atomic>
#include <condition_variable>
#include <bits/types/siginfo_t.h>
#include <csignal>
#include <cstring>
#include <signal.h>
namespace FEXCore::CPU {
@@ -5,8 +5,8 @@
#include "Interface/Context/Context.h"
#include "Interface/Core/ArchHelpers/MContext.h"
#include <bits/types/stack_t.h>
#include <cstdint>
#include <signal.h>
#include <stddef.h>
#include <stack>
#include <tuple>
@@ -95,7 +95,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
// L1 Cache
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.L1Pointer)]);
mov(rax, rdx);
and_(rax, LookupCache::L1_ENTRIES_MASK);
@@ -114,8 +114,9 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
mov(r13, Thread->LookupCache->GetPagePointer());
// Full lookup
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
mov(rax, rdx);
mov(rbx, Thread->LookupCache->GetVirtualMemorySize() - 1);
mov(rbx, VirtualMemorySize - 1);
and_(rax, rbx);
shr(rax, 12);
@@ -142,8 +143,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
je(NoBlock);
// Update L1
mov(r13, Thread->LookupCache->GetL1Pointer());
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.L1Pointer)]);
mov(rcx, rdx);
and_(rcx, LookupCache::L1_ENTRIES_MASK);
shl(rcx, 1);
@@ -252,8 +252,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
// XXX: XMM?
// Make sure to adjust the refcounter so we don't clear the cache now
mov(rax, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
add(dword [rax], 1);
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalHandlerRefCountPointer)], 1);
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
@@ -337,6 +336,20 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
}
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto &Pointers = ThreadState->CurrentFrame->Pointers.X86;
Pointers.DispatcherLoopTop = AbsoluteLoopTopAddress;
Pointers.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Pointers.ThreadStopHandler = ThreadStopHandlerAddress;
Pointers.ThreadPauseHandler = ThreadPauseHandlerAddress;
Pointers.UnimplementedInstructionHandler = UnimplementedInstructionAddress;
Pointers.OverflowExceptionHandler = OverflowExceptionInstructionAddress;
Pointers.SignalReturnHandler = SignalHandlerReturnAddress;
Pointers.L1Pointer = Thread->LookupCache->GetL1Pointer();
}
}
X86Dispatcher::~X86Dispatcher() {
+12 -5
View File
@@ -583,8 +583,11 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
return false;
}
else {
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
(this->*Disp)(&NonGPR, ModRM);
// Only decode if we haven't pre-decoded
if (NonGPR.IsNone()) {
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
(this->*Disp)(&NonGPR, ModRM);
}
}
return true;
@@ -857,7 +860,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
case 0x0F: {// Escape Op
uint8_t EscapeOp = ReadByte();
switch (EscapeOp) {
case 0x0F: { // 3DNow!
case 0x0F: [[unlikely]] { // 3DNow!
// 3DNow! Instruction Encoding: 0F 0F [ModRM] [SIB] [Displacement] [Opcode]
// Decode ModRM
uint8_t ModRMByte = ReadByte();
@@ -870,8 +873,12 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
const bool Has16BitAddressing = !CTX->Config.Is64BitMode &&
DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
(this->*Disp)(&DecodeInst->Src[0], ModRM);
// All 3DNow! instructions have the second argument as the rm handler
// We need to decode it upfront to get the displacement out of the way
if (ModRM.mod != 0b11) {
auto Disp = DecodeModRMs_Disp[Has16BitAddressing];
(this->*Disp)(&DecodeInst->Src[0], ModRM);
}
// Take a peek at the op just past the displacement
uint8_t LocalOp = ReadByte();
@@ -55,6 +55,8 @@ HostFeatures::HostFeatures() {
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
SupportsRAND = Features.Has(vixl::CPUFeatures::Feature::kRNG);
// Only supported when FEAT_AFP is supported
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
@@ -80,6 +82,8 @@ HostFeatures::HostFeatures() {
Xbyak::util::Cpu Features{};
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
SupportsFlushInputsToZero = true;
SupportsFloatExceptions = true;
#else
+1
View File
@@ -19,6 +19,7 @@ class HostFeatures final {
bool SupportsCLZERO{};
bool SupportsAtomics{};
bool SupportsRCPC{};
bool SupportsRAND{};
// Float exception behaviour
bool SupportsFlushInputsToZero{};
@@ -18,7 +18,7 @@ namespace FEXCore::CPU {
DEF_OP(TruncElementPair) {
auto Op = IROp->C<IR::IROp_TruncElementPair>();
switch (Op->Size) {
switch (IROp->Size) {
case 4: {
uint64_t *Src = GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
uint64_t Result{};
@@ -27,7 +27,7 @@ DEF_OP(TruncElementPair) {
GD = Result;
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", Op->Size); break;
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", IROp->Size); break;
}
}
@@ -900,7 +900,7 @@ DEF_OP(VExtractToGPR) {
if (SourceSize == 16) {
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
uint64_t Shift = Op->Header.ElementSize * Op->Idx * 8;
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
if (Op->Header.ElementSize == 8)
SourceMask = ~0ULL;
@@ -911,7 +911,7 @@ DEF_OP(VExtractToGPR) {
}
else {
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
uint64_t Shift = Op->Header.ElementSize * Op->Idx * 8;
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
if (Op->Header.ElementSize == 8)
SourceMask = ~0ULL;
+112 -113
View File
@@ -313,30 +313,29 @@ uint64_t AtomicCompareAndSwap(uint64_t expected, uint64_t desired, uint64_t *add
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CASPair>();
uint8_t OpSize = IROp->Size;
// Size is the size of each pair element
switch (OpSize) {
switch (IROp->ElementSize) {
case 4: {
GD = AtomicCompareAndSwap(
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint64_t**>(Data->SSAData, Op->Header.Args[2])
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 8: {
std::atomic<__uint128_t> *MemData = *GetSrc<std::atomic<__uint128_t> **>(Data->SSAData, Op->Header.Args[2]);
std::atomic<__uint128_t> *MemData = *GetSrc<std::atomic<__uint128_t> **>(Data->SSAData, Op->Addr);
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Expected);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Desired);
__uint128_t Expected = Src1;
bool Result = MemData->compare_exchange_strong(Expected, Src2);
memcpy(GDP, Result ? &Src1 : &Expected, 16);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", OpSize); break;
default: LOGMAN_MSG_A_FMT("Unknown CAS size: {}", IROp->ElementSize); break;
}
}
@@ -347,33 +346,33 @@ DEF_OP(CAS) {
switch (OpSize) {
case 1: {
GD = AtomicCompareAndSwap(
*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint8_t**>(Data->SSAData, Op->Header.Args[2])
*GetSrc<uint8_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint8_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint8_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 2: {
GD = AtomicCompareAndSwap(
*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint16_t**>(Data->SSAData, Op->Header.Args[2])
*GetSrc<uint16_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint16_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint16_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 4: {
GD = AtomicCompareAndSwap(
*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint32_t**>(Data->SSAData, Op->Header.Args[2])
*GetSrc<uint32_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint32_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint32_t**>(Data->SSAData, Op->Addr)
);
break;
}
case 8: {
GD = AtomicCompareAndSwap(
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]),
*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]),
*GetSrc<uint64_t**>(Data->SSAData, Op->Header.Args[2])
*GetSrc<uint64_t*>(Data->SSAData, Op->Expected),
*GetSrc<uint64_t*>(Data->SSAData, Op->Desired),
*GetSrc<uint64_t**>(Data->SSAData, Op->Addr)
);
break;
}
@@ -385,26 +384,26 @@ DEF_OP(AtomicAdd) {
auto Op = IROp->C<IR::IROp_AtomicAdd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData += Src;
break;
}
@@ -416,26 +415,26 @@ DEF_OP(AtomicSub) {
auto Op = IROp->C<IR::IROp_AtomicSub>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData -= Src;
break;
}
@@ -447,26 +446,26 @@ DEF_OP(AtomicAnd) {
auto Op = IROp->C<IR::IROp_AtomicAnd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData &= Src;
break;
}
@@ -478,26 +477,26 @@ DEF_OP(AtomicOr) {
auto Op = IROp->C<IR::IROp_AtomicOr>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData |= Src;
break;
}
@@ -509,26 +508,26 @@ DEF_OP(AtomicXor) {
auto Op = IROp->C<IR::IROp_AtomicXor>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
*MemData ^= Src;
break;
}
@@ -540,29 +539,29 @@ DEF_OP(AtomicSwap) {
auto Op = IROp->C<IR::IROp_AtomicSwap>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->exchange(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->exchange(Src);
GD = Previous;
break;
@@ -575,29 +574,29 @@ DEF_OP(AtomicFetchAdd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_add(Src);
GD = Previous;
break;
@@ -610,29 +609,29 @@ DEF_OP(AtomicFetchSub) {
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_sub(Src);
GD = Previous;
break;
@@ -645,29 +644,29 @@ DEF_OP(AtomicFetchAnd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_and(Src);
GD = Previous;
break;
@@ -680,29 +679,29 @@ DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_or(Src);
GD = Previous;
break;
@@ -715,29 +714,29 @@ DEF_OP(AtomicFetchXor) {
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
switch (IROp->Size) {
case 1: {
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Header.Args[0]);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint8_t> *MemData = *GetSrc<std::atomic<uint8_t> **>(Data->SSAData, Op->Addr);
uint8_t Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
uint8_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 2: {
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Header.Args[0]);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint16_t> *MemData = *GetSrc<std::atomic<uint16_t> **>(Data->SSAData, Op->Addr);
uint16_t Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Value);
uint16_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 4: {
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Header.Args[0]);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint32_t> *MemData = *GetSrc<std::atomic<uint32_t> **>(Data->SSAData, Op->Addr);
uint32_t Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
uint32_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
}
case 8: {
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Header.Args[0]);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
std::atomic<uint64_t> *MemData = *GetSrc<std::atomic<uint64_t> **>(Data->SSAData, Op->Addr);
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
uint64_t Previous = MemData->fetch_xor(Src);
GD = Previous;
break;
@@ -751,22 +750,22 @@ DEF_OP(AtomicFetchNeg) {
switch (IROp->Size) {
case 1: {
using Type = uint8_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
case 2: {
using Type = uint16_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
case 4: {
using Type = uint32_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
case 8: {
using Type = uint64_t;
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Header.Args[0]));
GD = AtomicFetchNeg(*GetSrc<Type**>(Data->SSAData, Op->Addr));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
@@ -33,10 +33,6 @@ DEF_OP(GuestCallIndirect) {
LogMan::Msg::DFmt("Unimplemented");
}
DEF_OP(GuestReturn) {
LogMan::Msg::DFmt("Unimplemented");
}
DEF_OP(SignalReturn) {
SignalReturn(Data->State);
}
@@ -19,7 +19,7 @@ DEF_OP(VInsGPR) {
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
uint64_t Offset = Op->Index * Op->Header.ElementSize * 8;
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
if (Op->Header.ElementSize == 8) {
Mask = ~0ULL;
@@ -158,7 +158,7 @@ DEF_OP(F80CVTINT) {
DEF_OP(F80CVTTO) {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch (Op->Size) {
switch (Op->SrcSize) {
case 4: {
float Src = *GetSrc<float *>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp = Src;
@@ -171,14 +171,14 @@ DEF_OP(F80CVTTO) {
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->Size);
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
}
}
DEF_OP(F80CVTTOINT) {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->Size) {
switch (Op->SrcSize) {
case 2: {
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Header.Args[0]);
X80SoftFloat Tmp = Src;
@@ -191,7 +191,7 @@ DEF_OP(F80CVTTOINT) {
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->Size);
default: LogMan::Msg::DFmt("Unhandled size: {}", Op->SrcSize);
}
}
@@ -11,7 +11,6 @@
#include <FEXCore/Utils/LogManager.h>
#include <memory>
#include <bits/types/stack_t.h>
#include <signal.h>
#include <stdint.h>
#include <unordered_map>
@@ -1,3 +1,4 @@
#include "FEXCore/Core/CoreState.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/F80Ops.h"
@@ -7,93 +8,140 @@
namespace FEXCore::CPU {
template<typename R, typename... Args>
static FallbackInfo GetFallbackInfo(R(*fn)(Args...)) {
return {FABI_UNKNOWN, (void*)fn};
static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_UNKNOWN, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float)) {
return {FABI_F80_F32, (void*)fn};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F32, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double)) {
return {FABI_F80_F64, (void*)fn};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t)) {
return {FABI_F80_I16, (void*)fn};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t)) {
return {FABI_VOID_U16, (void*)fn};
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_VOID_U16, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t)) {
return {FABI_F80_I32, (void*)fn};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I32, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat)) {
return {FABI_F32_F80, (void*)fn};
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F32_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat)) {
return {FABI_F64_F80, (void*)fn};
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat)) {
return {FABI_I16_F80, (void*)fn};
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I16_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat)) {
return {FABI_I32_F80, (void*)fn};
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I32_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat)) {
return {FABI_I64_F80, (void*)fn};
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat)) {
return {FABI_I64_F80_F80, (void*)fn};
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_F80_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat)) {
return {FABI_F80_F80, (void*)fn};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat)) {
return {FABI_F80_F80_F80, (void*)fn};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F80_F80, (void*)fn, HandlerIndex};
}
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
Info[Core::OPINDEX_F80LOADFCW] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW).fn);
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4).fn);
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8).fn);
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4).fn);
Info[Core::OPINDEX_F80CVT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8).fn);
Info[Core::OPINDEX_F80CVTINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2).fn);
Info[Core::OPINDEX_F80CVTINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4).fn);
Info[Core::OPINDEX_F80CVTINT_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8).fn);
Info[Core::OPINDEX_F80CVTINT_TRUNC2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2).fn);
Info[Core::OPINDEX_F80CVTINT_TRUNC4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4).fn);
Info[Core::OPINDEX_F80CVTINT_TRUNC8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8).fn);
Info[Core::OPINDEX_F80CMP_0] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>, Core::OPINDEX_F80CMP_0).fn);
Info[Core::OPINDEX_F80CMP_1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>, Core::OPINDEX_F80CMP_1).fn);
Info[Core::OPINDEX_F80CMP_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>, Core::OPINDEX_F80CMP_2).fn);
Info[Core::OPINDEX_F80CMP_3] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>, Core::OPINDEX_F80CMP_3).fn);
Info[Core::OPINDEX_F80CMP_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>, Core::OPINDEX_F80CMP_4).fn);
Info[Core::OPINDEX_F80CMP_5] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>, Core::OPINDEX_F80CMP_5).fn);
Info[Core::OPINDEX_F80CMP_6] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>, Core::OPINDEX_F80CMP_6).fn);
Info[Core::OPINDEX_F80CMP_7] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>, Core::OPINDEX_F80CMP_7).fn);
Info[Core::OPINDEX_F80CVTTOINT_2] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2).fn);
Info[Core::OPINDEX_F80CVTTOINT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4).fn);
// Unary
Info[Core::OPINDEX_F80ROUND] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ROUND>::handle, Core::OPINDEX_F80ROUND).fn);
Info[Core::OPINDEX_F80F2XM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80F2XM1>::handle, Core::OPINDEX_F80F2XM1).fn);
Info[Core::OPINDEX_F80TAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80TAN>::handle, Core::OPINDEX_F80TAN).fn);
Info[Core::OPINDEX_F80SQRT] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SQRT>::handle, Core::OPINDEX_F80SQRT).fn);
Info[Core::OPINDEX_F80SIN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SIN>::handle, Core::OPINDEX_F80SIN).fn);
Info[Core::OPINDEX_F80COS] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80COS>::handle, Core::OPINDEX_F80COS).fn);
Info[Core::OPINDEX_F80XTRACT_EXP] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle, Core::OPINDEX_F80XTRACT_EXP).fn);
Info[Core::OPINDEX_F80XTRACT_SIG] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle, Core::OPINDEX_F80XTRACT_SIG).fn);
Info[Core::OPINDEX_F80BCDSTORE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle, Core::OPINDEX_F80BCDSTORE).fn);
Info[Core::OPINDEX_F80BCDLOAD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle, Core::OPINDEX_F80BCDLOAD).fn);
// Binary
Info[Core::OPINDEX_F80ADD] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ADD>::handle, Core::OPINDEX_F80ADD).fn);
Info[Core::OPINDEX_F80SUB] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SUB>::handle, Core::OPINDEX_F80SUB).fn);
Info[Core::OPINDEX_F80MUL] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80MUL>::handle, Core::OPINDEX_F80MUL).fn);
Info[Core::OPINDEX_F80DIV] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80DIV>::handle, Core::OPINDEX_F80DIV).fn);
Info[Core::OPINDEX_F80FYL2X] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2X>::handle, Core::OPINDEX_F80FYL2X).fn);
Info[Core::OPINDEX_F80ATAN] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80ATAN>::handle, Core::OPINDEX_F80ATAN).fn);
Info[Core::OPINDEX_F80FPREM1] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM1>::handle, Core::OPINDEX_F80FPREM1).fn);
Info[Core::OPINDEX_F80FPREM] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80FPREM>::handle, Core::OPINDEX_F80FPREM).fn);
Info[Core::OPINDEX_F80SCALE] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle, Core::OPINDEX_F80SCALE).fn);
}
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
uint8_t OpSize = IROp->Size;
switch(IROp->Op) {
case IR::OP_F80LOADFCW: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle);
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW);
return true;
}
case IR::OP_F80CVTTO: {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
switch (Op->Size) {
switch (Op->SrcSize) {
case 4: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4);
return true;
}
case 8: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8);
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
@@ -103,11 +151,11 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Inf
case IR::OP_F80CVT: {
switch (OpSize) {
case 4: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4);
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4);
return true;
}
case 8: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8);
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8);
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
@@ -119,15 +167,30 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Inf
switch (OpSize) {
case 2: {
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2);
if (Op->Truncate) {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2);
}
else {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2);
}
return true;
}
case 4: {
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4);
if (Op->Truncate) {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4);
}
else {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4);
}
return true;
}
case 8: {
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8);
if (Op->Truncate) {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8);
}
else {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8);
}
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
@@ -148,20 +211,20 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Inf
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
};
*Info = GetFallbackInfo(handlers[Op->Flags]);
*Info = GetFallbackInfo(handlers[Op->Flags], (Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP_0 + Op->Flags));
return true;
}
case IR::OP_F80CVTTOINT: {
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
switch (Op->Size) {
switch (Op->SrcSize) {
case 2: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2);
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2);
return true;
}
case 4: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4);
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4);
return true;
}
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
@@ -171,7 +234,7 @@ bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Inf
#define COMMON_X87_OP(OP) \
case IR::OP_F80##OP: { \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle); \
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP); \
return true; \
}
@@ -114,7 +114,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
// Branch ops
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
REGISTER_OP(GUESTRETURN, GuestReturn);
REGISTER_OP(SIGNALRETURN, SignalReturn);
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
REGISTER_OP(EXITFUNCTION, ExitFunction);
@@ -176,6 +175,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
REGISTER_OP(INVALIDATEFLAGS, NoOp);
REGISTER_OP(PROCESSORID, ProcessorID);
REGISTER_OP(RDRAND, RDRAND);
// Move ops
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
@@ -185,8 +185,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
// Vector ops
REGISTER_OP(VECTORZERO, VectorZero);
REGISTER_OP(VECTORIMM, VectorImm);
REGISTER_OP(CREATEVECTOR2, CreateVector2);
REGISTER_OP(CREATEVECTOR4, CreateVector4);
REGISTER_OP(SPLATVECTOR2, SplatVector);
REGISTER_OP(SPLATVECTOR4, SplatVector);
REGISTER_OP(VMOV, VMov);
@@ -275,6 +273,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
REGISTER_OP(VSMULL2, VSMull2);
REGISTER_OP(VUABDL, VUABDL);
REGISTER_OP(VTBL1, VTBL1);
REGISTER_OP(VREV64, VRev64);
// Encryption ops
REGISTER_OP(VAESIMC, AESImc);
@@ -1,6 +1,7 @@
#pragma once
#include <stdint.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IntrusiveIRList.h>
@@ -38,12 +39,14 @@ namespace FEXCore::CPU {
struct FallbackInfo {
FallbackABI ABI;
void *fn;
FEXCore::Core::FallbackHandlerIndex HandlerIndex;
};
class InterpreterOps {
public:
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
static void FillFallbackIndexPointers(uint64_t *Info);
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
struct IROpData {
@@ -139,7 +142,6 @@ namespace FEXCore::CPU {
///< Branch ops
DEF_OP(GuestCallDirect);
DEF_OP(GuestCallIndirect);
DEF_OP(GuestReturn);
DEF_OP(SignalReturn);
DEF_OP(CallbackReturn);
DEF_OP(ExitFunction);
@@ -194,6 +196,7 @@ namespace FEXCore::CPU {
DEF_OP(GetRoundingMode);
DEF_OP(SetRoundingMode);
DEF_OP(ProcessorID);
DEF_OP(RDRAND);
///< Move ops
DEF_OP(ExtractElementPair);
@@ -203,8 +206,6 @@ namespace FEXCore::CPU {
///< Vector ops
DEF_OP(VectorZero);
DEF_OP(VectorImm);
DEF_OP(CreateVector2);
DEF_OP(CreateVector4);
DEF_OP(SplatVector);
DEF_OP(VMov);
DEF_OP(VAnd);
@@ -290,6 +291,7 @@ namespace FEXCore::CPU {
DEF_OP(VSMull2);
DEF_OP(VUABDL);
DEF_OP(VTBL1);
DEF_OP(VRev64);
///< Encryption ops
DEF_OP(AESImc);
@@ -14,6 +14,7 @@ $end_info$
#ifdef _M_X86_64
#include <xmmintrin.h>
#endif
#include <sys/random.h>
namespace FEXCore::CPU {
[[noreturn]]
@@ -148,6 +149,14 @@ DEF_OP(ProcessorID) {
GD = (CPUNode << 12) | CPU;
}
DEF_OP(RDRAND) {
// We are ignoring Op->GetReseeded in the interpreter
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
ssize_t Result = ::getrandom(&DstPtr[0], 8, 0);
// Second result is if we managed to read a valid random number or not
DstPtr[1] = Result == 8 ? 1 : 0;
}
#undef DEF_OP
} // namespace FEXCore::CPU
@@ -26,8 +26,8 @@ DEF_OP(CreateElementPair) {
uint8_t *Dst = GetDest<uint8_t*>(Data->SSAData, Node);
memcpy(Dst, Src_Lower, Op->Header.Size);
memcpy(Dst + Op->Header.Size, Src_Upper, Op->Header.Size);
memcpy(Dst, Src_Lower, IROp->ElementSize);
memcpy(Dst + IROp->ElementSize, Src_Upper, IROp->ElementSize);
}
DEF_OP(Mov) {
@@ -7,6 +7,7 @@ $end_info$
#include "Interface/Core/Interpreter/InterpreterClass.h"
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include "Interface/Core/Interpreter/InterpreterDefines.h"
#include <FEXCore/Utils/BitUtils.h>
#include <bit>
#include <cstdint>
@@ -38,39 +39,6 @@ DEF_OP(VectorImm) {
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(CreateVector2) {
auto Op = IROp->C<IR::IROp_CreateVector2>();
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_A_FMT(OpSize <= 16, "Can't handle a vector of size: {}", OpSize);
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
uint8_t Tmp[16];
uint8_t ElementSize = OpSize / 2;
#define CREATE_VECTOR(elementsize, type) \
case elementsize: { \
auto *Dst_d = reinterpret_cast<type*>(Tmp); \
auto *Src1_d = reinterpret_cast<type*>(Src1); \
auto *Src2_d = reinterpret_cast<type*>(Src2); \
Dst_d[0] = *Src1_d; \
Dst_d[1] = *Src2_d; \
break; \
}
switch (ElementSize) {
CREATE_VECTOR(1, uint8_t)
CREATE_VECTOR(2, uint16_t)
CREATE_VECTOR(4, uint32_t)
CREATE_VECTOR(8, uint64_t)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize); break;
}
#undef CREATE_VECTOR
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(CreateVector4) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(SplatVector) {
auto Op = IROp->C<IR::IROp_SplatVector2>();
uint8_t OpSize = IROp->Size;
@@ -1930,6 +1898,39 @@ DEF_OP(VTBL1) {
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VRev64) {
auto Op = IROp->C<IR::IROp_VRev64>();
uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
uint8_t Tmp[16];
uint8_t Elements = OpSize / 8;
// The element working size is always 64-bit
// The defined element size in the op is the operating size of the element swapping
auto Func8 = [](auto a) { return BSwap64(a); };
auto Func16 = [](auto a) {
return (a >> 48) | // Element[3] -> Element[0]
((a >> 16) & 0xFFFF'0000U) | // Element[2] -> Element[1]
((a << 16) & 0xFFFF'0000'0000ULL) | // Element[1] -> Element[2]
(a << 48); // Element[0] -> Element[3]
};
auto Func32 = [](auto a) {
return (a >> 32) | (a << 32);
};
switch (Op->Header.ElementSize) {
DO_VECTOR_1SRC_OP(1, uint64_t, Func8)
DO_VECTOR_1SRC_OP(2, uint64_t, Func16)
DO_VECTOR_1SRC_OP(4, uint64_t, Func32)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
}
memcpy(GDP, Tmp, Op->Header.Size);
}
#undef DEF_OP
} // namespace FEXCore::CPU
+11 -34
View File
@@ -12,37 +12,13 @@ namespace FEXCore::CPU {
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
__uint128_t Res = Source / Divisor;
return Res;
}
static int64_t LDIV(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
__int128_t Res = Source / Divisor;
return Res;
}
static uint64_t LUREM(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
__uint128_t Res = Source % Divisor;
return Res;
}
static int64_t LREM(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
__int128_t Res = Source % Divisor;
return Res;
}
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
DEF_OP(TruncElementPair) {
auto Op = IROp->C<IR::IROp_TruncElementPair>();
switch (Op->Size) {
switch (IROp->Size) {
case 4: {
auto Dst = GetSrcPair<RA_32>(Node);
auto Src = GetSrcPair<RA_32>(Op->Header.Args[0].ID());
@@ -50,7 +26,7 @@ DEF_OP(TruncElementPair) {
mov(Dst.second, Src.second);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", Op->Size); break;
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", IROp->Size); break;
}
}
@@ -666,7 +642,8 @@ DEF_OP(LDiv) {
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
mov(x2, GetReg<RA_64>(Op->Header.Args[2].ID()));
LoadConstant(x3, reinterpret_cast<uint64_t>(LDIV));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIV)));
SpillStaticRegs();
blr(x3);
FillStaticRegs();
@@ -709,7 +686,7 @@ DEF_OP(LUDiv) {
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
mov(x2, GetReg<RA_64>(Op->Header.Args[2].ID()));
LoadConstant(x3, reinterpret_cast<uint64_t>(LUDIV));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIV)));
SpillStaticRegs();
blr(x3);
FillStaticRegs();
@@ -762,7 +739,7 @@ DEF_OP(LRem) {
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
mov(x2, GetReg<RA_64>(Op->Header.Args[2].ID()));
LoadConstant(x3, reinterpret_cast<uint64_t>(LREM));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREM)));
SpillStaticRegs();
blr(x3);
FillStaticRegs();
@@ -812,7 +789,7 @@ DEF_OP(LURem) {
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
mov(x2, GetReg<RA_64>(Op->Header.Args[2].ID()));
LoadConstant(x3, reinterpret_cast<uint64_t>(LUREM));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREM)));
SpillStaticRegs();
blr(x3);
FillStaticRegs();
@@ -1093,16 +1070,16 @@ DEF_OP(VExtractToGPR) {
uint8_t OpSize = IROp->Size;
switch (OpSize) {
case 1:
umov(GetReg<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()).V16B(), Op->Idx);
umov(GetReg<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()).V16B(), Op->Index);
break;
case 2:
umov(GetReg<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()).V8H(), Op->Idx);
umov(GetReg<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()).V8H(), Op->Index);
break;
case 4:
umov(GetReg<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()).V4S(), Op->Idx);
umov(GetReg<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()).V4S(), Op->Index);
break;
case 8:
umov(GetReg<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()).V2D(), Op->Idx);
umov(GetReg<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()).V2D(), Op->Index);
break;
default: LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
}
@@ -12,18 +12,17 @@ using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CASPair>();
uint8_t OpSize = IROp->Size;
// Size is the size of each pair element
auto Dst = GetSrcPair<RA_64>(Node);
auto Expected = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
auto Desired = GetSrcPair<RA_64>(Op->Header.Args[1].ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
auto Expected = GetSrcPair<RA_64>(Op->Expected.ID());
auto Desired = GetSrcPair<RA_64>(Op->Desired.ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
mov(TMP3, Expected.first);
mov(TMP4, Expected.second);
switch (OpSize) {
switch (IROp->ElementSize) {
case 4:
caspal(TMP3.W(), TMP4.W(), Desired.first.W(), Desired.second.W(), MemOperand(MemSrc));
mov(Dst.first.W(), TMP3.W());
@@ -34,11 +33,11 @@ DEF_OP(CASPair) {
mov(Dst.first, TMP3);
mov(Dst.second, TMP4);
break;
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
}
}
else {
switch (OpSize) {
switch (IROp->ElementSize) {
case 4: {
aarch64::Label LoopTop;
aarch64::Label LoopNotExpected;
@@ -91,7 +90,7 @@ DEF_OP(CASPair) {
bind(&LoopExpected);
break;
}
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
}
}
}
@@ -99,16 +98,13 @@ DEF_OP(CASPair) {
DEF_OP(CAS) {
auto Op = IROp->C<IR::IROp_CAS>();
uint8_t OpSize = IROp->Size;
// Args[0]: Expected
// Args[1]: Desired
// Args[2]: Pointer
// DataSrc = *Src1
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
// This will write to memory! Careful!
auto Expected = GetReg<RA_64>(Op->Header.Args[0].ID());
auto Desired = GetReg<RA_64>(Op->Header.Args[1].ID());
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
auto Expected = GetReg<RA_64>(Op->Expected.ID());
auto Desired = GetReg<RA_64>(Op->Desired.ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
mov(TMP2, Expected);
@@ -216,14 +212,14 @@ DEF_OP(CAS) {
DEF_OP(AtomicAdd) {
auto Op = IROp->C<IR::IROp_AtomicAdd>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: staddlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 2: staddlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 4: staddl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 8: staddl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 1: staddlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 2: staddlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 4: staddl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 8: staddl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
@@ -234,7 +230,7 @@ DEF_OP(AtomicAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -243,7 +239,7 @@ DEF_OP(AtomicAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -252,7 +248,7 @@ DEF_OP(AtomicAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
add(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -261,7 +257,7 @@ DEF_OP(AtomicAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
add(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
add(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
@@ -274,10 +270,10 @@ DEF_OP(AtomicAdd) {
DEF_OP(AtomicSub) {
auto Op = IROp->C<IR::IROp_AtomicSub>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
neg(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
case 1: staddlb(TMP2.W(), MemOperand(MemSrc)); break;
case 2: staddlh(TMP2.W(), MemOperand(MemSrc)); break;
@@ -293,7 +289,7 @@ DEF_OP(AtomicSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -302,7 +298,7 @@ DEF_OP(AtomicSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -311,7 +307,7 @@ DEF_OP(AtomicSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
sub(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -320,7 +316,7 @@ DEF_OP(AtomicSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
sub(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
sub(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
@@ -333,10 +329,10 @@ DEF_OP(AtomicSub) {
DEF_OP(AtomicAnd) {
auto Op = IROp->C<IR::IROp_AtomicAnd>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
mvn(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
case 1: stclrlb(TMP2.W(), MemOperand(MemSrc)); break;
case 2: stclrlh(TMP2.W(), MemOperand(MemSrc)); break;
@@ -352,7 +348,7 @@ DEF_OP(AtomicAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -361,7 +357,7 @@ DEF_OP(AtomicAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -370,7 +366,7 @@ DEF_OP(AtomicAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
and_(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -379,7 +375,7 @@ DEF_OP(AtomicAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
and_(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
and_(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
@@ -392,14 +388,14 @@ DEF_OP(AtomicAnd) {
DEF_OP(AtomicOr) {
auto Op = IROp->C<IR::IROp_AtomicOr>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: stsetlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 2: stsetlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 4: stsetl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 8: stsetl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 1: stsetlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 2: stsetlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 4: stsetl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 8: stsetl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
@@ -410,7 +406,7 @@ DEF_OP(AtomicOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -419,7 +415,7 @@ DEF_OP(AtomicOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -428,7 +424,7 @@ DEF_OP(AtomicOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
orr(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -437,7 +433,7 @@ DEF_OP(AtomicOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
orr(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
orr(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
@@ -450,14 +446,14 @@ DEF_OP(AtomicOr) {
DEF_OP(AtomicXor) {
auto Op = IROp->C<IR::IROp_AtomicXor>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: steorlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 2: steorlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 4: steorl(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 8: steorl(GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
case 1: steorlb(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 2: steorlh(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 4: steorl(GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc)); break;
case 8: steorl(GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
@@ -468,7 +464,7 @@ DEF_OP(AtomicXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -477,7 +473,7 @@ DEF_OP(AtomicXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -486,7 +482,7 @@ DEF_OP(AtomicXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
eor(TMP2.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP2.W(), TMP2.W(), MemOperand(MemSrc));
cbnz(TMP2.W(), &LoopTop);
break;
@@ -495,7 +491,7 @@ DEF_OP(AtomicXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
eor(TMP2, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
eor(TMP2, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP2, TMP2, MemOperand(MemSrc));
cbnz(TMP2, &LoopTop);
break;
@@ -508,10 +504,10 @@ DEF_OP(AtomicXor) {
DEF_OP(AtomicSwap) {
auto Op = IROp->C<IR::IROp_AtomicSwap>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
mov(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
mov(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
case 1: swplb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: swplh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
@@ -527,7 +523,7 @@ DEF_OP(AtomicSwap) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
stlxrb(TMP4.W(), GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
stlxrb(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
uxtb(GetReg<RA_32>(Node), TMP2.W());
break;
@@ -536,7 +532,7 @@ DEF_OP(AtomicSwap) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
stlxrh(TMP4.W(), GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
stlxrh(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
uxtw(GetReg<RA_32>(Node), TMP2.W());
break;
@@ -545,7 +541,7 @@ DEF_OP(AtomicSwap) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
stlxr(TMP4.W(), GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
stlxr(TMP4.W(), GetReg<RA_32>(Op->Value.ID()), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
break;
@@ -554,7 +550,7 @@ DEF_OP(AtomicSwap) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
stlxr(TMP4, GetReg<RA_64>(Op->Header.Args[1].ID()), MemOperand(MemSrc));
stlxr(TMP4, GetReg<RA_64>(Op->Value.ID()), MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2.X());
break;
@@ -566,14 +562,14 @@ DEF_OP(AtomicSwap) {
DEF_OP(AtomicFetchAdd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: ldaddalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldaddalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldaddal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldaddal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
case 1: ldaddalb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldaddalh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldaddal(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldaddal(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
@@ -584,7 +580,7 @@ DEF_OP(AtomicFetchAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -594,7 +590,7 @@ DEF_OP(AtomicFetchAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -604,7 +600,7 @@ DEF_OP(AtomicFetchAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
add(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -614,7 +610,7 @@ DEF_OP(AtomicFetchAdd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
add(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
add(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
@@ -627,10 +623,10 @@ DEF_OP(AtomicFetchAdd) {
DEF_OP(AtomicFetchSub) {
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
neg(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
case 1: ldaddalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldaddalh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
@@ -646,7 +642,7 @@ DEF_OP(AtomicFetchSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -656,7 +652,7 @@ DEF_OP(AtomicFetchSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -666,7 +662,7 @@ DEF_OP(AtomicFetchSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
sub(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -676,7 +672,7 @@ DEF_OP(AtomicFetchSub) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
sub(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
sub(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
@@ -689,10 +685,10 @@ DEF_OP(AtomicFetchSub) {
DEF_OP(AtomicFetchAnd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
mvn(TMP2, GetReg<RA_64>(Op->Value.ID()));
switch (IROp->Size) {
case 1: ldclralb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldclralh(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
@@ -708,7 +704,7 @@ DEF_OP(AtomicFetchAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -718,7 +714,7 @@ DEF_OP(AtomicFetchAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -728,7 +724,7 @@ DEF_OP(AtomicFetchAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
and_(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -738,7 +734,7 @@ DEF_OP(AtomicFetchAnd) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
and_(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
and_(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
@@ -751,14 +747,14 @@ DEF_OP(AtomicFetchAnd) {
DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: ldsetalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldsetalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldsetal(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldsetal(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
case 1: ldsetalb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldsetalh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldsetal(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldsetal(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
@@ -769,7 +765,7 @@ DEF_OP(AtomicFetchOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -779,7 +775,7 @@ DEF_OP(AtomicFetchOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -789,7 +785,7 @@ DEF_OP(AtomicFetchOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
orr(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -799,7 +795,7 @@ DEF_OP(AtomicFetchOr) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
orr(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
orr(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
@@ -812,14 +808,14 @@ DEF_OP(AtomicFetchOr) {
DEF_OP(AtomicFetchXor) {
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
if (CTX->HostFeatures.SupportsAtomics) {
switch (IROp->Size) {
case 1: ldeoralb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldeoralh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldeoral(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldeoral(GetReg<RA_64>(Op->Header.Args[1].ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
case 1: ldeoralb(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 2: ldeoralh(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 4: ldeoral(GetReg<RA_32>(Op->Value.ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
case 8: ldeoral(GetReg<RA_64>(Op->Value.ID()), GetReg<RA_64>(Node), MemOperand(MemSrc)); break;
default: LOGMAN_MSG_A_FMT("Unhandled Atomic size: {}", IROp->Size);
}
}
@@ -830,7 +826,7 @@ DEF_OP(AtomicFetchXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrb(TMP2.W(), MemOperand(MemSrc));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -840,7 +836,7 @@ DEF_OP(AtomicFetchXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxrh(TMP2.W(), MemOperand(MemSrc));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxrh(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -850,7 +846,7 @@ DEF_OP(AtomicFetchXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2.W(), MemOperand(MemSrc));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Header.Args[1].ID()));
eor(TMP3.W(), TMP2.W(), GetReg<RA_32>(Op->Value.ID()));
stlxr(TMP4.W(), TMP3.W(), MemOperand(MemSrc));
cbnz(TMP4.W(), &LoopTop);
mov(GetReg<RA_32>(Node), TMP2.W());
@@ -860,7 +856,7 @@ DEF_OP(AtomicFetchXor) {
aarch64::Label LoopTop;
bind(&LoopTop);
ldaxr(TMP2, MemOperand(MemSrc));
eor(TMP3, TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
eor(TMP3, TMP2, GetReg<RA_64>(Op->Value.ID()));
stlxr(TMP4, TMP3, MemOperand(MemSrc));
cbnz(TMP4, &LoopTop);
mov(GetReg<RA_64>(Node), TMP2);
@@ -873,7 +869,7 @@ DEF_OP(AtomicFetchXor) {
DEF_OP(AtomicFetchNeg) {
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemSrc = GetReg<RA_64>(Op->Addr.ID());
// TMP2-TMP3
switch (IROp->Size) {
@@ -4,6 +4,7 @@ tags: backend|arm64
$end_info$
*/
#include "FEXCore/IR/IR.h"
#include "Interface/Core/LookupCache.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
@@ -26,17 +27,13 @@ DEF_OP(GuestCallIndirect) {
LogMan::Msg::DFmt("Unimplemented");
}
DEF_OP(GuestReturn) {
LogMan::Msg::DFmt("Unimplemented");
}
DEF_OP(SignalReturn) {
// First we must reset the stack
ResetStack();
// Now branch to our signal return helper
// This can't be a direct branch since the code needs to live at a constant location
LoadConstant(x0, ThreadSharedData.SignalReturnInstruction);
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SignalReturnHandler)));
br(x0);
}
@@ -49,7 +46,7 @@ DEF_OP(CallbackReturn) {
ResetStack();
// We can now lower the ref counter again
LoadConstant(x0, reinterpret_cast<uint64_t>(ThreadSharedData.SignalHandlerRefCounterPtr));
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SignalHandlerRefCountPointer)));
ldr(w2, MemOperand(x0));
sub(w2, w2, 1);
str(w2, MemOperand(x0));
@@ -88,7 +85,7 @@ DEF_OP(ExitFunction) {
RipReg = GetReg<RA_64>(Op->Header.Args[0].ID());
// L1 Cache
LoadConstant(x0, ThreadState->LookupCache->GetL1Pointer());
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.L1Pointer)));
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x3, Shift::LSL, 4));
@@ -99,7 +96,7 @@ DEF_OP(ExitFunction) {
br(x1);
bind(&FullLookup);
LoadConstant(TMP1, ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress);
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.DispatcherLoopTop)));
str(RipReg, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
br(TMP1);
}
@@ -184,8 +181,16 @@ DEF_OP(Syscall) {
// X1: ThreadState
// X2: Pointer to SyscallArguments
FEXCore::IR::SyscallFlags Flags = Op->Flags;
PushDynamicRegsAndLR();
SpillStaticRegs();
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
SpillStaticRegs();
}
else {
// Need to spill all caller saved registers still
SpillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
}
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
sub(sp, sp, SPOffset);
@@ -194,22 +199,30 @@ DEF_OP(Syscall) {
str(GetReg<RA_64>(Op->Header.Args[i].ID()), MemOperand(sp, i * 8));
}
LoadConstant(x0, reinterpret_cast<uint64_t>(CTX->SyscallHandler));
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SyscallHandlerObj)));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SyscallHandlerFunc)));
mov(x1, STATE);
mov(x2, sp);
LoadConstant(x3, reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall));
blr(x3);
add(sp, sp, SPOffset);
// Result is now in x0
// Fix the stack and any values that were stepped on
FillStaticRegs();
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) != FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY &&
(Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
FillStaticRegs();
}
else {
// Result is now in x0
// Fix the stack and any values that were stepped on
FillStaticRegs(true, CALLER_GPR_MASK, CALLER_FPR_MASK);
}
PopDynamicRegsAndLR();
// Move result to its destination register
mov(GetReg<RA_64>(Node), x0);
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
// Move result to its destination register
mov(GetReg<RA_64>(Node), x0);
}
}
DEF_OP(InlineSyscall) {
@@ -340,21 +353,23 @@ DEF_OP(InlineSyscall) {
svc(0);
// On updated signal mask we can receive a signal RIGHT HERE
// Now that we are done in the syscall we need to carefully peel back the state
// First unspill the registers from before
FillStaticRegs(false, SpillMask);
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
// Now that we are done in the syscall we need to carefully peel back the state
// First unspill the registers from before
FillStaticRegs(false, SpillMask);
// Now the registers we've spilled are back in their original host registers
// We can safely claim we are no longer in a syscall
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
// Now the registers we've spilled are back in their original host registers
// We can safely claim we are no longer in a syscall
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
// Result is now in x0
// Move result to its destination register
if (CTX->Config.Is64BitMode()) {
mov(GetReg<RA_64>(Node), x0);
}
else {
uxtw(GetReg<RA_64>(Node), x0);
// Result is now in x0
// Move result to its destination register
if (CTX->Config.Is64BitMode()) {
mov(GetReg<RA_64>(Node), x0);
}
else {
uxtw(GetReg<RA_64>(Node), x0);
}
}
}
@@ -437,7 +452,7 @@ DEF_OP(RemoveCodeEntry) {
mov(x0, STATE);
LoadConstant(x1, Entry);
LoadConstant(x2, reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit));
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.RemoveCodeEntryFromJIT)));
SpillStaticRegs();
blr(x2);
FillStaticRegs();
@@ -454,19 +469,10 @@ DEF_OP(CPUID) {
// x0 = CPUID Handler
// x1 = CPUID Function
// x2 = CPUID Leaf
LoadConstant(x0, reinterpret_cast<uint64_t>(&CTX->CPUID));
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.CPUIDObj)));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.CPUIDFunction)));
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
mov(x2, GetReg<RA_64>(Op->Header.Args[1].ID()));
using ClassPtrType = FEXCore::CPUID::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t, uint32_t);
union PtrCast {
ClassPtrType ClassPtr;
uintptr_t Data;
};
PtrCast Ptr;
Ptr.ClassPtr = &FEXCore::CPUIDEmu::RunFunction;
LoadConstant(x3, Ptr.Data);
SpillStaticRegs();
blr(x3);
FillStaticRegs();
@@ -485,7 +491,6 @@ void Arm64JITCore::RegisterBranchHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
REGISTER_OP(GUESTRETURN, GuestReturn);
REGISTER_OP(SIGNALRETURN, SignalReturn);
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
REGISTER_OP(EXITFUNCTION, ExitFunction);
@@ -16,19 +16,19 @@ DEF_OP(VInsGPR) {
mov(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
switch (Op->Header.ElementSize) {
case 1: {
ins(GetDst(Node).V16B(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Header.Args[1].ID()));
break;
}
case 2: {
ins(GetDst(Node).V8H(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Header.Args[1].ID()));
break;
}
case 4: {
ins(GetDst(Node).V4S(), Op->Index, GetReg<RA_32>(Op->Header.Args[1].ID()));
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Header.Args[1].ID()));
break;
}
case 8: {
ins(GetDst(Node).V2D(), Op->Index, GetReg<RA_64>(Op->Header.Args[1].ID()));
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Header.Args[1].ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
+79 -25
View File
@@ -21,10 +21,13 @@ $end_info$
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Utils/MemberFunctionToPointer.h"
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/Core/UContext.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include "Interface/Core/Interpreter/InterpreterOps.h"
#include <sys/mman.h>
@@ -32,6 +35,40 @@ $end_info$
#include <unistd.h>
#include <string.h>
namespace {
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
__uint128_t Res = Source / Divisor;
return Res;
}
static int64_t LDIV(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
__int128_t Res = Source / Divisor;
return Res;
}
static uint64_t LUREM(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
__uint128_t Res = Source % Divisor;
return Res;
}
static int64_t LREM(int64_t SrcHigh, int64_t SrcLow, int64_t Divisor) {
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
__int128_t Res = Source % Divisor;
return Res;
}
static void PrintValue(uint64_t Value) {
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
}
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
}
}
namespace FEXCore::CPU {
void Arm64JITCore::CopyNecessaryDataForCompileThread(CPUBackend *Original) {
@@ -56,8 +93,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
LoadConstant(x1, (uintptr_t)Info.fn);
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x1);
PopDynamicRegsAndLR();
@@ -72,8 +108,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
fmov(v0.S(), GetSrc(IROp->Args[0].ID()).S()) ;
LoadConstant(x0, (uintptr_t)Info.fn);
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x0);
PopDynamicRegsAndLR();
@@ -92,8 +127,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushDynamicRegsAndLR();
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
LoadConstant(x0, (uintptr_t)Info.fn);
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x0);
PopDynamicRegsAndLR();
@@ -118,8 +152,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
else {
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
}
LoadConstant(x1, (uintptr_t)Info.fn);
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x1);
PopDynamicRegsAndLR();
@@ -140,8 +173,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
LoadConstant(x2, (uintptr_t)Info.fn);
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x2);
PopDynamicRegsAndLR();
@@ -160,8 +192,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
LoadConstant(x2, (uintptr_t)Info.fn);
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x2);
PopDynamicRegsAndLR();
@@ -180,8 +211,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
LoadConstant(x2, (uintptr_t)Info.fn);
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x2);
PopDynamicRegsAndLR();
@@ -199,8 +229,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
LoadConstant(x2, (uintptr_t)Info.fn);
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x2);
PopDynamicRegsAndLR();
@@ -218,8 +247,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
LoadConstant(x2, (uintptr_t)Info.fn);
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x2);
PopDynamicRegsAndLR();
@@ -240,8 +268,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
LoadConstant(x4, (uintptr_t)Info.fn);
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x4);
PopDynamicRegsAndLR();
@@ -259,8 +286,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
LoadConstant(x2, (uintptr_t)Info.fn);
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x2);
PopDynamicRegsAndLR();
@@ -283,8 +309,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
LoadConstant(x4, (uintptr_t)Info.fn);
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
blr(x4);
PopDynamicRegsAndLR();
@@ -336,6 +361,35 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
: Arm64Emitter(ctx, 0)
, CTX {ctx}
, ThreadState {Thread} {
{
// Set up pointers that the JIT needs to load
auto &Pointers = ThreadState->CurrentFrame->Pointers.AArch64;
// Process specific
Pointers.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
Pointers.LDIV = reinterpret_cast<uint64_t>(LDIV);
Pointers.LUREM = reinterpret_cast<uint64_t>(LUREM);
Pointers.LREM = reinterpret_cast<uint64_t>(LREM);
Pointers.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
Pointers.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
Pointers.RemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit);
Pointers.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
{
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
Pointers.CPUIDFunction = PMF.GetConvertedPointer();
}
Pointers.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
Pointers.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
// Fill in the fallback handlers
InterpreterOps::FillFallbackIndexPointers(Pointers.FallbackHandlerPointers);
// Thread Specific
Pointers.SignalHandlerRefCountPointer = reinterpret_cast<uint64_t>(&Dispatcher->SignalHandlerRefCounter);
}
{
DispatcherConfig config;
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
@@ -673,7 +727,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
// Stop the thread
LoadConstant(x0, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddressSpillSRA);
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.ThreadPauseHandlerSpillSRA)));
br(x0);
}
bind(&RunBlock);
@@ -277,7 +277,6 @@ private:
///< Branch ops
DEF_OP(GuestCallDirect);
DEF_OP(GuestCallIndirect);
DEF_OP(GuestReturn);
DEF_OP(SignalReturn);
DEF_OP(CallbackReturn);
DEF_OP(ExitFunction);
@@ -336,6 +335,7 @@ private:
DEF_OP(GetRoundingMode);
DEF_OP(SetRoundingMode);
DEF_OP(ProcessorID);
DEF_OP(RDRAND);
///< Move ops
DEF_OP(ExtractElementPair);
@@ -345,8 +345,6 @@ private:
///< Vector ops
DEF_OP(VectorZero);
DEF_OP(VectorImm);
DEF_OP(CreateVector2);
DEF_OP(CreateVector4);
DEF_OP(SplatVector2);
DEF_OP(SplatVector4);
DEF_OP(VMov);
@@ -434,6 +432,7 @@ private:
DEF_OP(VSMull2);
DEF_OP(VUABDL);
DEF_OP(VTBL1);
DEF_OP(VRev64);
///< Encryption ops
DEF_OP(AESImc);
@@ -698,29 +698,29 @@ DEF_OP(LoadMemTSO) {
DEF_OP(StoreMem) {
auto Op = IROp->C<IR::IROp_StoreMem>();
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class == FEXCore::IR::GPRClass) {
switch (IROp->Size) {
case 1:
strb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
strb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
break;
case 2:
strh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
strh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
break;
case 4:
str(GetReg<RA_32>(Op->Header.Args[1].ID()), MemSrc);
str(GetReg<RA_32>(Op->Value.ID()), MemSrc);
break;
case 8:
str(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
str(GetReg<RA_64>(Op->Value.ID()), MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
}
}
else {
auto Src = GetSrc(Op->Header.Args[1].ID());
auto Src = GetSrc(Op->Value.ID());
switch (IROp->Size) {
case 1:
str(Src.B(), MemSrc);
@@ -744,7 +744,7 @@ DEF_OP(StoreMem) {
DEF_OP(StoreMemTSO) {
auto Op = IROp->C<IR::IROp_StoreMemTSO>();
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Addr.ID()));
if (!Op->Offset.IsInvalid()) {
LOGMAN_MSG_A_FMT("StoreMemTSO: No offset allowed");
@@ -753,19 +753,19 @@ DEF_OP(StoreMemTSO) {
if (Op->Class == FEXCore::IR::GPRClass) {
if (IROp->Size == 1) {
// 8bit load is always aligned to natural alignment
stlrb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
stlrb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
}
else {
nop();
switch (IROp->Size) {
case 2:
stlrh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
stlrh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
break;
case 4:
stlr(GetReg<RA_32>(Op->Header.Args[1].ID()), MemSrc);
stlr(GetReg<RA_32>(Op->Value.ID()), MemSrc);
break;
case 8:
stlr(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
stlr(GetReg<RA_64>(Op->Value.ID()), MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", IROp->Size);
}
@@ -774,7 +774,7 @@ DEF_OP(StoreMemTSO) {
}
else {
dmb(InnerShareable, BarrierAll);
auto Src = GetSrc(Op->Header.Args[1].ID());
auto Src = GetSrc(Op->Value.ID());
switch (IROp->Size) {
case 1:
str(Src.B(), MemSrc);
@@ -800,7 +800,7 @@ DEF_OP(StoreMemTSO) {
DEF_OP(ParanoidLoadMemTSO) {
auto Op = IROp->C<IR::IROp_LoadMemTSO>();
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Addr.ID()));
if (!Op->Offset.IsInvalid()) {
LOGMAN_MSG_A_FMT("ParanoidLoadMemTSO: No offset allowed");
@@ -857,7 +857,7 @@ DEF_OP(ParanoidLoadMemTSO) {
DEF_OP(ParanoidStoreMemTSO) {
auto Op = IROp->C<IR::IROp_StoreMemTSO>();
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Header.Args[0].ID()));
auto MemSrc = MemOperand(GetReg<RA_64>(Op->Addr.ID()));
if (!Op->Offset.IsInvalid()) {
LOGMAN_MSG_A_FMT("ParanoidStoreMemTSO: No offset allowed");
@@ -866,25 +866,25 @@ DEF_OP(ParanoidStoreMemTSO) {
if (Op->Class == FEXCore::IR::GPRClass) {
if (IROp->Size == 1) {
// 8bit load is always aligned to natural alignment
stlrb(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
stlrb(GetReg<RA_64>(Op->Value.ID()), MemSrc);
}
else {
switch (IROp->Size) {
case 2:
stlrh(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
stlrh(GetReg<RA_64>(Op->Value.ID()), MemSrc);
break;
case 4:
stlr(GetReg<RA_32>(Op->Header.Args[1].ID()), MemSrc);
stlr(GetReg<RA_32>(Op->Value.ID()), MemSrc);
break;
case 8:
stlr(GetReg<RA_64>(Op->Header.Args[1].ID()), MemSrc);
stlr(GetReg<RA_64>(Op->Value.ID()), MemSrc);
break;
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", IROp->Size);
}
}
}
else {
auto Src = GetSrc(Op->Header.Args[1].ID());
auto Src = GetSrc(Op->Value.ID());
if (IROp->Size == 1) {
// 8bit load is always aligned to natural alignment
mov(TMP1.W(), Src.V16B(), 0);
+26 -15
View File
@@ -7,14 +7,6 @@ $end_info$
#include "Interface/Core/JIT/Arm64/JITClass.h"
namespace FEXCore::CPU {
static void PrintValue(uint64_t Value) {
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
}
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
}
using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
@@ -44,7 +36,7 @@ DEF_OP(Break) {
break;
case FEXCore::IR::Break_Overflow: // overflow
ResetStack();
LoadConstant(TMP1, ThreadSharedData.Dispatcher->OverflowExceptionInstructionAddress);
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.OverflowExceptionHandler)));
br(TMP1);
break;
case FEXCore::IR::Break_Halt: { // HLT
@@ -54,14 +46,13 @@ DEF_OP(Break) {
add(sp, TMP1, 0);
// Now we need to jump to the thread stop handler
LoadConstant(TMP1, ThreadSharedData.Dispatcher->ThreadStopHandlerAddressSpillSRA);
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.ThreadStopHandlerSpillSRA)));
br(TMP1);
break;
}
case FEXCore::IR::Break_Interrupt3: { // INT3
ResetStack();
LoadConstant(TMP1, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddressSpillSRA);
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.ThreadPauseHandlerSpillSRA)));
br(TMP1);
break;
}
@@ -69,7 +60,7 @@ DEF_OP(Break) {
{
ResetStack();
LoadConstant(TMP1, ThreadSharedData.Dispatcher->UnimplementedInstructionAddress);
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.UnimplementedInstructionHandler)));
br(TMP1);
break;
@@ -143,13 +134,13 @@ DEF_OP(Print) {
if (IsGPR(Op->Header.Args[0].ID())) {
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
LoadConstant(x3, reinterpret_cast<uint64_t>(PrintValue));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.PrintValue)));
}
else {
fmov(x0, GetSrc(Op->Header.Args[0].ID()).V1D());
// Bug in vixl that source vector needs to b V1D rather than V2D?
fmov(x1, GetSrc(Op->Header.Args[0].ID()).V1D(), 1);
LoadConstant(x3, reinterpret_cast<uint64_t>(PrintVectorValue));
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.PrintVectorValue)));
}
SpillStaticRegs();
blr(x3);
@@ -210,6 +201,24 @@ DEF_OP(ProcessorID) {
orr(GetReg<RA_64>(Node), x0, Operand(x1, LSL, 12));
}
DEF_OP(RDRAND) {
auto Op = IROp->C<IR::IROp_RDRAND>();
// Results are in x0, x1
// Results want to be in a i64v2 vector
auto Dst = GetSrcPair<RA_64>(Node);
if (Op->GetReseeded) {
mrs(Dst.first, RNDRRS);
}
else {
mrs(Dst.first, RNDR);
}
// If the rng number is valid then NZCV is 0b0000, otherwise NZCV is 0b0100
cset(Dst.second, Condition::ne);
}
#undef DEF_OP
void Arm64JITCore::RegisterMiscHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
@@ -227,6 +236,8 @@ void Arm64JITCore::RegisterMiscHandlers() {
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
REGISTER_OP(INVALIDATEFLAGS, NoOp);
REGISTER_OP(PROCESSORID, ProcessorID);
REGISTER_OP(RDRAND, RDRAND);
#undef REGISTER_OP
}
}
@@ -37,7 +37,7 @@ DEF_OP(CreateElementPair) {
aarch64::Register RegSecond;
aarch64::Register RegTmp;
switch (Op->Header.Size) {
switch (IROp->ElementSize) {
case 4: {
Dst = GetSrcPair<RA_32>(Node);
RegFirst = GetReg<RA_32>(Op->Header.Args[0].ID());
@@ -42,14 +42,6 @@ DEF_OP(VectorImm) {
}
}
DEF_OP(CreateVector2) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(CreateVector4) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(SplatVector2) {
auto Op = IROp->C<IR::IROp_SplatVector2>();
uint8_t OpSize = IROp->Size;
@@ -1758,8 +1750,7 @@ DEF_OP(VInsScalarElement) {
DEF_OP(VExtractElement) {
auto Op = IROp->C<IR::IROp_VExtractElement>();
uint8_t OpSize = IROp->Size;
switch (OpSize) {
switch (Op->Header.Size) {
case 1:
mov(GetDst(Node).B(), GetSrc(Op->Header.Args[0].ID()).V16B(), Op->Index);
break;
@@ -1772,7 +1763,7 @@ DEF_OP(VExtractElement) {
case 8:
mov(GetDst(Node).D(), GetSrc(Op->Header.Args[0].ID()).V2D(), Op->Index);
break;
default: LOGMAN_MSG_A_FMT("Unhandled VExtractElement element size: {}", OpSize);
default: LOGMAN_MSG_A_FMT("Unhandled VExtractElement element size: {}", Op->Header.Size);
}
}
@@ -2318,13 +2309,27 @@ DEF_OP(VTBL1) {
}
}
DEF_OP(VRev64) {
auto Op = IROp->C<IR::IROp_VRev64>();
uint8_t OpSize = IROp->Size;
uint8_t Elements = OpSize / Op->Header.ElementSize;
// Vector
switch (Op->Header.ElementSize) {
case 1:
case 2:
case 4:
rev64(GetDst(Node).VCast(OpSize * 8, Elements), GetSrc(Op->Header.Args[0].ID()).VCast(OpSize * 8, Elements));
break;
case 8:
default: LOGMAN_MSG_A_FMT("Invalid Element Size: {}", Op->Header.ElementSize); break;
}
}
#undef DEF_OP
void Arm64JITCore::RegisterVectorHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
REGISTER_OP(VECTORZERO, VectorZero);
REGISTER_OP(VECTORIMM, VectorImm);
REGISTER_OP(CREATEVECTOR2, CreateVector2);
REGISTER_OP(CREATEVECTOR4, CreateVector4);
REGISTER_OP(SPLATVECTOR2, SplatVector2);
REGISTER_OP(SPLATVECTOR4, SplatVector4);
REGISTER_OP(VMOV, VMov);
@@ -2413,6 +2418,7 @@ void Arm64JITCore::RegisterVectorHandlers() {
REGISTER_OP(VSMULL2, VSMull2);
REGISTER_OP(VUABDL, VUABDL);
REGISTER_OP(VTBL1, VTBL1);
REGISTER_OP(VREV64, VRev64);
#undef REGISTER_OP
}
}
@@ -24,7 +24,7 @@ namespace FEXCore::CPU {
DEF_OP(TruncElementPair) {
auto Op = IROp->C<IR::IROp_TruncElementPair>();
switch (Op->Size) {
switch (IROp->Size) {
case 4: {
auto Dst = GetSrcPair<RA_32>(Node);
auto Src = GetSrcPair<RA_32>(Op->Header.Args[0].ID());
@@ -32,7 +32,7 @@ DEF_OP(TruncElementPair) {
mov(Dst.second, Src.second);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", Op->Size); break;
default: LOGMAN_MSG_A_FMT("Unhandled Truncation size: {}", IROp->Size); break;
}
}
@@ -1150,19 +1150,19 @@ DEF_OP(VExtractToGPR) {
switch (Op->Header.ElementSize) {
case 1: {
pextrb(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx);
pextrb(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()), Op->Index);
break;
}
case 2: {
pextrw(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx);
pextrw(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()), Op->Index);
break;
}
case 4: {
pextrd(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx);
pextrd(GetDst<RA_32>(Node), GetSrc(Op->Header.Args[0].ID()), Op->Index);
break;
}
case 8: {
pextrq(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx);
pextrq(GetDst<RA_64>(Node), GetSrc(Op->Header.Args[0].ID()), Op->Index);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
@@ -18,20 +18,16 @@ namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
DEF_OP(CASPair) {
auto Op = IROp->C<IR::IROp_CAS>();
uint8_t OpSize = IROp->Size;
// Args[0]: Desired
// Args[1]: Expected
// Args[2]: Pointer
// DataSrc = *Src1
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
// This will write to memory! Careful!
// Third operand must be a calculated guest memory address
//OrderedNode *CASResult = _CAS(Src3, Src2, Src1);
auto Dst = GetSrcPair<RA_64>(Node);
auto Expected = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
auto Desired = GetSrcPair<RA_64>(Op->Header.Args[1].ID());
auto MemSrc = GetSrc<RA_64>(Op->Header.Args[2].ID());
auto Expected = GetSrcPair<RA_64>(Op->Expected.ID());
auto Desired = GetSrcPair<RA_64>(Op->Desired.ID());
auto MemSrc = GetSrc<RA_64>(Op->Addr.ID());
Xbyak::Reg MemReg = MemSrc;
@@ -47,7 +43,7 @@ DEF_OP(CASPair) {
lock();
switch (OpSize) {
switch (IROp->ElementSize) {
case 4: {
cmpxchg8b(dword [MemReg]);
// EDX:EAX now contains the result
@@ -62,7 +58,7 @@ DEF_OP(CASPair) {
mov(Dst.second, rdx);
break;
}
default: LOGMAN_MSG_A_FMT("Unsupported: {}", OpSize);
default: LOGMAN_MSG_A_FMT("Unsupported: {}", IROp->ElementSize);
}
}
@@ -70,18 +66,15 @@ DEF_OP(CAS) {
auto Op = IROp->C<IR::IROp_CAS>();
uint8_t OpSize = IROp->Size;
// Args[0]: Desired
// Args[1]: Expected
// Args[2]: Pointer
// DataSrc = *Src1
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
// This will write to memory! Careful!
// Third operand must be a calculated guest memory address
//OrderedNode *CASResult = _CAS(Src3, Src2, Src1);
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[2].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
mov(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
mov(rax, GetSrc<RA_64>(Op->Expected.ID()));
// RCX now contains pointer
// RAX contains our expected value
@@ -90,23 +83,23 @@ DEF_OP(CAS) {
lock();
switch (OpSize) {
case 1: {
cmpxchg(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
cmpxchg(byte [MemReg], GetSrc<RA_8>(Op->Desired.ID()));
movzx(GetDst<RA_64>(Node), al);
break;
}
case 2: {
cmpxchg(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
cmpxchg(word [MemReg], GetSrc<RA_16>(Op->Desired.ID()));
movzx(GetDst<RA_64>(Node), ax);
break;
}
case 4: {
cmpxchg(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
cmpxchg(dword [MemReg], GetSrc<RA_32>(Op->Desired.ID()));
// RAX now contains the result
mov (GetDst<RA_64>(Node), eax);
break;
}
case 8: {
cmpxchg(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
cmpxchg(qword [MemReg], GetSrc<RA_64>(Op->Desired.ID()));
// RAX now contains the result
mov (GetDst<RA_64>(Node), rax);
break;
@@ -118,21 +111,21 @@ DEF_OP(CAS) {
DEF_OP(AtomicAdd) {
auto Op = IROp->C<IR::IROp_AtomicAdd>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
lock();
switch (IROp->Size) {
case 1:
add(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
add(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
break;
case 2:
add(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
add(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
break;
case 4:
add(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
add(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
break;
case 8:
add(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
add(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
}
@@ -141,20 +134,20 @@ DEF_OP(AtomicAdd) {
DEF_OP(AtomicSub) {
auto Op = IROp->C<IR::IROp_AtomicSub>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
lock();
switch (IROp->Size) {
case 1:
sub(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
sub(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
break;
case 2:
sub(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
sub(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
break;
case 4:
sub(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
sub(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
break;
case 8:
sub(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
sub(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
}
@@ -163,20 +156,20 @@ DEF_OP(AtomicSub) {
DEF_OP(AtomicAnd) {
auto Op = IROp->C<IR::IROp_AtomicAnd>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
lock();
switch (IROp->Size) {
case 1:
and_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
and_(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
break;
case 2:
and_(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
and_(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
break;
case 4:
and_(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
and_(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
break;
case 8:
and_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
and_(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
}
@@ -185,20 +178,20 @@ DEF_OP(AtomicAnd) {
DEF_OP(AtomicOr) {
auto Op = IROp->C<IR::IROp_AtomicOr>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
lock();
switch (IROp->Size) {
case 1:
or_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
or_(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
break;
case 2:
or_(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
or_(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
break;
case 4:
or_(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
or_(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
break;
case 8:
or_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
or_(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
}
@@ -207,20 +200,20 @@ DEF_OP(AtomicOr) {
DEF_OP(AtomicXor) {
auto Op = IROp->C<IR::IROp_AtomicXor>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
lock();
switch (IROp->Size) {
case 1:
xor_(byte [MemReg], GetSrc<RA_8>(Op->Header.Args[1].ID()));
xor_(byte [MemReg], GetSrc<RA_8>(Op->Value.ID()));
break;
case 2:
xor_(word [MemReg], GetSrc<RA_16>(Op->Header.Args[1].ID()));
xor_(word [MemReg], GetSrc<RA_16>(Op->Value.ID()));
break;
case 4:
xor_(dword [MemReg], GetSrc<RA_32>(Op->Header.Args[1].ID()));
xor_(dword [MemReg], GetSrc<RA_32>(Op->Value.ID()));
break;
case 8:
xor_(qword [MemReg], GetSrc<RA_64>(Op->Header.Args[1].ID()));
xor_(qword [MemReg], GetSrc<RA_64>(Op->Value.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled AtomicAdd size: {}", IROp->Size);
}
@@ -230,26 +223,26 @@ DEF_OP(AtomicSwap) {
auto Op = IROp->C<IR::IROp_AtomicSwap>();
Xbyak::Reg MemReg = rax;
mov(MemReg, GetSrc<RA_64>(Op->Header.Args[0].ID()));
mov(MemReg, GetSrc<RA_64>(Op->Addr.ID()));
switch (IROp->Size) {
case 1:
movzx(GetDst<RA_64>(Node), GetSrc<RA_8>(Op->Header.Args[1].ID()));
movzx(GetDst<RA_64>(Node), GetSrc<RA_8>(Op->Value.ID()));
lock();
xchg(byte [MemReg], GetDst<RA_8>(Node));
break;
case 2:
movzx(GetDst<RA_64>(Node), GetSrc<RA_16>(Op->Header.Args[1].ID()));
movzx(GetDst<RA_64>(Node), GetSrc<RA_16>(Op->Value.ID()));
lock();
xchg(word [MemReg], GetDst<RA_16>(Node));
break;
case 4:
mov(GetDst<RA_64>(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()));
mov(GetDst<RA_64>(Node), GetSrc<RA_32>(Op->Value.ID()));
lock();
xchg(dword [MemReg], GetDst<RA_32>(Node));
break;
case 8:
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Value.ID()));
lock();
xchg(qword [MemReg], GetDst<RA_64>(Node));
break;
@@ -260,28 +253,28 @@ DEF_OP(AtomicSwap) {
DEF_OP(AtomicFetchAdd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
switch (IROp->Size) {
case 1:
movzx(rcx, GetSrc<RA_8>(Op->Header.Args[1].ID()));
movzx(rcx, GetSrc<RA_8>(Op->Value.ID()));
lock();
xadd(byte [MemReg], cl);
movzx(GetDst<RA_32>(Node), cl);
break;
case 2:
movzx(rcx, GetSrc<RA_16>(Op->Header.Args[1].ID()));
movzx(rcx, GetSrc<RA_16>(Op->Value.ID()));
lock();
xadd(word [MemReg], cx);
movzx(GetDst<RA_32>(Node), cx);
break;
case 4:
mov(ecx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
mov(ecx, GetSrc<RA_32>(Op->Value.ID()));
lock();
xadd(dword [MemReg], ecx);
mov(GetDst<RA_64>(Node), ecx);
break;
case 8:
mov(rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov(rcx, GetSrc<RA_64>(Op->Value.ID()));
lock();
xadd(qword [MemReg], rcx);
mov(GetDst<RA_64>(Node), rcx);
@@ -293,31 +286,31 @@ DEF_OP(AtomicFetchAdd) {
DEF_OP(AtomicFetchSub) {
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
switch (IROp->Size) {
case 1:
mov(cl, GetSrc<RA_8>(Op->Header.Args[1].ID()));
mov(cl, GetSrc<RA_8>(Op->Value.ID()));
neg(cl);
lock();
xadd(byte [MemReg], cl);
movzx(GetDst<RA_32>(Node), cl);
break;
case 2:
mov(cx, GetSrc<RA_16>(Op->Header.Args[1].ID()));
mov(cx, GetSrc<RA_16>(Op->Value.ID()));
neg(cx);
lock();
xadd(word [MemReg], cx);
movzx(GetDst<RA_32>(Node), cx);
break;
case 4:
mov(ecx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
mov(ecx, GetSrc<RA_32>(Op->Value.ID()));
neg(ecx);
lock();
xadd(dword [MemReg], ecx);
mov(GetDst<RA_32>(Node), ecx);
break;
case 8:
mov(rcx, GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov(rcx, GetSrc<RA_64>(Op->Value.ID()));
neg(rcx);
lock();
xadd(qword [MemReg], rcx);
@@ -331,7 +324,7 @@ DEF_OP(AtomicFetchAnd) {
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
// TMP1 = rax
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
switch (IROp->Size) {
case 1: {
@@ -341,7 +334,7 @@ DEF_OP(AtomicFetchAnd) {
L(Loop);
mov(TMP2.cvt8(), TMP1.cvt8());
mov(TMP3.cvt8(), TMP1.cvt8());
and_(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
and_(TMP2.cvt8(), GetSrc<RA_8>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
@@ -357,7 +350,7 @@ DEF_OP(AtomicFetchAnd) {
L(Loop);
mov(TMP2.cvt16(), TMP1.cvt16());
mov(TMP3.cvt16(), TMP1.cvt16());
and_(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
and_(TMP2.cvt16(), GetSrc<RA_16>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
@@ -374,7 +367,7 @@ DEF_OP(AtomicFetchAnd) {
L(Loop);
mov(TMP2.cvt32(), TMP1.cvt32());
mov(TMP3.cvt32(), TMP1.cvt32());
and_(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
and_(TMP2.cvt32(), GetSrc<RA_32>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
@@ -391,7 +384,7 @@ DEF_OP(AtomicFetchAnd) {
L(Loop);
mov(TMP2.cvt64(), TMP1.cvt64());
mov(TMP3.cvt64(), TMP1.cvt64());
and_(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
and_(TMP2.cvt64(), GetSrc<RA_64>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
@@ -409,7 +402,7 @@ DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
// TMP1 = rax
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
switch (IROp->Size) {
case 1: {
mov(TMP1.cvt8(), byte [MemReg]);
@@ -418,7 +411,7 @@ DEF_OP(AtomicFetchOr) {
L(Loop);
mov(TMP2.cvt8(), TMP1.cvt8());
mov(TMP3.cvt8(), TMP1.cvt8());
or_(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
or_(TMP2.cvt8(), GetSrc<RA_8>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
@@ -434,7 +427,7 @@ DEF_OP(AtomicFetchOr) {
L(Loop);
mov(TMP2.cvt16(), TMP1.cvt16());
mov(TMP3.cvt16(), TMP1.cvt16());
or_(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
or_(TMP2.cvt16(), GetSrc<RA_16>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
@@ -451,7 +444,7 @@ DEF_OP(AtomicFetchOr) {
L(Loop);
mov(TMP2.cvt32(), TMP1.cvt32());
mov(TMP3.cvt32(), TMP1.cvt32());
or_(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
or_(TMP2.cvt32(), GetSrc<RA_32>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
@@ -468,7 +461,7 @@ DEF_OP(AtomicFetchOr) {
L(Loop);
mov(TMP2.cvt64(), TMP1.cvt64());
mov(TMP3.cvt64(), TMP1.cvt64());
or_(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
or_(TMP2.cvt64(), GetSrc<RA_64>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
@@ -486,7 +479,7 @@ DEF_OP(AtomicFetchXor) {
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
// TMP1 = rax
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
switch (IROp->Size) {
case 1: {
mov(TMP1.cvt8(), byte [MemReg]);
@@ -495,7 +488,7 @@ DEF_OP(AtomicFetchXor) {
L(Loop);
mov(TMP2.cvt8(), TMP1.cvt8());
mov(TMP3.cvt8(), TMP1.cvt8());
xor_(TMP2.cvt8(), GetSrc<RA_8>(Op->Header.Args[1].ID()));
xor_(TMP2.cvt8(), GetSrc<RA_8>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
@@ -511,7 +504,7 @@ DEF_OP(AtomicFetchXor) {
L(Loop);
mov(TMP2.cvt16(), TMP1.cvt16());
mov(TMP3.cvt16(), TMP1.cvt16());
xor_(TMP2.cvt16(), GetSrc<RA_16>(Op->Header.Args[1].ID()));
xor_(TMP2.cvt16(), GetSrc<RA_16>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
@@ -528,7 +521,7 @@ DEF_OP(AtomicFetchXor) {
L(Loop);
mov(TMP2.cvt32(), TMP1.cvt32());
mov(TMP3.cvt32(), TMP1.cvt32());
xor_(TMP2.cvt32(), GetSrc<RA_32>(Op->Header.Args[1].ID()));
xor_(TMP2.cvt32(), GetSrc<RA_32>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
@@ -545,7 +538,7 @@ DEF_OP(AtomicFetchXor) {
L(Loop);
mov(TMP2.cvt64(), TMP1.cvt64());
mov(TMP3.cvt64(), TMP1.cvt64());
xor_(TMP2.cvt64(), GetSrc<RA_64>(Op->Header.Args[1].ID()));
xor_(TMP2.cvt64(), GetSrc<RA_64>(Op->Value.ID()));
// Updates RAX with the value from memory
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
@@ -562,7 +555,7 @@ DEF_OP(AtomicFetchXor) {
DEF_OP(AtomicFetchNeg) {
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Header.Args[0].ID());
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
switch (IROp->Size) {
case 1: {
mov(TMP1.cvt8(), byte [MemReg]);
@@ -37,18 +37,13 @@ DEF_OP(GuestCallIndirect) {
LogMan::Msg::DFmt("Unimplemented");
}
DEF_OP(GuestReturn) {
LogMan::Msg::DFmt("Unimplemented");
}
DEF_OP(SignalReturn) {
// Adjust the stack first for a regular return
if (SpillSlots) {
add(rsp, SpillSlots * 16); // + 8 to consume return address
}
mov(TMP1, ThreadSharedData.SignalHandlerReturnAddress);
jmp(TMP1);
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalReturnHandler)]);
}
DEF_OP(CallbackReturn) {
@@ -58,8 +53,7 @@ DEF_OP(CallbackReturn) {
}
// Make sure to adjust the refcounter so we don't clear the cache now
mov(rax, reinterpret_cast<uint64_t>(ThreadSharedData.SignalHandlerRefCounterPtr));
sub(dword [rax], 1);
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalHandlerRefCountPointer)], 1);
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 8);
@@ -104,7 +98,8 @@ DEF_OP(ExitFunction) {
Xbyak::Reg RipReg = GetSrc<RA_64>(Op->NewRIP.ID());
// L1 Cache
mov(rcx, ThreadState->LookupCache->GetL1Pointer());
mov(rcx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.L1Pointer)]);
mov(rax, RipReg);
and_(rax, LookupCache::L1_ENTRIES_MASK);
@@ -117,9 +112,8 @@ DEF_OP(ExitFunction) {
jmp(qword[LookupBase + 0]);
L(FullLookup);
mov(rax, ThreadSharedData.Dispatcher->AbsoluteLoopTopAddress);
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], RipReg);
jmp(rax);
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.DispatcherLoopTop)]);
}
#ifdef BLOCKSTATS
@@ -187,15 +181,13 @@ DEF_OP(Syscall) {
}
mov(rsi, STATE); // Move thread in to rsi
mov(rdi, reinterpret_cast<uint64_t>(CTX->SyscallHandler));
mov(rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SyscallHandlerObj)]);
mov(rdx, rsp);
mov(rax, reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall));
if (NumPush & 1)
sub(rsp, 8); // Align
// {rdi, rsi, rdx}
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SyscallHandlerFunc)]);
if (NumPush & 1)
add(rsp, 8); // Align
@@ -280,9 +272,7 @@ DEF_OP(RemoveCodeEntry) {
mov(rax, Entry); // imm64 move
mov(rsi, rax);
mov(rax, reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit));
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.RemoveCodeEntryFromJIT)]);
if (NumPush & 1)
add(rsp, 8); // Align
@@ -294,13 +284,6 @@ DEF_OP(RemoveCodeEntry) {
DEF_OP(CPUID) {
auto Op = IROp->C<IR::IROp_CPUID>();
using ClassPtrType = FEXCore::CPUID::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function, uint32_t Leaf);
union {
ClassPtrType ClassPtr;
uint64_t Raw;
} Ptr;
Ptr.ClassPtr = &CPUIDEmu::RunFunction;
for (auto &Reg : RA64)
push(Reg);
@@ -313,18 +296,15 @@ DEF_OP(CPUID) {
// rsi can be in the source registers, so copy argument to edx first
mov (edx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
mov (esi, GetSrc<RA_32>(Op->Header.Args[0].ID()));
mov (rdi, reinterpret_cast<uint64_t>(&CTX->CPUID));
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.CPUIDObj)]);
auto NumPush = RA64.size();
if (NumPush & 1)
sub(rsp, 8); // Align
mov(rax, Ptr.Raw);
// {rdi, rsi, rdx}
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.CPUIDFunction)]);
if (NumPush & 1)
add(rsp, 8); // Align
@@ -342,7 +322,6 @@ void X86JITCore::RegisterBranchHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
REGISTER_OP(GUESTRETURN, GuestReturn);
REGISTER_OP(SIGNALRETURN, SignalReturn);
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
REGISTER_OP(EXITFUNCTION, ExitFunction);
@@ -18,23 +18,23 @@ namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
DEF_OP(VInsGPR) {
auto Op = IROp->C<IR::IROp_VInsGPR>();
movapd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
movapd(GetDst(Node), GetSrc(Op->DestVector.ID()));
switch (Op->Header.ElementSize) {
case 1: {
pinsrb(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()), Op->Index);
pinsrb(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
break;
}
case 2: {
pinsrw(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()), Op->Index);
pinsrw(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
break;
}
case 4: {
pinsrd(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[1].ID()), Op->Index);
pinsrd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
break;
}
case 8: {
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[1].ID()), Op->Index);
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()), Op->DestIdx);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
+49 -37
View File
@@ -15,6 +15,8 @@ $end_info$
#include "Interface/IR/PassManager.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Utils/MemberFunctionToPointer.h"
#include <FEXCore/Core/CPUBackend.h>
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/SignalDelegator.h>
@@ -27,7 +29,6 @@ $end_info$
#include <algorithm>
#include <array>
#include <bits/types/stack_t.h>
#include <memory>
#include <stddef.h>
#include <stdint.h>
@@ -42,6 +43,16 @@ $end_info$
// #define DEBUG_RA 1
// #define DEBUG_CYCLES
namespace {
static void PrintValue(uint64_t Value) {
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
}
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
}
}
namespace FEXCore::CPU {
CodeBuffer AllocateNewCodeBuffer(FEXCore::Context::Context *CTX, size_t Size) {
@@ -109,9 +120,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
case FABI_VOID_U16: {
PushRegs();
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
break;
@@ -120,9 +129,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushRegs();
movss(xmm0, GetSrc(IROp->Args[0].ID()));
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -136,9 +143,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushRegs();
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -153,9 +158,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
PushRegs();
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -171,9 +174,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -187,9 +188,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -203,9 +202,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -218,9 +215,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -233,9 +228,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -251,9 +244,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
movq(rdx, GetSrc(IROp->Args[1].ID()));
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -266,9 +257,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -286,9 +275,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
movq(rdx, GetSrc(IROp->Args[1].ID()));
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
mov(rax, (uintptr_t)Info.fn);
call(rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -317,6 +304,31 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
, ThreadState {Thread}
, InitialCodeBuffer {Buffer}
{
{
// Set up pointers that the JIT needs to load
auto &Pointers = ThreadState->CurrentFrame->Pointers.X86;
// Process specific
Pointers.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
Pointers.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
Pointers.RemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveCodeEntryFromJit);
Pointers.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
{
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
Pointers.CPUIDFunction = PMF.GetConvertedPointer();
}
Pointers.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
Pointers.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
// Fill in the fallback handlers
InterpreterOps::FillFallbackIndexPointers(Pointers.FallbackHandlerPointers);
// Thread Specific
Pointers.SignalHandlerRefCountPointer = reinterpret_cast<uint64_t>(&Dispatcher->SignalHandlerRefCounter);
}
CurrentCodeBuffer = &InitialCodeBuffer;
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
@@ -276,7 +276,6 @@ private:
///< Branch ops
DEF_OP(GuestCallDirect);
DEF_OP(GuestCallIndirect);
DEF_OP(GuestReturn);
DEF_OP(SignalReturn);
DEF_OP(CallbackReturn);
DEF_OP(ExitFunction);
@@ -329,6 +328,7 @@ private:
DEF_OP(GetRoundingMode);
DEF_OP(SetRoundingMode);
DEF_OP(ProcessorID);
DEF_OP(RDRAND);
///< Move ops
DEF_OP(ExtractElementPair);
@@ -338,8 +338,6 @@ private:
///< Vector ops
DEF_OP(VectorZero);
DEF_OP(VectorImm);
DEF_OP(CreateVector2);
DEF_OP(CreateVector4);
DEF_OP(SplatVector);
DEF_OP(VMov);
DEF_OP(VAnd);
@@ -426,6 +424,7 @@ private:
DEF_OP(VSMull2);
DEF_OP(VUABDL);
DEF_OP(VTBL1);
DEF_OP(VRev64);
///< Encryption ops
DEF_OP(AESImc);
@@ -23,7 +23,7 @@ DEF_OP(LoadContext) {
auto Op = IROp->C<IR::IROp_LoadContext>();
uint8_t OpSize = IROp->Size;
if (Op->Class.Val == 0) {
if (Op->Class == IR::GPRClass) {
switch (OpSize) {
case 1: {
movzx(GetDst<RA_32>(Node), byte [STATE + Op->Offset]);
@@ -84,23 +84,23 @@ DEF_OP(StoreContext) {
auto Op = IROp->C<IR::IROp_StoreContext>();
uint8_t OpSize = IROp->Size;
if (Op->Class.Val == 0) {
if (Op->Class == IR::GPRClass) {
switch (OpSize) {
case 1: {
mov(byte [STATE + Op->Offset], GetSrc<RA_8>(Op->Header.Args[0].ID()));
mov(byte [STATE + Op->Offset], GetSrc<RA_8>(Op->Value.ID()));
}
break;
case 2: {
mov(word [STATE + Op->Offset], GetSrc<RA_16>(Op->Header.Args[0].ID()));
mov(word [STATE + Op->Offset], GetSrc<RA_16>(Op->Value.ID()));
}
break;
case 4: {
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Header.Args[0].ID()));
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
}
break;
case 8: {
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Header.Args[0].ID()));
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
}
break;
case 16:
@@ -112,27 +112,27 @@ DEF_OP(StoreContext) {
else {
switch (OpSize) {
case 1: {
pextrb(byte [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()), 0);
pextrb(byte [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
}
break;
case 2: {
pextrw(word [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()), 0);
pextrw(word [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
}
break;
case 4: {
vmovd(dword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()));
vmovd(dword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
}
break;
case 8: {
vmovq(qword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()));
vmovq(qword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
}
break;
case 16: {
if (Op->Offset % 16 == 0)
movaps(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()));
movaps(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
else
movups(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID()));
movups(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
@@ -143,9 +143,9 @@ DEF_OP(StoreContext) {
DEF_OP(LoadContextIndexed) {
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
size_t size = IROp->Size;
Reg index = GetSrc<RA_64>(Op->Header.Args[0].ID());
Reg index = GetSrc<RA_64>(Op->Index.ID());
if (Op->Class.Val == 0) {
if (Op->Class == IR::GPRClass) {
switch (Op->Stride) {
case 1:
case 2:
@@ -245,11 +245,11 @@ DEF_OP(LoadContextIndexed) {
DEF_OP(StoreContextIndexed) {
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
Reg index = GetSrc<RA_64>(Op->Header.Args[1].ID());
Reg index = GetSrc<RA_64>(Op->Index.ID());
size_t size = IROp->Size;
if (Op->Class.Val == 0) {
auto value = GetSrc<RA_64>(Op->Header.Args[0].ID());
if (Op->Class == IR::GPRClass) {
auto value = GetSrc<RA_64>(Op->Value.ID());
lea(rax, dword [STATE + Op->BaseOffset]);
switch (Op->Stride) {
@@ -269,7 +269,7 @@ DEF_OP(StoreContextIndexed) {
}
}
else {
auto value = GetSrc(Op->Header.Args[0].ID());
auto value = GetSrc(Op->Value.ID());
switch (Op->Stride) {
case 1:
case 2:
@@ -339,19 +339,19 @@ DEF_OP(SpillRegister) {
if (Op->Class == FEXCore::IR::GPRClass) {
switch (OpSize) {
case 1: {
mov(byte [rsp + SlotOffset], GetSrc<RA_8>(Op->Header.Args[0].ID()));
mov(byte [rsp + SlotOffset], GetSrc<RA_8>(Op->Value.ID()));
break;
}
case 2: {
mov(word [rsp + SlotOffset], GetSrc<RA_16>(Op->Header.Args[0].ID()));
mov(word [rsp + SlotOffset], GetSrc<RA_16>(Op->Value.ID()));
break;
}
case 4: {
mov(dword [rsp + SlotOffset], GetSrc<RA_32>(Op->Header.Args[0].ID()));
mov(dword [rsp + SlotOffset], GetSrc<RA_32>(Op->Value.ID()));
break;
}
case 8: {
mov(qword [rsp + SlotOffset], GetSrc<RA_64>(Op->Header.Args[0].ID()));
mov(qword [rsp + SlotOffset], GetSrc<RA_64>(Op->Value.ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
@@ -359,15 +359,15 @@ DEF_OP(SpillRegister) {
} else if (Op->Class == FEXCore::IR::FPRClass) {
switch (OpSize) {
case 4: {
movss(dword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID()));
movss(dword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
break;
}
case 8: {
movsd(qword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID()));
movsd(qword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
break;
}
case 16: {
movaps(xword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID()));
movaps(xword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
@@ -435,7 +435,7 @@ DEF_OP(LoadFlag) {
DEF_OP(StoreFlag) {
auto Op = IROp->C<IR::IROp_StoreFlag>();
mov (rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
mov (rax, GetSrc<RA_64>(Op->Value.ID()));
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
}
@@ -469,7 +469,7 @@ DEF_OP(LoadMem) {
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class.Val == 0) {
if (Op->Class == IR::GPRClass) {
auto Dst = GetDst<RA_64>(Node);
switch (IROp->Size) {
@@ -537,19 +537,19 @@ DEF_OP(StoreMem) {
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class.Val == 0) {
if (Op->Class == IR::GPRClass) {
switch (IROp->Size) {
case 1:
mov(byte [MemPtr], GetSrc<RA_8>(Op->Header.Args[1].ID()));
mov(byte [MemPtr], GetSrc<RA_8>(Op->Value.ID()));
break;
case 2:
mov(word [MemPtr], GetSrc<RA_16>(Op->Header.Args[1].ID()));
mov(word [MemPtr], GetSrc<RA_16>(Op->Value.ID()));
break;
case 4:
mov(dword [MemPtr], GetSrc<RA_32>(Op->Header.Args[1].ID()));
mov(dword [MemPtr], GetSrc<RA_32>(Op->Value.ID()));
break;
case 8:
mov(qword [MemPtr], GetSrc<RA_64>(Op->Header.Args[1].ID()));
mov(qword [MemPtr], GetSrc<RA_64>(Op->Value.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
}
@@ -557,22 +557,22 @@ DEF_OP(StoreMem) {
else {
switch (IROp->Size) {
case 1:
pextrb(byte [MemPtr], GetSrc(Op->Header.Args[1].ID()), 0);
pextrb(byte [MemPtr], GetSrc(Op->Value.ID()), 0);
break;
case 2:
pextrw(word [MemPtr], GetSrc(Op->Header.Args[1].ID()), 0);
pextrw(word [MemPtr], GetSrc(Op->Value.ID()), 0);
break;
case 4:
vmovd(dword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
vmovd(dword [MemPtr], GetSrc(Op->Value.ID()));
break;
case 8:
vmovq(qword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
vmovq(qword [MemPtr], GetSrc(Op->Value.ID()));
break;
case 16:
if (IROp->Size == Op->Align)
movups(xword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
else
movups(xword [MemPtr], GetSrc(Op->Header.Args[1].ID()));
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
}
+25 -24
View File
@@ -18,14 +18,6 @@ $end_info$
#include <xbyak/xbyak.h>
namespace FEXCore::CPU {
static void PrintValue(uint64_t Value) {
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
}
static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
LogMan::Msg::DFmt("Value: 0x{:016x}'{:016x}", ValueUpper, Value);
}
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
DEF_OP(Fence) {
@@ -53,8 +45,7 @@ DEF_OP(Break) {
break;
case FEXCore::IR::Break_Overflow: // overflow
// Need to be outside of JIT cache space to ensure cache clearing correctness
mov(TMP1, ThreadSharedData.Dispatcher->OverflowExceptionInstructionAddress);
jmp(TMP1);
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.OverflowExceptionHandler)]);
break;
case FEXCore::IR::Break_Halt: { // HLT
// Time to quit
@@ -62,8 +53,7 @@ DEF_OP(Break) {
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
// Now we need to jump to the thread stop handler
mov(TMP1, ThreadSharedData.Dispatcher->ThreadStopHandlerAddress);
jmp(TMP1);
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.ThreadStopHandler)]);
break;
}
case FEXCore::IR::Break_Interrupt3: // INT3
@@ -75,8 +65,7 @@ DEF_OP(Break) {
}
// This jump target needs to be a constant offset here
mov(TMP1, ThreadSharedData.Dispatcher->ThreadPauseHandlerAddress);
jmp(TMP1);
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.ThreadPauseHandler)]);
}
else {
// If we don't have a gdb server attached then....crash?
@@ -84,8 +73,7 @@ DEF_OP(Break) {
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
// Now we need to jump to the thread stop handler
mov(TMP1, ThreadSharedData.Dispatcher->ThreadStopHandlerAddress);
jmp(TMP1);
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.ThreadStopHandler)]);
}
break;
}
@@ -96,9 +84,7 @@ DEF_OP(Break) {
}
// Need to be outside of JIT cache space to ensure cache clearing correctness
mov(TMP1, ThreadSharedData.Dispatcher->UnimplementedInstructionAddress);
jmp(TMP1);
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.UnimplementedInstructionHandler)]);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
@@ -144,18 +130,15 @@ DEF_OP(Print) {
PushRegs();
if (IsGPR(Op->Header.Args[0].ID())) {
mov (rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
mov(rax, reinterpret_cast<uintptr_t>(PrintValue));
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.PrintValue)]);
}
else {
pextrq(rdi, GetSrc(Op->Header.Args[0].ID()), 0);
pextrq(rsi, GetSrc(Op->Header.Args[0].ID()), 1);
mov(rax, reinterpret_cast<uintptr_t>(PrintVectorValue));
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.PrintVectorValue)]);
}
call(rax);
PopRegs();
}
@@ -166,6 +149,23 @@ DEF_OP(ProcessorID) {
mov (GetDst<RA_32>(Node), ecx);
}
DEF_OP(RDRAND) {
auto Op = IROp->C<IR::IROp_RDRAND>();
auto Dst = GetSrcPair<RA_64>(Node);
if (Op->GetReseeded) {
rdrand(Dst.first);
}
else {
rdseed(Dst.first);
}
// In the case of RDRAND or RDSEED returning a valid number then CF = 1, else 0
mov (Dst.second, 0);
setc(Dst.second.cvt8());
}
#undef DEF_OP
void X86JITCore::RegisterMiscHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
@@ -183,6 +183,7 @@ void X86JITCore::RegisterMiscHandlers() {
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
REGISTER_OP(INVALIDATEFLAGS, NoOp);
REGISTER_OP(PROCESSORID, ProcessorID);
REGISTER_OP(RDRAND, RDRAND);
#undef REGISTER_OP
}
}
@@ -42,7 +42,7 @@ DEF_OP(CreateElementPair) {
Xbyak::Reg RegSecond;
Xbyak::Reg RegTmp;
switch (Op->Header.Size) {
switch (IROp->ElementSize) {
case 4: {
Dst = GetSrcPair<RA_32>(Node);
RegFirst = GetSrc<RA_32>(Op->Header.Args[0].ID());
@@ -67,14 +67,6 @@ DEF_OP(VectorImm) {
}
}
DEF_OP(CreateVector2) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(CreateVector4) {
LOGMAN_MSG_A_FMT("Unimplemented");
}
DEF_OP(SplatVector) {
auto Op = IROp->C<IR::IROp_SplatVector2>();
uint8_t OpSize = IROp->Size;
@@ -1545,7 +1537,7 @@ DEF_OP(VInsScalarElement) {
DEF_OP(VExtractElement) {
auto Op = IROp->C<IR::IROp_VExtractElement>();
switch (Op->Header.ElementSize) {
switch (Op->Header.Size) {
case 1: {
pextrb(eax, GetSrc(Op->Header.Args[0].ID()), Op->Index);
pinsrb(GetDst(Node), eax, 0);
@@ -1566,7 +1558,7 @@ DEF_OP(VExtractElement) {
pinsrq(GetDst(Node), rax, 0);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.Size); break;
}
}
@@ -2151,13 +2143,79 @@ DEF_OP(VTBL1) {
}
}
DEF_OP(VRev64) {
auto Op = IROp->C<IR::IROp_VDupElement>();
switch (Op->Header.ElementSize) {
case 1: {
mov(rax, 0x00'01'02'03'04'05'06'07); // Lower
vmovq(xmm15, rax);
if (IROp->Size == 16) {
// Full 8bit byteswap in each 64-bit element
mov(rcx, 0x08'09'0A'0B'0C'0D'0E'0F); // Upper
pinsrq(xmm15, rcx, 1);
}
else {
// 8byte, upper bits get zero
// Full 8bit byteswap in each 64-bit element
mov(rcx, 0x80'80'80'80'80'80'80'80); // Upper
pinsrq(xmm15, rcx, 1);
}
vpshufb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), xmm15);
break;
}
case 2: {
// Full 16-bit byteswap in each 64-bit element
mov(rax, 0x01'00'03'02'05'04'07'06); // Lower
vmovq(xmm15, rax);
if (IROp->Size == 16) {
mov(rcx, 0x09'08'0B'0A'0D'0C'0F'0E); // Upper
pinsrq(xmm15, rcx, 1);
}
else {
// 8byte, upper bits get zero
// Full 8bit byteswap in each 64-bit element
mov(rcx, 0x80'80'80'80'80'80'80'80); // Upper
pinsrq(xmm15, rcx, 1);
}
vpshufb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), xmm15);
break;
}
case 4: {
if (IROp->Size == 16) {
vpshufd(GetDst(Node),
GetSrc(Op->Header.Args[0].ID()),
(0b11 << 0) |
(0b10 << 2) |
(0b01 << 4) |
(0b00 << 6));
}
else {
vpshufd(GetDst(Node),
GetSrc(Op->Header.Args[0].ID()),
(0b01 << 0) |
(0b00 << 2) |
(0b11 << 4) | // Last two don't matter, will be overwritten with zero
(0b11 << 6));
// Zero upper 64-bits
mov(rcx, 0);
pinsrq(GetDst(Node), rcx, 1);
}
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
}
}
#undef DEF_OP
void X86JITCore::RegisterVectorHandlers() {
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
REGISTER_OP(VECTORZERO, VectorZero);
REGISTER_OP(VECTORIMM, VectorImm);
REGISTER_OP(CREATEVECTOR2, CreateVector2);
REGISTER_OP(CREATEVECTOR4, CreateVector4);
REGISTER_OP(SPLATVECTOR2, SplatVector);
REGISTER_OP(SPLATVECTOR4, SplatVector);
REGISTER_OP(VMOV, VMov);
@@ -2246,6 +2304,7 @@ void X86JITCore::RegisterVectorHandlers() {
REGISTER_OP(VSMULL2, VSMull2);
REGISTER_OP(VUABDL, VUABDL);
REGISTER_OP(VTBL1, VTBL1);
REGISTER_OP(VREV64, VRev64);
#undef REGISTER_OP
}
}
File diff suppressed because it is too large. Load diff
+34 -6
View File
@@ -65,6 +65,7 @@ public:
TYPE_TZCNT,
TYPE_LZCNT,
TYPE_BITSELECT,
TYPE_RDRAND,
};
SelectionFlag flagsOp{};
@@ -273,6 +274,8 @@ public:
void XADDOp(OpcodeArgs);
void PopcountOp(OpcodeArgs);
void XLATOp(OpcodeArgs);
template<bool Reseed>
void RDRANDOp(OpcodeArgs);
enum class Segment {
FS,
@@ -296,9 +299,14 @@ public:
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void VectorALUOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void VectorALUROp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void VectorScalarALUOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize, bool Scalar>
void VectorUnaryOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void VectorUnaryDuplicateOp(OpcodeArgs);
void MOVQOp(OpcodeArgs);
template<size_t ElementSize>
void PADDQOp(OpcodeArgs);
@@ -485,6 +493,17 @@ public:
template<size_t ElementSize>
void ADDSUBPOp(OpcodeArgs);
void PFNACCOp(OpcodeArgs);
void PFPNACCOp(OpcodeArgs);
void PSWAPDOp(OpcodeArgs);
template<uint8_t CompType>
void VPFCMPOp(OpcodeArgs);
void PI2FWOp(OpcodeArgs);
void PF2IWOp(OpcodeArgs);
void PMULHRWOp(OpcodeArgs);
void PMADDWD(OpcodeArgs);
void PMADDUBSW(OpcodeArgs);
@@ -629,7 +648,7 @@ private:
OrderedNode *Res{};
union {
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, BITSELECT
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, BITSELECT, RDRAND
struct {
} NoSource;
@@ -726,6 +745,7 @@ private:
void CalculcateFlags_TZCNT(OrderedNode *Src);
void CalculcateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
void CalculcateFlags_BITSELECT(OrderedNode *Src);
void CalculcateFlags_RDRAND(OrderedNode *Src);
/** @} */
/**
@@ -1106,6 +1126,14 @@ private:
};
}
void GenerateFlags_RDRAND(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
CurrentDeferredFlags = DeferredFlagData {
.Type = FlagsGenerationType::TYPE_RDRAND,
.SrcSize = GetSrcSize(Op),
.Res = Src,
};
}
/** @} */
/** @} */
@@ -1134,18 +1162,18 @@ private:
bool Multiblock{};
uint64_t Entry;
OrderedNode* _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Align = 1) {
OrderedNode* _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
if (CTX->Config.TSOEnabled)
return _StoreMemTSO(ssa0, ssa1, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
else
return _StoreMem(ssa0, ssa1, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
OrderedNode* _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
if (CTX->Config.TSOEnabled)
return _LoadMemTSO(ssa0, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
else
return _LoadMem(ssa0, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
};
@@ -249,6 +249,9 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
case FlagsGenerationType::TYPE_BITSELECT:
CalculcateFlags_BITSELECT(CurrentDeferredFlags.Res);
break;
case FlagsGenerationType::TYPE_RDRAND:
CalculcateFlags_RDRAND(CurrentDeferredFlags.Res);
break;
case FlagsGenerationType::TYPE_NONE:
default: ERROR_AND_DIE_FMT("Unhandled flags type {}", CurrentDeferredFlags.Type);
}
@@ -1259,4 +1262,17 @@ void OpDispatchBuilder::CalculcateFlags_BITSELECT(OrderedNode *Src) {
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFSelectOp);
}
void OpDispatchBuilder::CalculcateFlags_RDRAND(OrderedNode *Src) {
// OF, SF, ZF, AF, PF all zero
// CF is set to the incoming source
auto ZeroConst = _Constant(0);
SetRFLAG<X86State::RFLAG_OF_LOC>(ZeroConst);
SetRFLAG<X86State::RFLAG_SF_LOC>(ZeroConst);
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZeroConst);
SetRFLAG<X86State::RFLAG_AF_LOC>(ZeroConst);
SetRFLAG<X86State::RFLAG_PF_LOC>(ZeroConst);
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Src);
}
}
@@ -297,6 +297,24 @@ void OpDispatchBuilder::VectorALUOp<IR::OP_VUQSUB, 1>(OpcodeArgs);
template
void OpDispatchBuilder::VectorALUOp<IR::OP_VUQSUB, 2>(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void OpDispatchBuilder::VectorALUROp(OpcodeArgs) {
auto Size = GetSrcSize(Op);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
auto ALUOp = _VAdd(Size, ElementSize, Src, Dest);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
StoreResult(FPRClass, Op, ALUOp, -1);
}
template
void OpDispatchBuilder::VectorALUROp<IR::OP_VFSUB, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorALUROp<IR::OP_VFSUB, 8>(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) {
auto Size = GetSrcSize(Op);
@@ -392,14 +410,35 @@ void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2, false>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4, false>(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void OpDispatchBuilder::VectorUnaryDuplicateOp(OpcodeArgs) {
auto Size = GetSrcSize(Op);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
auto ALUOp = _VFSqrt(ElementSize, ElementSize, Src);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
// Duplicate the lower bits
auto Result = _VDupElement(Size, ElementSize, ALUOp, 0);
StoreResult(FPRClass, Op, Result, -1);
}
template
void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, 4>(OpcodeArgs);
void OpDispatchBuilder::MOVQOp(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
// This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 128bit
if (Op->Dest.IsGPR()) {
const auto gpr = Op->Dest.Data.GPR.GPR;
_StoreContext(FPRClass, 8, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][0]), Src);
_StoreContext(8, FPRClass, Src, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][0]));
auto Const = _Constant(0);
_StoreContext(GPRClass, 8, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][1]), Const);
_StoreContext(8, GPRClass, Const, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][1]));
}
else {
// This is simple, just store the result
@@ -440,14 +479,15 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) {
//TODO: We could remove this VCastFromGOR + VInsGPR pair if we had a VDUPFromGPR instruction that maps directly to AArch64.
auto M = _Constant(0x80'40'20'10'08'04'02'01ULL);
OrderedNode *VMask = _VCastFromGPR(16, 8, M);
VMask = _VInsGPR(16, 8, VMask, M, 1);
auto VCMP = _VCMPLTZ(Src, 16, 1);
auto VAnd = _VAnd(VCMP, VMask, 16, 1);
VMask = _VInsGPR(16, 8, 1, VMask, M);
auto VAdd1 = _VAddP(VAnd, VAnd, 16, 1);
auto VAdd2 = _VAddP(VAdd1, VAdd1, 8, 1);
auto VAdd3 = _VAddP(VAdd2, VAdd2, 8, 1);
auto VCMP = _VCMPLTZ(16, 1, Src);
auto VAnd = _VAnd(16, 1, VCMP, VMask);
auto VAdd1 = _VAddP(16, 1, VAnd, VAnd);
auto VAdd2 = _VAddP(8, 1, VAdd1, VAdd1);
auto VAdd3 = _VAddP(8, 1, VAdd2, VAdd2);
StoreResult(GPRClass, Op, _VExtractToGPR(16, 2, VAdd3, 0), -1);
}
@@ -504,11 +544,11 @@ void OpDispatchBuilder::PSHUFBOp(OpcodeArgs) {
// Bits [6:4] is reserved for 128bit
// Bits [6:3] is reserved for 64bit
if (Size == 8) {
auto MaskVector = _VectorImm(0b1000'0111, Size, 1);
auto MaskVector = _VectorImm(Size, 1, 0b1000'0111);
Src = _VAnd(Size, Size, Src, MaskVector);
}
else {
auto MaskVector = _VectorImm(0b1000'1111, Size, 1);
auto MaskVector = _VectorImm(Size, 1, 0b1000'1111);
Src = _VAnd(Size, Size, Src, MaskVector);
}
auto Res = _VTBL1(Size, Dest, Src);
@@ -629,7 +669,7 @@ void OpDispatchBuilder::PINSROp(OpcodeArgs) {
Index &= NumElements - 1;
// This maps 1:1 to an AArch64 NEON Op
auto ALUOp = _VInsGPR(Size, ElementSize, Dest, Src, Index);
auto ALUOp = _VInsGPR(Size, ElementSize, Index, Dest, Src);
StoreResult(FPRClass, Op, ALUOp, -1);
}
@@ -672,10 +712,10 @@ void OpDispatchBuilder::InsertPSOp(OpcodeArgs) {
// ZMask happens after insert
if (ZMask == 0xF) {
Dest = _VectorImm(0, 16, 4);
Dest = _VectorImm(16, 4, 0);
}
else if (ZMask) {
auto Zero = _VectorImm(0, 16, 4);
auto Zero = _VectorImm(16, 4, 0);
for (size_t i = 0; i < 4; ++i) {
if (ZMask & (1 << i)) {
Dest = _VInsElement(GetDstSize(Op), 4, i, 0, Dest, Zero);
@@ -768,7 +808,7 @@ void OpDispatchBuilder::PSRLDOp(OpcodeArgs) {
OrderedNode *Result{};
// Incoming element size for the shift source is always 8
auto MaxShift = _VectorImm(ElementSize * 8, 8, 8);
auto MaxShift = _VectorImm(8, 8, ElementSize * 8);
Src = _VUMin(8, 8, MaxShift, Src);
Result = _VUShrS(Size, ElementSize, Dest, Src);
@@ -832,7 +872,7 @@ void OpDispatchBuilder::PSLL(OpcodeArgs) {
OrderedNode *Result{};
// Incoming element size for the shift source is always 8
auto MaxShift = _VectorImm(ElementSize * 8, 8, 8);
auto MaxShift = _VectorImm(8, 8, ElementSize * 8);
Src = _VUMin(8, 8, MaxShift, Src);
Result = _VUShlS(Size, ElementSize, Dest, Src);
@@ -856,7 +896,7 @@ void OpDispatchBuilder::PSRAOp(OpcodeArgs) {
OrderedNode *Result{};
// Incoming element size for the shift source is always 8
auto MaxShift = _VectorImm(ElementSize * 8, 8, 8);
auto MaxShift = _VectorImm(8, 8, ElementSize * 8);
Src = _VUMin(8, 8, MaxShift, Src);
Result = _VSShrS(Size, ElementSize, Dest, Src);
@@ -959,12 +999,11 @@ void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) {
// Source Element size is determined by instruction
size_t GPRSize = GetDstSize(Op);
size_t ElementSize = SrcElementSize;
if constexpr (HostRoundingMode) {
Src = _Float_ToGPR_S(Src, ElementSize, GPRSize);
Src = _Float_ToGPR_S(GPRSize, SrcElementSize, Src);
}
else {
Src = _Float_ToGPR_ZS(Src, ElementSize, GPRSize);
Src = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
}
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, GPRSize, -1);
@@ -987,11 +1026,11 @@ void OpDispatchBuilder::Vector_CVT_Int_To_Float(OpcodeArgs) {
size_t ElementSize = SrcElementSize;
size_t Size = GetDstSize(Op);
if constexpr (Widen) {
Src = _VSXTL(Src, Size, ElementSize);
Src = _VSXTL(Size, ElementSize, Src);
ElementSize <<= 1;
}
Src = _Vector_SToF(Src, Size, ElementSize);
Src = _Vector_SToF(Size, ElementSize, Src);
StoreResult(FPRClass, Op, Src, -1);
}
@@ -1009,15 +1048,15 @@ void OpDispatchBuilder::Vector_CVT_Float_To_Int(OpcodeArgs) {
size_t Size = GetDstSize(Op);
if constexpr (Narrow) {
Src = _Vector_FToF(Size, SrcElementSize >> 1, SrcElementSize, Src);
Src = _Vector_FToF(Size, SrcElementSize >> 1, Src, SrcElementSize);
ElementSize >>= 1;
}
if constexpr (HostRoundingMode) {
Src = _Vector_FToS(Src, Size, ElementSize);
Src = _Vector_FToS(Size, ElementSize, Src);
}
else {
Src = _Vector_FToZS(Src, Size, ElementSize);
Src = _Vector_FToZS(Size, ElementSize, Src);
}
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, -1);
@@ -1027,6 +1066,8 @@ template
void OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>(OpcodeArgs);
template
void OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, true>(OpcodeArgs);
template
void OpDispatchBuilder::Vector_CVT_Float_To_Int<4, true, false>(OpcodeArgs);
template
void OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, true>(OpcodeArgs);
@@ -1055,10 +1096,10 @@ void OpDispatchBuilder::Vector_CVT_Float_To_Float(OpcodeArgs) {
size_t Size = GetDstSize(Op);
if constexpr (DstElementSize > SrcElementSize) {
Src = _Vector_FToF(Size, SrcElementSize << 1, SrcElementSize, Src);
Src = _Vector_FToF(Size, SrcElementSize << 1, Src, SrcElementSize);
}
else {
Src = _Vector_FToF(Size, SrcElementSize >> 1, SrcElementSize, Src);
Src = _Vector_FToF(Size, SrcElementSize >> 1, Src, SrcElementSize);
}
StoreResult(FPRClass, Op, Src, -1);
@@ -1076,12 +1117,12 @@ void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) {
size_t ElementSize = SrcElementSize;
size_t DstSize = GetDstSize(Op);
if constexpr (Widen) {
Src = _VSXTL(Src, DstSize, ElementSize);
Src = _VSXTL(DstSize, ElementSize, Src);
ElementSize <<= 1;
}
// Always signed
Src = _Vector_SToF(Src, DstSize, ElementSize);
Src = _Vector_SToF(DstSize, ElementSize, Src);
OrderedNode *Dest{};
if constexpr (Widen) {
@@ -1109,14 +1150,14 @@ void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) {
size_t Size = GetDstSize(Op);
// Always narrows
Src = _Vector_FToF(Size, SrcElementSize >> 1, SrcElementSize, Src);
Src = _Vector_FToF(Size, SrcElementSize >> 1, Src, SrcElementSize);
ElementSize >>= 1;
if constexpr (HostRoundingMode) {
Src = _Vector_FToS(Src, Size, ElementSize);
Src = _Vector_FToS(Size, ElementSize, Src);
}
else {
Src = _Vector_FToZS(Src, Size, ElementSize);
Src = _Vector_FToZS(Size, ElementSize, Src);
}
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, -1);
@@ -1135,7 +1176,7 @@ void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *MemDest = _LoadContext(GPRSize, GPROffset(X86State::REG_RDI), GPRClass);
OrderedNode *MemDest = _LoadContext(GPRSize, GPRClass, GPROffset(X86State::REG_RDI));
const size_t NumElements = Size / 64;
for (size_t Element = 0; Element < NumElements; ++Element) {
@@ -1263,7 +1304,7 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
}
{
auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass);
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMem(GPRClass, 2, Mem, FCW, 2);
}
@@ -1289,7 +1330,7 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
{
// FTW
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
auto FTW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FTW), GPRClass);
auto FTW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FTW));
_StoreMem(GPRClass, 2, MemLocation, FTW, 2);
}
@@ -1338,7 +1379,7 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
// This is implementation dependent
for (unsigned i = 0; i < 8; ++i) {
OrderedNode *MMReg = _LoadContext(16, offsetof(FEXCore::Core::CPUState, mm[i]), FPRClass);
OrderedNode *MMReg = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, mm[i]));
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
_StoreMem(FPRClass, 16, MemLocation, MMReg, 16);
@@ -1346,7 +1387,7 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
unsigned NumRegs = CTX->Config.Is64BitMode ? 16 : 8;
for (unsigned i = 0; i < NumRegs; ++i) {
OrderedNode *XMMReg = _LoadContext(16, offsetof(FEXCore::Core::CPUState, xmm[i]), FPRClass);
OrderedNode *XMMReg = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm[i]));
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
_StoreMem(FPRClass, 16, MemLocation, XMMReg, 16);
@@ -1359,7 +1400,7 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
_F80LoadFCW(NewFCW);
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
{
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
@@ -1384,20 +1425,20 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
// FTW
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
auto NewFTW = _LoadMem(GPRClass, 2, MemLocation, 2);
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FTW), NewFTW);
_StoreContext(2, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, FTW));
}
for (unsigned i = 0; i < 8; ++i) {
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
auto MMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
_StoreContext(FPRClass, 16, offsetof(FEXCore::Core::CPUState, mm[i]), MMReg);
_StoreContext(16, FPRClass, MMReg, offsetof(FEXCore::Core::CPUState, mm[i]));
}
unsigned NumRegs = CTX->Config.Is64BitMode ? 16 : 8;
for (unsigned i = 0; i < NumRegs; ++i) {
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
_StoreContext(FPRClass, 16, offsetof(FEXCore::Core::CPUState, xmm[i]), XMMReg);
_StoreContext(16, FPRClass, XMMReg, offsetof(FEXCore::Core::CPUState, xmm[i]));
}
}
@@ -1422,7 +1463,7 @@ template<size_t ElementSize>
void OpDispatchBuilder::UCOMISxOp(OpcodeArgs) {
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Res = _FCmp(Src1, Src2, ElementSize,
OrderedNode *Res = _FCmp(ElementSize, Src1, Src2,
(1 << FCMP_FLAG_EQ) |
(1 << FCMP_FLAG_LT) |
(1 << FCMP_FLAG_UNORDERED));
@@ -1543,8 +1584,8 @@ void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) {
// This instruction is a bit special in that if the source is MMX then it zexts to 128bit
if constexpr (ToXMM) {
Src = _VMov(Src, 16);
_StoreContext(FPRClass, 16, offsetof(FEXCore::Core::CPUState, xmm[Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0][0]), Src);
Src = _VMov(16, Src);
_StoreContext(16, FPRClass, Src, offsetof(FEXCore::Core::CPUState, xmm[Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0][0]));
}
else {
// This is simple, just store the result
@@ -1633,11 +1674,151 @@ void OpDispatchBuilder::ADDSUBPOp(OpcodeArgs) {
StoreResult(FPRClass, Op, ResAdd, -1);
}
void OpDispatchBuilder::PFNACCOp(OpcodeArgs) {
auto Size = GetSrcSize(Op);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *ResSubSrc{};
OrderedNode *ResSubDest{};
auto UpperSubDest = _VExtractElement(Size, 4, Dest, 1);
auto UpperSubSrc = _VExtractElement(Size, 4, Src, 1);
ResSubDest = _VFSub(4, 4, Dest, UpperSubDest);
ResSubSrc = _VFSub(4, 4, Src, UpperSubSrc);
auto Result = _VInsElement(8, 4, 1, 0, ResSubDest, ResSubSrc);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::PFPNACCOp(OpcodeArgs) {
auto Size = GetSrcSize(Op);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *ResAdd{};
OrderedNode *ResSub{};
auto UpperSubDest = _VExtractElement(Size, 4, Dest, 1);
ResSub = _VFSub(4, 4, Dest, UpperSubDest);
ResAdd = _VFAddP(Size, 4, Src, Src);
auto Result = _VInsElement(8, 4, 1, 0, ResSub, ResAdd);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::PSWAPDOp(OpcodeArgs) {
auto Size = GetSrcSize(Op);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
auto Result = _VRev64(Size, 4, Src);
StoreResult(FPRClass, Op, Result, -1);
}
template
void OpDispatchBuilder::ADDSUBPOp<4>(OpcodeArgs);
template
void OpDispatchBuilder::ADDSUBPOp<8>(OpcodeArgs);
void OpDispatchBuilder::PI2FWOp(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
size_t Size = GetDstSize(Op);
// We now need to transpose the lower 16-bits of each element together
// Only needing to move the upper element down in this case
Src = _VInsElement(Size, 2, 1, 2, Src, Src);
// Now we need to sign extend the 16bit value to 32-bit
Src = _VSXTL(Size, 2, Src);
// int32_t to float
Src = _Vector_SToF(Size, 4, Src);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, -1);
}
void OpDispatchBuilder::PF2IWOp(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
size_t Size = GetDstSize(Op);
// Float to int32_t
Src = _Vector_FToZS(Size, 4, Src);
// We now need to transpose the lower 16-bits of each element together
// Only needing to move the upper element down in this case
Src = _VInsElement(Size, 2, 1, 2, Src, Src);
// Now we need to sign extend the 16bit value to 32-bit
Src = _VSXTL(Size, 2, Src);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, -1);
}
void OpDispatchBuilder::PMULHRWOp(OpcodeArgs) {
auto Size = GetSrcSize(Op);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Res{};
// Implementation is more efficient for 8byte registers
// Multiplies 4 16bit values in to 4 32bit values
Res = _VSMull(Size * 2, 2, Dest, Src);
//TODO: We could remove this VCastFromGOR + VInsGPR pair if we had a VDUPFromGPR instruction that maps directly to AArch64.
auto M = _Constant(0x0000'8000'0000'8000ULL);
OrderedNode *VConstant = _VCastFromGPR(16, 8, M);
VConstant = _VInsGPR(16, 8, 1, VConstant, M);
Res = _VAdd(Size * 2, 4, Res, VConstant);
// Now shift and narrow to convert 32-bit values to 16bit, storing the top 16bits
Res = _VUShrNI(Size * 2, 4, Res, 16);
StoreResult(FPRClass, Op, Res, -1);
}
template<uint8_t CompType>
void OpDispatchBuilder::VPFCMPOp(OpcodeArgs) {
auto Size = GetSrcSize(Op);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetDstSize(Op), Op->Flags, -1);
OrderedNode *Result{};
// This maps 1:1 to an AArch64 NEON Op
//auto ALUOp = _VCMPGT(Size, 4, Dest, Src);
LogMan::Msg::DFmt("CompType: {} Size: {}", CompType, Size);
switch (CompType) {
case 0x00: // EQ
Result = _VFCMPEQ(Size, 4, Dest, Src);
break;
case 0x01: // GE(Swapped operand)
Result = _VFCMPLE(Size, 4, Src, Dest);
break;
case 0x02: // GT
Result = _VFCMPGT(Size, 4, Dest, Src);
break;
default:
LOGMAN_MSG_A_FMT("Unknown Comparison type: {}", CompType);
break;
}
StoreResult(FPRClass, Op, Result, -1);
ShouldDump = true;
}
template
void OpDispatchBuilder::VPFCMPOp<0>(OpcodeArgs);
template
void OpDispatchBuilder::VPFCMPOp<1>(OpcodeArgs);
template
void OpDispatchBuilder::VPFCMPOp<2>(OpcodeArgs);
void OpDispatchBuilder::PMADDWD(OpcodeArgs) {
// This is a pretty curious operation
// Does two MADD operations across 4 16bit signed integers and accumulates to 32bit integers in the destination
@@ -1790,7 +1971,7 @@ void OpDispatchBuilder::PMULHRSW(OpcodeArgs) {
// Implementation is more efficient for 8byte registers
Res = _VSMull(Size * 2, 2, Dest, Src);
Res = _VSShrI(Size * 2, 4, Res, 14);
auto OneVector = _VectorImm(1, Size * 2, 4);
auto OneVector = _VectorImm(Size * 2, 4, 1);
Res = _VAdd(Size * 2, 4, Res, OneVector);
Res = _VUShrNI(Size * 2, 4, Res, 1);
}
@@ -1804,7 +1985,7 @@ void OpDispatchBuilder::PMULHRSW(OpcodeArgs) {
ResultLow = _VSShrI(Size, 4, ResultLow, 14);
ResultHigh = _VSShrI(Size, 4, ResultHigh, 14);
auto OneVector = _VectorImm(1, Size, 4);
auto OneVector = _VectorImm(Size, 4, 1);
ResultLow = _VAdd(Size, 4, ResultLow, OneVector);
ResultHigh = _VAdd(Size, 4, ResultHigh, OneVector);
@@ -2059,10 +2240,10 @@ void OpDispatchBuilder::ExtendVectorElements(OpcodeArgs) {
CurrentElementSize != DstElementSize;
CurrentElementSize <<= 1) {
if constexpr (Signed) {
Result = _VSXTL(Result, Size, CurrentElementSize);
Result = _VSXTL(Size, CurrentElementSize, Result);
}
else {
Result = _VUXTL(Result, Size, CurrentElementSize);
Result = _VUXTL(Size, CurrentElementSize, Result);
}
}
StoreResult(FPRClass, Op, Result, -1);
@@ -2116,7 +2297,7 @@ void OpDispatchBuilder::VectorRound(OpcodeArgs) {
FEXCore::IR::Round_Host,
};
Src = _Vector_FToI(Src, SourceModes[(RoundControlSource << 2) | RoundControl], Size, ElementSize);
Src = _Vector_FToI(Size, ElementSize, Src, SourceModes[(RoundControlSource << 2) | RoundControl]);
if constexpr (Scalar) {
// Insert the lower bits
@@ -2170,13 +2351,14 @@ void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
// The mask is hardcoded to be xmm0 in this instruction
OrderedNode *Mask = _LoadContext(16, offsetof(FEXCore::Core::CPUState, xmm[0]), FPRClass);
OrderedNode *Mask = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm[0]));
// Each element is selected by the high bit of that element size
// Dest[ElementIdx] = Xmm0[ElementIndex][HighBit] ? Src : Dest;
//
// To emulate this on AArch64
// Arithmetic shift right by the element size, then use BSL to select the registers
Mask = _VSShrI(Size, ElementSize, Mask, (ElementSize * 8) - 1);
auto Result = _VBSL(Mask, Src, Dest);
StoreResult(FPRClass, Op, Result, -1);
@@ -2197,8 +2379,8 @@ void OpDispatchBuilder::PTestOp(OpcodeArgs) {
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Test1 = _VAnd(Dest, Src, Size, 1);
OrderedNode *Test2 = _VBic(Src, Dest, Size, 1);
OrderedNode *Test1 = _VAnd(Size, 1, Dest, Src);
OrderedNode *Test2 = _VBic(Size, 1, Src, Dest);
Test1 = _VPopcount(Size, 1, Test1);
Test2 = _VPopcount(Size, 1, Test2);
@@ -2260,10 +2442,10 @@ void OpDispatchBuilder::PHMINPOSUWOp(OpcodeArgs) {
}
// Insert the minimum in to bits [15:0]
OrderedNode *Result = _VMov(Min, 2);
OrderedNode *Result = _VMov(2, Min);
// Insert position in to bits [18:16]
Result = _VInsGPR(16, 2, Result, Pos, 1);
Result = _VInsGPR(16, 2, 1, Result, Pos);
StoreResult(FPRClass, Op, Result, -1);
}
@@ -25,12 +25,12 @@ class OrderedNode;
OrderedNode *OpDispatchBuilder::GetX87Top() {
// Yes, we are storing 3 bits in a single flag register.
// Deal with it
return _LoadContext(1, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC, GPRClass);
return _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
}
void OpDispatchBuilder::SetX87TopTag(OrderedNode *Value, X87Tag Tag) {
// if we are popping then we must first mark this location as empty
auto FTW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FTW), GPRClass);
auto FTW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FTW));
OrderedNode *Mask = _Constant(0b11);
auto TopOffset = _Lshl(Value, _Constant(1));
Mask = _Lshl(Mask, TopOffset);
@@ -40,11 +40,11 @@ void OpDispatchBuilder::SetX87TopTag(OrderedNode *Value, X87Tag Tag) {
NewFTW = _Or(NewFTW, TagVal);
}
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FTW), NewFTW);
_StoreContext(2, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, FTW));
}
OrderedNode *OpDispatchBuilder::GetX87FTW(OrderedNode *Value) {
auto FTW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FTW), GPRClass);
auto FTW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FTW));
OrderedNode *Mask = _Constant(0b11);
auto TopOffset = _Lshl(Value, _Constant(1));
auto NewFTW = _Lshr(FTW, TopOffset);
@@ -52,7 +52,7 @@ OrderedNode *OpDispatchBuilder::GetX87FTW(OrderedNode *Value) {
}
void OpDispatchBuilder::SetX87Top(OrderedNode *Value) {
_StoreContext(GPRClass, 1, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC, Value);
_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
}
template<size_t width>
@@ -136,7 +136,7 @@ void OpDispatchBuilder::FLD_Const(OpcodeArgs) {
auto low = _Constant(Lower);
auto high = _Constant(Upper);
OrderedNode *data = _VCastFromGPR(16, 8, low);
data = _VInsGPR(16, 8, data, high, 1);
data = _VInsGPR(16, 8, 1, data, high);
// Write to ST[TOP]
_StoreContextIndexed(data, top, 16, MMBaseOffset(), 16, FPRClass);
}
@@ -202,7 +202,7 @@ void OpDispatchBuilder::FST(OpcodeArgs) {
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, data, 10, 1);
}
else if constexpr (width == 32 || width == 64) {
auto result = _F80CVT(data, width / 8);
auto result = _F80CVT(width / 8, data);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, result, width / 8, 1);
}
@@ -228,7 +228,7 @@ void OpDispatchBuilder::FIST(OpcodeArgs) {
auto orig_top = GetX87Top();
OrderedNode *data = _LoadContextIndexed(orig_top, 16, MMBaseOffset(), 16, FPRClass);
data = _F80CVTInt(data, Truncate, Size);
data = _F80CVTInt(Size, data, Truncate);
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, 1);
@@ -547,9 +547,9 @@ void OpDispatchBuilder::FCHS(OpcodeArgs) {
auto low = _Constant(0);
auto high = _Constant(0b1'000'0000'0000'0000ULL);
OrderedNode *data = _VCastFromGPR(16, 8, low);
data = _VInsGPR(16, 8, data, high, 1);
data = _VInsGPR(16, 8, 1, data, high);
auto result = _VXor(a, data, 16, 1);
auto result = _VXor(16, 1, a, data);
// Write to ST[TOP]
_StoreContextIndexed(result, top, 16, MMBaseOffset(), 16, FPRClass);
@@ -562,9 +562,9 @@ void OpDispatchBuilder::FABS(OpcodeArgs) {
auto low = _Constant(~0ULL);
auto high = _Constant(0b0'111'1111'1111'1111ULL);
OrderedNode *data = _VCastFromGPR(16, 8, low);
data = _VInsGPR(16, 8, data, high, 1);
data = _VInsGPR(16, 8, 1, data, high);
auto result = _VAnd(a, data, 16, 1);
auto result = _VAnd(16, 1, a, data);
// Write to ST[TOP]
_StoreContextIndexed(result, top, 16, MMBaseOffset(), 16, FPRClass);
@@ -624,7 +624,7 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
// Init FCW to 0x037
auto NewFCW = _Constant(16, 0x037);
_F80LoadFCW(NewFCW);
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
// Init FSW to 0
SetX87Top(_Constant(0));
@@ -635,7 +635,7 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Constant(0));
// Tags all get set to 0b11
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FTW), _Constant(0xFFFF));
_StoreContext(2, GPRClass, _Constant(0xFFFF), offsetof(FEXCore::Core::CPUState, FTW));
}
template<size_t width, bool Integer, OpDispatchBuilder::FCOMIFlags whichflags, bool poptwice>
@@ -875,7 +875,7 @@ void OpDispatchBuilder::X87FYL2X(OpcodeArgs) {
auto low = _Constant(0x8000'0000'0000'0000ULL);
auto high = _Constant(0b0'011'1111'1111'1111);
OrderedNode *data = _VCastFromGPR(16, 8, low);
data = _VInsGPR(16, 8, data, high, 1);
data = _VInsGPR(16, 8, 1, data, high);
st0 = _F80Add(st0, data);
}
@@ -898,7 +898,7 @@ void OpDispatchBuilder::X87TAN(OpcodeArgs) {
auto low = _Constant(0x8000'0000'0000'0000ULL);
auto high = _Constant(0b0'011'1111'1111'1111ULL);
OrderedNode *data = _VCastFromGPR(16, 8, low);
data = _VInsGPR(16, 8, data, high, 1);
data = _VInsGPR(16, 8, 1, data, high);
// Write to ST[TOP]
_StoreContextIndexed(result, orig_top, 16, MMBaseOffset(), 16, FPRClass);
@@ -928,7 +928,7 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
_F80LoadFCW(NewFCW);
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1));
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
@@ -951,7 +951,7 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
// FTW
OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 2));
auto NewFTW = _LoadMem(GPRClass, Size, MemLocation, Size);
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FTW), NewFTW);
_StoreContext(2, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, FTW));
}
}
@@ -980,7 +980,7 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
Mem = AppendSegmentOffset(Mem, Op->Flags);
{
auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass);
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMem(GPRClass, Size, Mem, FCW, Size);
}
@@ -1008,7 +1008,7 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
{
// FTW
OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 2));
auto FTW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FTW), GPRClass);
auto FTW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FTW));
_StoreMem(GPRClass, Size, MemLocation, FTW, Size);
}
@@ -1040,11 +1040,11 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
_F80LoadFCW(NewFCW);
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
}
void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass);
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
StoreResult(GPRClass, Op, FCW, -1);
}
@@ -1111,7 +1111,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
OrderedNode *Top = GetX87Top();
{
auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass);
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
_StoreMem(GPRClass, Size, Mem, FCW, Size);
}
@@ -1138,7 +1138,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
{
// FTW
OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 2));
auto FTW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FTW), GPRClass);
auto FTW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FTW));
_StoreMem(GPRClass, Size, MemLocation, FTW, Size);
}
@@ -1199,7 +1199,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
_F80LoadFCW(NewFCW);
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1));
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
@@ -1222,7 +1222,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
// FTW
OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 2));
auto NewFTW = _LoadMem(GPRClass, Size, MemLocation, Size);
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FTW), NewFTW);
_StoreContext(2, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, FTW));
}
OrderedNode *ST0Location = _Add(Mem, _Constant(Size * 7));
@@ -1234,7 +1234,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
auto low = _Constant(~0ULL);
auto high = _Constant(0xFFFF);
OrderedNode *Mask = _VCastFromGPR(16, 8, low);
Mask = _VInsGPR(16, 8, Mask, high, 1);
Mask = _VInsGPR(16, 8, 1, Mask, high);
for (int i = 0; i < 7; ++i) {
OrderedNode *Reg = _LoadMem(FPRClass, 16, ST0Location, 1);
@@ -1362,7 +1362,7 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
SrcCond = _Sbfe(1, 0, SrcCond);
OrderedNode *VecCond = _VCastFromGPR(16, 8, SrcCond);
VecCond = _VInsGPR(16, 8, VecCond, SrcCond, 1);
VecCond = _VInsGPR(16, 8, 1, VecCond, SrcCond);
auto top = GetX87Top();
OrderedNode* arg;
@@ -1383,7 +1383,7 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
void OpDispatchBuilder::X87EMMS(OpcodeArgs) {
// Tags all get set to 0b11
_StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FTW), _Constant(0xFFFF));
_StoreContext(2, GPRClass, _Constant(0xFFFF), offsetof(FEXCore::Core::CPUState, FTW));
}
void OpDispatchBuilder::X87FFREE(OpcodeArgs) {
@@ -15,37 +15,42 @@ using namespace InstFlags;
void InitializeDDDTables() {
static constexpr U8U8InfoStruct DDDNowOpTable[] = {
{0x0C, 1, X86InstInfo{"PI2FW", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x0D, 1, X86InstInfo{"PI2FD", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x1C, 1, X86InstInfo{"PF2IW", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x1D, 1, X86InstInfo{"PF2ID", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x0C, 1, X86InstInfo{"PI2FW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x0D, 1, X86InstInfo{"PI2FD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x1C, 1, X86InstInfo{"PF2IW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x1D, 1, X86InstInfo{"PF2ID", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x8A, 1, X86InstInfo{"PFNACC", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x8E, 1, X86InstInfo{"PFPNACC", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
// Inverse 3DNow! These two instructions are Geode product line specific
// No CPUID for these, you're expected to read ID_CONFIG_MSR (1250h) bit 1
{0x86, 1, X86InstInfo{"PFRCPV", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x87, 1, X86InstInfo{"PFRSQRTV", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x9A, 1, X86InstInfo{"PFSUB", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x9E, 1, X86InstInfo{"PFADD", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x8A, 1, X86InstInfo{"PFNACC", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x8E, 1, X86InstInfo{"PFPNACC", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xAA, 1, X86InstInfo{"PFSUBR", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xAE, 1, X86InstInfo{"PFACC", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x90, 1, X86InstInfo{"PFCMPGE", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x94, 1, X86InstInfo{"PFMIN", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x96, 1, X86InstInfo{"PFRCP", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x97, 1, X86InstInfo{"PFRSQRT", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xBB, 1, X86InstInfo{"PSWAPD", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xBF, 1, X86InstInfo{"PAVGUSB", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x9A, 1, X86InstInfo{"PFSUB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x9E, 1, X86InstInfo{"PFADD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x90, 1, X86InstInfo{"PFCMPGE", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x94, 1, X86InstInfo{"PFMIN", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x96, 1, X86InstInfo{"PFRCP", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0x97, 1, X86InstInfo{"PFRSQRT", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xA0, 1, X86InstInfo{"PFCMPGT", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xA4, 1, X86InstInfo{"PFMAX", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xA6, 1, X86InstInfo{"PFRCPIT1", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xA7, 1, X86InstInfo{"PFRSQIT1", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xA0, 1, X86InstInfo{"PFCMPGT", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xA4, 1, X86InstInfo{"PFMAX", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xA6, 1, X86InstInfo{"PFRCPIT1", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xA7, 1, X86InstInfo{"PFRSQIT1", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xAA, 1, X86InstInfo{"PFSUBR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xAE, 1, X86InstInfo{"PFACC", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xB0, 1, X86InstInfo{"PFCMPEQ", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xB4, 1, X86InstInfo{"PFMUL", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xB6, 1, X86InstInfo{"PFRCPIT2", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xB7, 1, X86InstInfo{"PMULHRW", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
{0xB0, 1, X86InstInfo{"PFCMPEQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xB4, 1, X86InstInfo{"PFMUL", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xB6, 1, X86InstInfo{"PFRCPIT2", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xB7, 1, X86InstInfo{"PMULHRW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xBB, 1, X86InstInfo{"PSWAPD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0xBF, 1, X86InstInfo{"PAVGUSB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
};
GenerateTable(&DDDNowOps.at(0), DDDNowOpTable, std::size(DDDNowOpTable));
@@ -144,38 +144,38 @@ void InitializeSecondaryGroupTables() {
// AMD documentation is a bit broken for Group 9
// Claims the entire group has n/a applied for the prefix (Implies that the prefix is ignored)
// RDRAND/RDSEED only work with no prefix
// RDRAND/RDSEED only work with no prefix (Other than 66h)
// CMPXCHG8B/16B works with all prefixes
// Tooling fails to decode CMPXCHG with prefix
{OPD(TYPE_GROUP_9, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 1), 1, X86InstInfo{"CMPXCHG16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 6), 1, X86InstInfo{"RDRAND", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 7), 1, X86InstInfo{"RDSEED", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 6), 1, X86InstInfo{"RDRAND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_NONE, 7), 1, X86InstInfo{"RDSEED", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 1), 1, X86InstInfo{"CMPXCHG16B", TYPE_INVALID, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"RDPID", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 1), 1, X86InstInfo{"CMPXCHG16B", TYPE_INVALID, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 6), 1, X86InstInfo{"RDRAND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_66, 7), 1, X86InstInfo{"RDSEED", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F2, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F2, 1), 1, X86InstInfo{"CMPXCHG16B", TYPE_INVALID, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F2, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F2, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F2, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{OPD(TYPE_GROUP_9, PF_F2, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
@@ -120,9 +120,9 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
{0x6F, 1, X86InstInfo{"MOVQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x70, 1, X86InstInfo{"PSHUFW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
{0x71, 1, X86InstInfo{"", TYPE_GROUP_12, FLAGS_NONE, 0, nullptr}},
{0x72, 1, X86InstInfo{"", TYPE_GROUP_13, FLAGS_NONE, 0, nullptr}},
{0x73, 1, X86InstInfo{"", TYPE_GROUP_14, FLAGS_NONE, 0, nullptr}},
{0x71, 1, X86InstInfo{"", TYPE_GROUP_12, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x72, 1, X86InstInfo{"", TYPE_GROUP_13, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x73, 1, X86InstInfo{"", TYPE_GROUP_14, FLAGS_NO_OVERLAY, 0, nullptr}},
{0x74, 1, X86InstInfo{"PCMPEQB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x75, 1, X86InstInfo{"PCMPEQW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
{0x76, 1, X86InstInfo{"PCMPEQD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
@@ -186,8 +186,8 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
{0xB6, 1, X86InstInfo{"MOVZX", TYPE_INST, GenFlagsSrcSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xB7, 1, X86InstInfo{"MOVZX", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xB8, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
{0xB9, 1, X86InstInfo{"", TYPE_GROUP_10, FLAGS_NONE, 0, nullptr}},
{0xBA, 1, X86InstInfo{"", TYPE_GROUP_8, FLAGS_NONE, 0, nullptr}},
{0xB9, 1, X86InstInfo{"", TYPE_GROUP_10, FLAGS_NO_OVERLAY, 0, nullptr}},
{0xBA, 1, X86InstInfo{"", TYPE_GROUP_8, FLAGS_NO_OVERLAY, 0, nullptr}},
{0xBB, 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xBC, 1, X86InstInfo{"BSF", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY66, 0, nullptr}},
{0xBD, 1, X86InstInfo{"BSR", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY66, 0, nullptr}},
@@ -201,7 +201,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
{0xC4, 1, X86InstInfo{"PINSRW", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX | FLAGS_SF_SRC_GPR, 1, nullptr}},
{0xC5, 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
{0xC6, 1, X86InstInfo{"SHUFPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
{0xC7, 1, X86InstInfo{"", TYPE_GROUP_9, FLAGS_NONE, 0, nullptr}},
{0xC7, 1, X86InstInfo{"", TYPE_GROUP_9, FLAGS_NO_OVERLAY, 0, nullptr}},
{0xC8, 8, X86InstInfo{"BSWAP", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_NO_OVERLAY, 0, nullptr}},
{0xD0, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
File diff suppressed because it is too large. Load diff
+13
View File
@@ -171,6 +171,19 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const*
}
}
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::SyscallFlags Arg) {
switch (Arg) {
case FEXCore::IR::SyscallFlags::DEFAULT: *out << "Default"; break;
case FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH: *out << "Optimize Through"; break;
case FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY: *out << "No Sync State on Entry"; break;
case FEXCore::IR::SyscallFlags::NORETURN: *out << "No Return"; break;
case FEXCore::IR::SyscallFlags::NOSIDEEFFECTS: *out << "No Side Effects"; break;
default: *out << "<Unknown Round Type>"; break;
}
}
void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData) {
auto HeaderOp = IR->GetHeader();
+56
View File
@@ -16,6 +16,62 @@ $end_info$
#include <vector>
namespace FEXCore::IR {
FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(OrderedNode *Node) {
auto Class = GetOpRegClass(Node);
switch (Class) {
case GPRClass:
case GPRPairClass:
case FPRClass:
case GPRFixedClass:
case FPRFixedClass:
case InvalidClass:
return Class;
default: break;
}
// Complex case, needs to be handled on an op by op basis
uintptr_t DataBegin = DualListData.DataBegin();
FEXCore::IR::IROp_Header *IROp = Node->Op(DataBegin);
switch (IROp->Op) {
case IROps::OP_LOADREGISTER: {
auto Op = IROp->C<IROp_LoadRegister>();
return Op->Class;
break;
}
case IROps::OP_LOADCONTEXT: {
auto Op = IROp->C<IROp_LoadContext>();
return Op->Class;
break;
}
case IROps::OP_LOADCONTEXTINDEXED: {
auto Op = IROp->C<IROp_LoadContextIndexed>();
return Op->Class;
break;
}
case IROps::OP_FILLREGISTER: {
auto Op = IROp->C<IROp_FillRegister>();
return Op->Class;
break;
}
case IROps::OP_LOADMEM: {
auto Op = IROp->C<IROp_LoadMem>();
return Op->Class;
break;
}
case IROps::OP_LOADMEMTSO: {
auto Op = IROp->C<IROp_LoadMemTSO>();
return Op->Class;
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled op type: {} {} in argument class validation", IROp->Op, GetOpName(Node));
break;
}
return InvalidClass;
}
void IREmitter::ResetWorkingList() {
DualListData.Reset();
CodeBlocks.clear();
+13 -5
View File
@@ -170,7 +170,7 @@ class IRParser: public FEXCore::IR::IREmitter {
}
Result.data[i] = high * 16 + low;
}
return {DecodeFailure::DECODE_OKAY, Result};
}
@@ -351,10 +351,6 @@ class IRParser: public FEXCore::IR::IREmitter {
bool Loaded = false;
#define IROP_PARSER_ALLOCATE_HELPERS
#include <FEXCore/IR/IRDefines.inc>
bool Parse() {
const auto CheckPrintError = [&](const LineDefinition &Def, DecodeFailure Failure) -> bool {
if (Failure != DecodeFailure::DECODE_OKAY) {
@@ -367,6 +363,18 @@ class IRParser: public FEXCore::IR::IREmitter {
return true;
};
const auto CheckPrintErrorArg = [&](const LineDefinition &Def, DecodeFailure Failure, size_t Arg) -> bool {
if (Failure != DecodeFailure::DECODE_OKAY) {
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
LogMan::Msg::EFmt("Argument Number {}: {}", Arg + 1, Def.Args[Arg]);
LogMan::Msg::EFmt("Value Couldn't be decoded due to {}", DecodeErrorToString(Failure));
return false;
}
return true;
};
// String parse every line for our definitions
for (size_t i = 0; i < Lines.size(); ++i) {
std::string Line = Lines[i];
+39 -28
View File
@@ -286,7 +286,7 @@ void ConstProp::FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR)
if (IROp->Op == OP_GETHOSTFLAG) {
auto ghf = IROp->CW<IR::IROp_GetHostFlag>();
auto fcmp = IREmit->GetOpHeader(ghf->GPR)->CW<IR::IROp_FCmp>();
auto fcmp = IREmit->GetOpHeader(ghf->Value)->CW<IR::IROp_FCmp>();
LOGMAN_THROW_A_FMT(fcmp->Header.Op == OP_FCMP || fcmp->Header.Op == OP_F80CMP, "Unexpected OP_GETHOSTFLAG source");
if(fcmp->Header.Op == OP_FCMP) {
fcmp->Flags |= 1 << ghf->Flag;
@@ -301,18 +301,29 @@ void ConstProp::LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListV
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
if (IROp->Op == OP_LOADMEM || IROp->Op == OP_STOREMEM) {
size_t AddrIndex = 0;
size_t OffsetIndex = 0;
if (IROp->Op == OP_LOADMEM) {
AddrIndex = IR::IROp_LoadMem::Addr_Index;
OffsetIndex = IR::IROp_LoadMem::Offset_Index;
}
else {
AddrIndex = IR::IROp_StoreMem::Addr_Index;
OffsetIndex = IR::IROp_StoreMem::Offset_Index;
}
uint64_t Addr;
if (IREmit->IsValueConstant(IROp->Args[0], &Addr) && IROp->Args[1].IsInvalid()) {
if (IREmit->IsValueConstant(IROp->Args[AddrIndex], &Addr) && IROp->Args[OffsetIndex].IsInvalid()) {
for (auto& Const: AddressgenConsts) {
if ((Addr - Const.second) < 65536) {
IREmit->ReplaceNodeArgument(CodeNode, 0, Const.first);
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_Constant(Addr - Const.second));
IREmit->ReplaceNodeArgument(CodeNode, AddrIndex, Const.first);
IREmit->ReplaceNodeArgument(CodeNode, OffsetIndex, IREmit->_Constant(Addr - Const.second));
goto doneOp;
}
}
AddressgenConsts[IREmit->UnwrapNode(IROp->Args[0])] = Addr;
AddressgenConsts[IREmit->UnwrapNode(IROp->Args[AddrIndex])] = Addr;
}
doneOp:
;
@@ -374,8 +385,8 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
auto Op = IROp->C<IR::IROp_Bfe>();
// Is this value already BFE'd?
if (IsBfeAlreadyDone(IREmit, IROp->Args[0], Op->Width)) {
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(IROp->Args[0]));
if (IsBfeAlreadyDone(IREmit, Op->Src, Op->Width)) {
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Src));
//printf("Removed BFE once \n");
break;
}
@@ -383,7 +394,7 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
// Is this value already ZEXT'd?
if (Op->lsb == 0) {
//LoadMem, LoadMemTSO & LoadContext ZExt
auto source = IROp->Args[0];
auto source = Op->Src;
auto sourceHeader = IREmit->GetOpHeader(source);
if (Op->Width >= (sourceHeader->Size*8) &&
@@ -401,10 +412,10 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
imm = (imm-1) *2 + 1;
imm <<= Op->lsb;
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[0], imm);
auto newArg = RemoveUselessMasking(IREmit, Op->Src, imm);
if (newArg.ID() != IROp->Args[0].ID()) {
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(newArg));
if (newArg.ID() != Op->Src.ID()) {
IREmit->ReplaceNodeArgument(CodeNode, Op->Src_Index, IREmit->UnwrapNode(newArg));
Changed = true;
}
break;
@@ -418,10 +429,10 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C
imm = (imm-1) *2 + 1;
imm <<= Op->lsb;
auto newArg = RemoveUselessMasking(IREmit, IROp->Args[0], imm);
auto newArg = RemoveUselessMasking(IREmit, Op->Src, imm);
if (newArg.ID() != IROp->Args[0].ID()) {
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(newArg));
if (newArg.ID() != Op->Src.ID()) {
IREmit->ReplaceNodeArgument(CodeNode, Op->Src_Index, IREmit->UnwrapNode(newArg));
Changed = true;
}
break;
@@ -528,15 +539,15 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
case OP_LOADMEM: {
auto Op = IROp->CW<IR::IROp_LoadMem>();
auto AddressHeader = IREmit->GetOpHeader(Op->Header.Args[0]);
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, 0, Arg0);
IREmit->ReplaceNodeArgument(CodeNode, 1, Arg1);
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
Changed = true;
}
@@ -545,15 +556,15 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
case OP_STOREMEM: {
auto Op = IROp->CW<IR::IROp_StoreMem>();
auto AddressHeader = IREmit->GetOpHeader(Op->Header.Args[0]);
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
Op->OffsetType = OffsetType;
Op->OffsetScale = OffsetScale;
IREmit->ReplaceNodeArgument(CodeNode, 0, Arg0);
IREmit->ReplaceNodeArgument(CodeNode, 2, Arg1);
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
Changed = true;
}
@@ -707,7 +718,7 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
case OP_BFE: {
auto Op = IROp->C<IR::IROp_Bfe>();
uint64_t Constant;
if (IROp->Size <= 8 && IREmit->IsValueConstant(Op->Header.Args[0], &Constant)) {
if (IROp->Size <= 8 && IREmit->IsValueConstant(Op->Src, &Constant)) {
uint64_t SourceMask = (1ULL << Op->Width) - 1;
if (Op->Width == 64)
SourceMask = ~0ULL;
@@ -897,7 +908,7 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
uint64_t Constant{};
if (IREmit->IsValueConstant(Op->NewRIP, &Constant)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineConstant(Constant));
@@ -940,11 +951,11 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
auto Op = IROp->CW<IR::IROp_LoadMem>();
uint64_t Constant2{};
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) {
if (IsImmMemory(Constant2, IROp->Size)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset));
IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->_InlineConstant(Constant2));
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, IREmit->_InlineConstant(Constant2));
Changed = true;
}
@@ -957,11 +968,11 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
auto Op = IROp->CW<IR::IROp_StoreMem>();
uint64_t Constant2{};
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Header.Args[2], &Constant2)) {
if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) {
if (IsImmMemory(Constant2, IROp->Size)) {
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[2]));
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset));
IREmit->ReplaceNodeArgument(CodeNode, 2, IREmit->_InlineConstant(Constant2));
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, IREmit->_InlineConstant(Constant2));
Changed = true;
}
@@ -37,9 +37,27 @@ bool DeadCodeElimination::Run(IREmitter *IREmit) {
while (1) {
auto [CodeNode, IROp] = CodeLast();
bool HasSideEffects = IR::HasSideEffects(IROp->Op);
if (IROp->Op == OP_SYSCALL ||
IROp->Op == OP_INLINESYSCALL) {
FEXCore::IR::SyscallFlags Flags{};
if (IROp->Op == OP_SYSCALL) {
auto Op = IROp->C<IR::IROp_Syscall>();
Flags = Op->Flags;
}
else {
auto Op = IROp->C<IR::IROp_InlineSyscall>();
Flags = Op->Flags;
}
if ((Flags & FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) == FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) {
HasSideEffects = false;
}
}
// Skip over anything that has side effects
// Use count tracking can't safely remove anything with side effects
if (!IR::HasSideEffects(IROp->Op)) {
if (!HasSideEffects) {
if (CodeNode->GetUses() == 0) {
NumRemoved++;
IREmit->Remove(CodeNode);
@@ -561,7 +561,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
// the vector element
IREmit->SetWriteCursor(CodeNode);
// zext to size
LastNode = IREmit->_VMov(LastNode, IROp->Size);
LastNode = IREmit->_VMov(IROp->Size, LastNode);
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
@@ -577,7 +577,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
IROp->Size < IREmit->GetOpSize(LastNode)) {
IREmit->SetWriteCursor(CodeNode);
// trucate to size
LastNode = IREmit->_VMov(LastNode, IROp->Size);
LastNode = IREmit->_VMov(IROp->Size, LastNode);
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
Changed = true;
@@ -585,7 +585,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
IROp->Size > IREmit->GetOpSize(LastNode)) {
IREmit->SetWriteCursor(CodeNode);
// zext to size
LastNode = IREmit->_VMov(LastNode, IROp->Size);
LastNode = IREmit->_VMov(IROp->Size, LastNode);
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
@@ -661,10 +661,25 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
Changed = true;
}
}
else if (IROp->Op == OP_SYSCALL ||
IROp->Op == OP_INLINESYSCALL) {
FEXCore::IR::SyscallFlags Flags{};
if (IROp->Op == OP_SYSCALL) {
auto Op = IROp->C<IR::IROp_Syscall>();
Flags = Op->Flags;
}
else {
auto Op = IROp->C<IR::IROp_InlineSyscall>();
Flags = Op->Flags;
}
if ((Flags & FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH) != FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH) {
// We can't track through these
ResetClassificationAccesses(&LocalInfo);
}
}
else if (IROp->Op == OP_STORECONTEXTINDEXED ||
IROp->Op == OP_LOADCONTEXTINDEXED ||
IROp->Op == OP_SYSCALL ||
IROp->Op == OP_INLINESYSCALL ||
IROp->Op == OP_BREAK) {
// We can't track through these
ResetClassificationAccesses(&LocalInfo);
@@ -25,6 +25,7 @@ $end_info$
#include <strings.h>
#include <unordered_map>
#include <unordered_set>
#include <sys/user.h>
#include <utility>
#include <vector>
@@ -61,7 +62,7 @@ namespace {
};
static_assert(sizeof(RegisterNode) == 128 * 4);
constexpr size_t REGISTER_NODES_PER_PAGE = FEXCore::Core::PAGE_SIZE / sizeof(RegisterNode);
constexpr size_t REGISTER_NODES_PER_PAGE = PAGE_SIZE / sizeof(RegisterNode);
struct RegisterSet {
std::vector<RegisterClass> Classes;
@@ -1498,7 +1499,7 @@ namespace {
auto IROp = IR->GetNode(IR->GetNode(CodeBlock->Last)->Header.Previous)->Op(IR->GetData());
if (IROp->Op == OP_JUMP) {
auto Op = IROp->C<IROp_Jump>();
Graph->BlockPredecessors[Op->Target.ID()].insert(IR->GetID(BlockNode));
Graph->BlockPredecessors[Op->TargetBlock.ID()].insert(IR->GetID(BlockNode));
} else if (IROp->Op == OP_CONDJUMP) {
auto Op = IROp->C<IROp_CondJump>();
Graph->BlockPredecessors[Op->TrueBlock.ID()].insert(IR->GetID(BlockNode));
@@ -26,14 +26,20 @@ bool SyscallOptimization::Run(IREmitter *IREmit) {
bool Changed = false;
auto CurrentIR = IREmit->ViewIR();
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
if (IROp->Op == FEXCore::IR::OP_SYSCALL) {
auto Op = IROp->CW<IR::IROp_Syscall>();
// Is the first argument a constant?
uint64_t Constant;
if (IREmit->IsValueConstant(IROp->Args[0], &Constant)) {
if (IREmit->IsValueConstant(Op->SyscallID, &Constant)) {
auto SyscallDef = Manager->SyscallHandler->GetSyscallABI(Constant);
auto SyscallFlags = Manager->SyscallHandler->GetSyscallFlags(Constant);
// Update the syscall flags
Op->Flags = SyscallFlags;
// XXX: Once we have the ability to do real function calls then we can call directly in to the syscall handler
if (SyscallDef.NumArgs < FEXCore::HLE::SyscallArguments::MAX_ARGS) {
// If the number of args are less than what the IR op supports then we can remove arg usage
@@ -53,7 +59,8 @@ bool SyscallOptimization::Run(IREmitter *IREmit) {
CurrentIR.GetNode(IROp->Args[4]),
CurrentIR.GetNode(IROp->Args[5]),
CurrentIR.GetNode(IROp->Args[6]),
SyscallDef.HostSyscallNumber);
SyscallDef.HostSyscallNumber,
Op->Flags);
// Replace all syscall uses with this inline one
IREmit->ReplaceAllUsesWith(CodeNode, InlineSyscall);
@@ -62,8 +69,9 @@ bool SyscallOptimization::Run(IREmitter *IREmit) {
IREmit->Remove(CodeNode);
}
#endif
Changed = true;
}
Changed = true;
}
}
}
+1
View File
@@ -6,6 +6,7 @@
#include <array>
#include <sys/mman.h>
#include <sys/user.h>
#ifdef ENABLE_JEMALLOC
#include <jemalloc/jemalloc.h>
#endif
+8 -10
View File
@@ -20,13 +20,12 @@
#include <sstream>
#include <sys/mman.h>
#include <sys/utsname.h>
#include <sys/user.h>
#include <type_traits>
#include <utility>
static constexpr uint64_t PAGE_SHIFT = 12;
static constexpr uint64_t PAGE_MASK = (1 << PAGE_SHIFT) - 1;
namespace Alloc::OSAllocator {
class OSAllocator_64Bit final : public Alloc::HostAllocator {
public:
OSAllocator_64Bit();
@@ -38,7 +37,6 @@ namespace Alloc::OSAllocator {
int Munmap(void *addr, size_t length) override;
private:
constexpr static uint64_t PAGE_SIZE = 4096;
// Upper bound is the maximum virtual address space of the host processor
uintptr_t UPPER_BOUND = (1ULL << 57);
@@ -107,8 +105,8 @@ namespace Alloc::OSAllocator {
static_assert(std::is_trivially_copyable<LiveVMARegion>::value, "Needs to be trivially copyable");
static_assert(offsetof(LiveVMARegion, UsedPages) == sizeof(LiveVMARegion), "FlexBitSet needs to be at the end");
using ReservedRegionListType = std::pmr::list<ReservedVMARegion*>;
using LiveRegionListType = std::pmr::list<LiveVMARegion*>;
using ReservedRegionListType = fex_pmr::list<ReservedVMARegion*>;
using LiveRegionListType = fex_pmr::list<LiveVMARegion*>;
ReservedRegionListType *ReservedRegions{};
LiveRegionListType *LiveRegions{};
@@ -162,13 +160,13 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
uint64_t Addr = reinterpret_cast<uint64_t>(addr);
// Addr must be page aligned
if (Addr & PAGE_MASK) {
if (Addr & ~PAGE_MASK) {
return reinterpret_cast<void*>(-EINVAL);
}
// If FD is provided then offset must also be page aligned
if (fd != -1 &&
offset & PAGE_MASK) {
offset & ~PAGE_MASK) {
return reinterpret_cast<void*>(-EINVAL);
}
@@ -430,11 +428,11 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
uint64_t Addr = reinterpret_cast<uint64_t>(addr);
if (Addr & PAGE_MASK) {
if (Addr & ~PAGE_MASK) {
return -EINVAL;
}
if (length & PAGE_MASK) {
if (length & ~PAGE_MASK) {
return -EINVAL;
}
@@ -5,8 +5,6 @@
#include <memory>
#include <sys/types.h>
constexpr static uint64_t PAGE_SIZE = 4096;
namespace Alloc {
// HostAllocator is just a page pased slab allocator
// Similar to mmap and munmap only mapping at the page level
@@ -7,12 +7,26 @@
#include <bitset>
#include <cstddef>
#ifdef TERMUX_BUILD
#ifdef __has_include
#if __has_include(<memory_resource>)
#error Termux <experimental/memory_resource> workaround can be removed
#endif
#endif
#include <experimental/memory_resource>
#include <experimental/list>
namespace fex_pmr = std::experimental::pmr;
#else
#include <memory_resource>
namespace fex_pmr = std::pmr;
#endif
#include <sys/user.h>
#include <mutex>
#include <vector>
namespace Alloc {
class ForwardOnlyIntrusiveArenaAllocator final : public std::pmr::memory_resource {
class ForwardOnlyIntrusiveArenaAllocator final : public fex_pmr::memory_resource {
public:
ForwardOnlyIntrusiveArenaAllocator(void* Ptr, size_t _Size)
: Begin {reinterpret_cast<uintptr_t>(Ptr)}
@@ -54,7 +68,7 @@ namespace Alloc {
// Do nothing
}
bool do_is_equal(const std::pmr::memory_resource& other) const noexcept override {
bool do_is_equal(const fex_pmr::memory_resource& other) const noexcept override {
// Only if the allocator pointers are the same are they equal
if (this == &other) {
return true;
@@ -68,7 +82,7 @@ namespace Alloc {
size_t LastAllocation{};
};
class IntrusiveArenaAllocator final : public std::pmr::memory_resource {
class IntrusiveArenaAllocator final : public fex_pmr::memory_resource {
public:
IntrusiveArenaAllocator(void* Ptr, size_t _Size)
: Begin {reinterpret_cast<uintptr_t>(Ptr)}
@@ -167,7 +181,7 @@ namespace Alloc {
FreePages += FreedPages;
}
bool do_is_equal(const std::pmr::memory_resource& other) const noexcept override {
bool do_is_equal(const fex_pmr::memory_resource& other) const noexcept override {
// Only if the allocator pointers are the same are they equal
if (this == &other) {
return true;
+50
View File
@@ -0,0 +1,50 @@
#pragma once
#include <FEXCore/Utils/LogManager.h>
#include <cstdint>
namespace FEXCore::Utils {
/**
* @brief Casts a class's member function pointer to a raw pointer that we can JIT
*
* Has additional validation to ensure we aren't casting a class member that is invalid
*/
template <typename PointerToMemberType>
class MemberFunctionToPointerCast final {
public:
MemberFunctionToPointerCast(PointerToMemberType Function) {
memcpy(&PMF, &Function, sizeof(PMF));
#ifdef _M_X86_64
// Itanium C++ ABI (https://itanium-cxx-abi.github.io/cxx-abi/abi.html#member-function-pointers)
// Low bit of ptr specifies if this Member function pointer is virtual or not
// Throw an assert if we were trying to cast a virtual member
LOGMAN_THROW_A_FMT((PMF.ptr & 1) == 0, "C++ Pointer-To-Member representation didn't have low bit set to 0. Are you trying to cast a virtual member?");
#elif defined(_M_ARM_64 )
// C++ ABI for the Arm 64-bit Architecture (IHI 0059E)
// 4.2.1 Representation of pointer to member function
// Differs from Itanium specification
LOGMAN_THROW_A_FMT(PMF.adj == 0, "C++ Pointer-To-Member representation didn't have adj == 0. Are you trying to cast a virtual member?");
#else
#error Don't know how to cast Member to function here. Likely just Itanium
#endif
}
uintptr_t GetConvertedPointer() const {
return PMF.ptr;
}
private:
struct PointerToMember {
uintptr_t ptr;
uintptr_t adj;
};
PointerToMember PMF;
// Ensure the representation of PointerToMember matches
static_assert(sizeof(PMF) == sizeof(PointerToMemberType));
};
}
+5 -1
View File
@@ -15,9 +15,13 @@ namespace FEXCore::Telemetry {
static std::array<Value, FEXCore::Telemetry::TelemetryType::TYPE_LAST> TelemetryValues = {{ }};
const std::array<std::string_view, FEXCore::Telemetry::TelemetryType::TYPE_LAST> TelemetryNames {
"64byte Split Locks",
"16Byte Split atomics",
"16byte Split atomics",
"VEX instructions (AVX)",
"EVEX instructions (AVX512)",
"16bit CAS Tear",
"32bit CAS Tear",
"64bit CAS Tear",
"128bit CAS Tear",
};
void Initialize() {
auto DataDirectory = Config::GetDataDirectory();
+1 -1
View File
@@ -31,7 +31,7 @@ namespace FEXCore::Threads {
std::lock_guard lk{DeadStackPoolMutex};
if (DeadStackPool.size() == 0) {
// Nothing in the pool, just allocate
return FEXCore::Allocator::mmap(nullptr, Size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_GROWSDOWN, -1, 0);
return FEXCore::Allocator::mmap(nullptr, Size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
}
// Keep the first item in the stack pool
+120 -7
View File
@@ -32,6 +32,120 @@ namespace FEXCore::Core {
struct InternalThreadState;
enum FallbackHandlerIndex {
OPINDEX_F80LOADFCW = 0,
OPINDEX_F80CVTTO_4,
OPINDEX_F80CVTTO_8,
OPINDEX_F80CVT_4,
OPINDEX_F80CVT_8,
OPINDEX_F80CVTINT_2,
OPINDEX_F80CVTINT_4,
OPINDEX_F80CVTINT_8,
OPINDEX_F80CVTINT_TRUNC2,
OPINDEX_F80CVTINT_TRUNC4,
OPINDEX_F80CVTINT_TRUNC8,
OPINDEX_F80CMP_0,
OPINDEX_F80CMP_1,
OPINDEX_F80CMP_2,
OPINDEX_F80CMP_3,
OPINDEX_F80CMP_4,
OPINDEX_F80CMP_5,
OPINDEX_F80CMP_6,
OPINDEX_F80CMP_7,
OPINDEX_F80CVTTOINT_2,
OPINDEX_F80CVTTOINT_4,
// Unary
OPINDEX_F80ROUND,
OPINDEX_F80F2XM1,
OPINDEX_F80TAN,
OPINDEX_F80SQRT,
OPINDEX_F80SIN,
OPINDEX_F80COS,
OPINDEX_F80XTRACT_EXP,
OPINDEX_F80XTRACT_SIG,
OPINDEX_F80BCDSTORE,
OPINDEX_F80BCDLOAD,
// Binary
OPINDEX_F80ADD,
OPINDEX_F80SUB,
OPINDEX_F80MUL,
OPINDEX_F80DIV,
OPINDEX_F80FYL2X,
OPINDEX_F80ATAN,
OPINDEX_F80FPREM1,
OPINDEX_F80FPREM,
OPINDEX_F80SCALE,
// Maximum
OPINDEX_MAX,
};
union JITPointers {
struct {
// Process specific
uint64_t LUDIV{};
uint64_t LDIV{};
uint64_t LUREM{};
uint64_t LREM{};
uint64_t PrintValue{};
uint64_t PrintVectorValue{};
uint64_t RemoveCodeEntryFromJIT{};
uint64_t CPUIDObj{};
uint64_t CPUIDFunction{};
uint64_t SyscallHandlerObj{};
uint64_t SyscallHandlerFunc{};
uint64_t FallbackHandlerPointers[FallbackHandlerIndex::OPINDEX_MAX];
// Thread Specific
uint64_t SignalHandlerRefCountPointer{};
/**
* @name Dispatcher pointers
* @{ */
uint64_t DispatcherLoopTop{};
uint64_t DispatcherLoopTopFillSRA{};
uint64_t ThreadStopHandlerSpillSRA{};
uint64_t ThreadPauseHandlerSpillSRA{};
uint64_t UnimplementedInstructionHandler{};
uint64_t OverflowExceptionHandler{};
uint64_t SignalReturnHandler{};
uint64_t L1Pointer{};
/** @} */
} AArch64;
struct {
// Process specific
uint64_t PrintValue{};
uint64_t PrintVectorValue{};
uint64_t RemoveCodeEntryFromJIT{};
uint64_t CPUIDObj{};
uint64_t CPUIDFunction{};
uint64_t SyscallHandlerObj{};
uint64_t SyscallHandlerFunc{};
uint64_t FallbackHandlerPointers[FallbackHandlerIndex::OPINDEX_MAX];
// Thread Specific
uint64_t SignalHandlerRefCountPointer{};
/**
* @name Dispatcher pointers
* @{ */
uint64_t DispatcherLoopTop{};
uint64_t DispatcherLoopTopFillSRA{};
uint64_t ThreadStopHandler{};
uint64_t ThreadPauseHandler{};
uint64_t UnimplementedInstructionHandler{};
uint64_t OverflowExceptionHandler{};
uint64_t SignalReturnHandler{};
uint64_t L1Pointer{};
/** @} */
} X86;
};
// Each guest JIT frame has one of these
struct CpuStateFrame {
CPUState State;
@@ -52,18 +166,17 @@ namespace FEXCore::Core {
*/
uint64_t InSyscallInfo{};
InternalThreadState* Thread;
// Pointers that the JIT needs to load to remove relocations
JITPointers Pointers;
};
static_assert(offsetof(CpuStateFrame, State) == 0, "CPUState must be first member in CpuStateFrame");
static_assert(offsetof(CpuStateFrame, State.rip) == 0, "rip must be zero offset in CpuStateFrame");
static_assert(offsetof(CpuStateFrame, Pointers) % 8 == 0, "JITPointers need to be aligned to 8 bytes");
static_assert(offsetof(CpuStateFrame, Pointers) + sizeof(CpuStateFrame::Pointers) <= 32760, "JITPointers maximum pointer needs to be less than architecture maximum 32768");
static_assert(std::is_standard_layout<CpuStateFrame>::value, "This needs to be standard layout");
#ifdef PAGE_SIZE
static_assert(PAGE_SIZE == 4096, "FEX only supports 4k pages");
#undef PAGE_SIZE
#endif
constexpr uint64_t PAGE_SIZE = 4096;
FEX_DEFAULT_VISIBILITY std::string_view const& GetFlagName(unsigned Flag);
FEX_DEFAULT_VISIBILITY std::string_view const& GetGRegName(unsigned Reg);
}
+1 -2
View File
@@ -3,11 +3,10 @@
#include <FEXCore/Utils/CompilerDefs.h>
#include <array>
#include <bits/types/siginfo_t.h>
#include <bits/types/stack_t.h>
#include <cstdint>
#include <functional>
#include <utility>
#include <signal.h>
#include <stddef.h>
namespace FEXCore {
+37 -11
View File
@@ -155,33 +155,59 @@ namespace FEXCore {
} _timer;
} _sifields;
union HostSigInfo_t {
// This anonymous struct needs to match the host definition
struct {
uint32_t si_signo;
uint32_t si_errno;
uint32_t si_code;
uint32_t __pad0;
// Pad[28] is a union for all the sifields
uint32_t _pad[28];
} FEXDef;
::siginfo_t host{};
};
static_assert(sizeof(HostSigInfo_t) == 128, "This needs to be the right size");
siginfo_t() = delete;
operator ::siginfo_t() const {
::siginfo_t val{};
val.si_signo = si_signo;
val.si_errno = si_errno;
val.si_code = si_code;
// The definition of siginfo_t changes depending on the host environment
// It is guaranteed to be 128 bytes and the kernel interface is the same for all of them
// Since we only run on Linux
HostSigInfo_t val{};
val.FEXDef.si_signo = si_signo;
val.FEXDef.si_errno = si_errno;
val.FEXDef.si_code = si_code;
// Host siginfo has a pad member that is set to zeros
val.__pad0 = 0;
val.FEXDef.__pad0 = 0;
// Copy over the union
// The union is different sizes on 64-bit versus 32-bit
memcpy(val._sifields._pad, _sifields.pad, std::min(sizeof(val._sifields._pad), sizeof(_sifields.pad)));
memcpy(val.FEXDef._pad, _sifields.pad, std::min(sizeof(val.FEXDef._pad), sizeof(_sifields.pad)));
return val;
return val.host;
}
siginfo_t(::siginfo_t val) {
si_signo = val.si_signo;
si_errno = val.si_errno;
si_code = val.si_code;
HostSigInfo_t host;
host.host = val;
si_signo = host.FEXDef.si_signo;
si_errno = host.FEXDef.si_errno;
si_code = host.FEXDef.si_code;
// Copy over the union
// The union is different sizes on 64-bit versus 32-bit
memcpy(val._sifields._pad, _sifields.pad, std::min(sizeof(val._sifields._pad), sizeof(_sifields.pad)));
memcpy(_sifields.pad, host.FEXDef._pad, std::min(sizeof(host.FEXDef._pad), sizeof(_sifields.pad)));
}
static_assert(offsetof(::siginfo_t, si_signo) == offsetof(HostSigInfo_t, FEXDef.si_signo), "si_signo in wrong location?");
static_assert(offsetof(::siginfo_t, si_errno) == offsetof(HostSigInfo_t, FEXDef.si_errno), "si_errno in wrong location?");
static_assert(offsetof(::siginfo_t, si_code) == offsetof(HostSigInfo_t, FEXDef.si_code), "si_code in wrong location?");
};
static_assert(sizeof(FEXCore::x86::siginfo_t) == 128, "This needs to be the right size");
+1 -4
View File
@@ -262,16 +262,13 @@ enum InstType {
TYPE_GROUP_EVEX,
// Just to make grepping easier
TYPE_3DNOW_TABLE = TYPE_INVALID,
TYPE_3DNOW_INST = TYPE_INVALID,
// Exists in the table but isn't decoded correctly
TYPE_UNDEC = TYPE_INVALID,
TYPE_MMX = TYPE_INVALID,
TYPE_PRIV = TYPE_INVALID,
TYPE_0F38_TABLE = TYPE_INVALID,
TYPE_0F3A_TABLE = TYPE_INVALID,
TYPE_3DNOW_TABLE = TYPE_INVALID,
};
namespace InstFlags {
+3
View File
@@ -2,6 +2,8 @@
#include <cstdint>
#include <string>
#include <FEXCore/IR/IR.h>
namespace FEXCore::Context {
struct Context;
}
@@ -43,6 +45,7 @@ namespace FEXCore::HLE {
virtual uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args) = 0;
virtual SyscallABI GetSyscallABI(uint64_t Syscall) = 0;
virtual FEXCore::IR::SyscallFlags GetSyscallFlags(uint64_t Syscall) const { return FEXCore::IR::SyscallFlags::DEFAULT; }
SyscallOSABI GetOSABI() const { return OSABI; }
+11
View File
@@ -1,6 +1,7 @@
#pragma once
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXHeaderUtils/EnumOperators.h>
#include <array>
#include <cassert>
@@ -482,6 +483,16 @@ protected:
OrderedNodeWrapper Node{};
};
enum class SyscallFlags : uint8_t {
DEFAULT = 0,
OPTIMIZETHROUGH = 1 << 0,
NOSYNCSTATEONENTRY = 1 << 1,
NORETURN = 1 << 2,
NOSIDEEFFECTS = 1 << 3
};
FEX_DEF_NUM_OPS(SyscallFlags)
#define IROP_ENUM
#define IROP_STRUCTS
#define IROP_SIZES
+20 -322
View File
@@ -33,6 +33,8 @@ friend class FEXCore::IR::PassManager;
*
* @{ */
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNode *Node);
// These handlers add cost to the constructor and destructor
// If it becomes an issue then blow them away
// GCC also generates some pretty atrocious code around these
@@ -40,7 +42,6 @@ friend class FEXCore::IR::PassManager;
#define IROP_ALLOCATE_HELPERS
#define IROP_DISPATCH_HELPERS
#include <FEXCore/IR/IRDefines.inc>
IRPair<IROp_Constant> _Constant(uint8_t Size, uint64_t Constant) {
auto Op = AllocateOp<IROp_Constant, IROps::OP_CONSTANT>();
uint64_t Mask = ~0ULL >> (64 - Size);
@@ -51,342 +52,39 @@ friend class FEXCore::IR::PassManager;
Op.first->Header.HasDest = true;
return Op;
}
IRPair<IROp_VBitcast> _VBitcast(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
auto Op = AllocateOp<IROp_VBitcast, IROps::OP_VBITCAST>();
Op.first->Header.Size = RegisterSize / 8;
Op.first->Header.ElementSize = ElementSize;
Op.first->Header.NumArgs = 1;
Op.first->Header.HasDest = true;
Op.first->Header.Args[0] = ssa0->Wrapped(DualListData.ListBegin());
ssa0->AddUse();
return Op;
}
IRPair<IROp_LoadContext> _LoadContext(uint8_t Size, uint32_t Offset, RegisterClassType Class) {
return _LoadContext(Offset, Class, Size);
}
IRPair<IROp_StoreContext> _StoreContext(RegisterClassType Class, uint8_t Size, uint32_t Offset, OrderedNode *ssa0) {
return _StoreContext(ssa0, Offset, Class, Size);
}
IRPair<IROp_Bfe> _Bfe(uint8_t Width, uint8_t lsb, OrderedNode *ssa0) {
return _Bfe(ssa0, Width, lsb, 0);
return _Bfe(0, Width, lsb, ssa0);
}
IRPair<IROp_Bfe> _Bfe(uint8_t DestSize, int8_t Width, uint8_t lsb, OrderedNode *ssa0) {
return _Bfe(ssa0, Width, lsb, DestSize);
IRPair<IROp_Sbfe> _Sext(uint8_t SrcSize, OrderedNode *ssa0) {
return _Sbfe(SrcSize, 0, ssa0);
}
IRPair<IROp_Sbfe> _Sbfe(uint8_t Width, uint8_t lsb, OrderedNode *ssa0) {
return _Sbfe(ssa0, Width, lsb);
IRPair<IROp_Jump> _Jump() {
return _Jump(InvalidNode);
}
IRPair<IROp_Bfi> _Bfi(uint8_t DestSize, uint8_t Width, uint8_t lsb, OrderedNode *ssa0, OrderedNode *ssa1) {
return _Bfi(ssa0, ssa1, Width, lsb, DestSize);
IRPair<IROp_CondJump> _CondJump(OrderedNode *ssa0, CondClassType cond = {COND_NEQ}) {
return _CondJump(ssa0, _Constant(0), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
}
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Align = 1) {
return _StoreMem(ssa0, ssa1, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
}
IRPair<IROp_StoreMemTSO> _StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Align = 1) {
return _StoreMemTSO(ssa0, ssa1, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
}
IRPair<IROp_VStoreMemElement> _VStoreMemElement(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Index, uint8_t Align = 1) {
return _VStoreMemElement(ssa0, ssa1, Index, Align, RegisterSize, ElementSize);
}
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
return _LoadMem(ssa0, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
}
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
return _LoadMemTSO(ssa0, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
}
IRPair<IROp_VLoadMemElement> _VLoadMemElement(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Index, uint8_t Align = 1) {
return _VLoadMemElement(ssa0, ssa1, Index, Align, RegisterSize, ElementSize);
}
IRPair<IROp_LoadContextIndexed> _LoadContextIndexed(OrderedNode *ssa0, uint8_t Size, uint32_t BaseOffset, uint32_t Stride, RegisterClassType Class) {
return _LoadContextIndexed(ssa0, BaseOffset, Stride, Class, Size);
}
IRPair<IROp_StoreContextIndexed> _StoreContextIndexed(OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Size, uint32_t BaseOffset, uint32_t Stride, RegisterClassType Class) {
return _StoreContextIndexed(ssa0, ssa1, BaseOffset, Stride, Class, Size);
IRPair<IROp_CondJump> _CondJump(OrderedNode *ssa0, OrderedNode *ssa1, OrderedNode *ssa2, CondClassType cond = {COND_NEQ}) {
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
}
IRPair<IROp_Select> _Select(uint8_t Cond, OrderedNode *ssa0, OrderedNode *ssa1, OrderedNode *ssa2, OrderedNode *ssa3, uint8_t CompareSize = 0) {
if (CompareSize == 0)
CompareSize = std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa0), GetOpSize(ssa1)));
return _Select(ssa0, ssa1, ssa2, ssa3, {Cond}, CompareSize);
return _Select(CondClassType{Cond}, ssa0, ssa1, ssa2, ssa3, CompareSize);
}
IRPair<IROp_Sbfe> _Sext(uint8_t SrcSize, OrderedNode *ssa0) {
return _Sbfe(SrcSize, 0, ssa0);
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
IRPair<IROp_VInsElement> _VInsElement(uint8_t RegisterSize, uint8_t ElementSize, uint8_t DestIdx, uint8_t SrcIdx, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VInsElement(ssa0, ssa1, DestIdx, SrcIdx, RegisterSize, ElementSize);
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
IRPair<IROp_VInsScalarElement> _VInsScalarElement(uint8_t RegisterSize, uint8_t ElementSize, uint8_t DestIdx, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VInsScalarElement(ssa0, ssa1, DestIdx, RegisterSize, ElementSize);
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
IRPair<IROp_VExtractElement> _VExtractElement(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t Index) {
return _VExtractElement(ssa0, Index, RegisterSize, ElementSize);
IRPair<IROp_StoreMemTSO> _StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
}
IRPair<IROp_VDupElement> _VDupElement(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t Index) {
return _VDupElement(ssa0, Index, RegisterSize, ElementSize);
}
IRPair<IROp_VAnd> _VAnd(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VAnd(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VBic> _VBic(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VBic(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VOr> _VOr(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VOr(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VXor> _VXor(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VXor(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VAdd> _VAdd(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VAdd(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSub> _VSub(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSub(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUQAdd> _VUQAdd(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUQAdd(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUQSub> _VUQSub(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUQSub(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSQAdd> _VSQAdd(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSQAdd(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSQSub> _VSQSub(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSQSub(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VAddP> _VAddP(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VAddP(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VAddV> _VAddV(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VAddV(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VUMinV> _VUMinV(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VUMinV(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VURAvg> _VURAvg(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VURAvg(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VAbs> _VAbs(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VAbs(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VPopcount> _VPopcount(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VPopcount(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VFMul> _VFMul(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFMul(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUMin> _VUMin(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUMin(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSMin> _VSMin(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSMin(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUMax> _VUMax(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUMax(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSMax> _VSMax(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSMax(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VZip> _VZip(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VZip(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VZip2> _VZip2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VZip2(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUnZip> _VUnZip(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUnZip(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUnZip2> _VUnZip2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUnZip2(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VCMPEQ> _VCMPEQ(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VCMPEQ(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VCMPEQZ> _VCMPEQZ(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VCMPEQZ(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VCMPGT> _VCMPGT(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VCMPGT(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VCMPGTZ> _VCMPGTZ(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VCMPGTZ(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VCMPLTZ> _VCMPLTZ(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VCMPLTZ(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VFAdd> _VFAdd(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFAdd(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VFAddP> _VFAddP(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFAddP(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VFSub> _VFSub(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFSub(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VFCMPEQ> _VFCMPEQ(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFCMPEQ(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VFCMPNEQ> _VFCMPNEQ(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFCMPNEQ(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VFCMPLT> _VFCMPLT(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFCMPLT(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VFCMPGT> _VFCMPGT(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFCMPGT(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VFCMPLE> _VFCMPLE(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFCMPLE(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VFCMPUNO> _VFCMPUNO(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFCMPUNO(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VFCMPORD> _VFCMPORD(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VFCMPORD(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUShl> _VUShl(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUShl(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUShlS> _VUShlS(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUShlS(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUShrS> _VUShrS(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUShrS(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSShrS> _VSShrS(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSShrS(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUShr> _VUShr(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUShr(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSShr> _VSShr(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSShr(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VExtr> _VExtr(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Index) {
return _VExtr(ssa0, ssa1, Index, RegisterSize, ElementSize);
}
IRPair<IROp_VSLI> _VSLI(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t ByteShift) {
return _VSLI(ssa0, ByteShift, RegisterSize, ElementSize);
}
IRPair<IROp_VSRI> _VSRI(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t ByteShift) {
return _VSRI(ssa0, ByteShift, RegisterSize, ElementSize);
}
IRPair<IROp_VUShrI> _VUShrI(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t BitShift) {
return _VUShrI(ssa0, BitShift, RegisterSize, ElementSize);
}
IRPair<IROp_VSShrI> _VSShrI(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t BitShift) {
return _VSShrI(ssa0, BitShift, RegisterSize, ElementSize);
}
IRPair<IROp_VUShrNI> _VUShrNI(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t BitShift) {
return _VUShrNI(ssa0, BitShift, RegisterSize, ElementSize);
}
IRPair<IROp_VUShrNI2> _VUShrNI2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t BitShift) {
return _VUShrNI2(ssa0, ssa1, BitShift, RegisterSize, ElementSize);
}
IRPair<IROp_VShlI> _VShlI(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t BitShift) {
return _VShlI(ssa0, BitShift, RegisterSize, ElementSize);
}
IRPair<IROp_VFSqrt> _VFSqrt(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VFSqrt(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VFRSqrt> _VFRSqrt(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VFRSqrt(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VNeg> _VNeg(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VNeg(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VFNeg> _VFNeg(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VFNeg(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VNot> _VNot(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VNot(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VSQXTN> _VSQXTN(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VSQXTN(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VSQXTN2> _VSQXTN2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSQXTN2(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSQXTUN> _VSQXTUN(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VSQXTUN(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VSQXTUN2> _VSQXTUN2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSQXTUN2(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VCastFromGPR> _VCastFromGPR(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VCastFromGPR(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VExtractToGPR> _VExtractToGPR(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, uint8_t Index) {
return _VExtractToGPR(ssa0, Index, RegisterSize, ElementSize);
}
IRPair<IROp_VInsGPR> _VInsGPR(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Index) {
return _VInsGPR(ssa0, ssa1, Index, RegisterSize, ElementSize);
}
IRPair<IROp_Vector_FToF> _Vector_FToF(uint8_t RegisterSize, uint8_t DstElementSize, uint8_t SrcElementSize, OrderedNode *ssa0) {
return _Vector_FToF(ssa0, SrcElementSize, RegisterSize, DstElementSize);
}
IRPair<IROp_Float_FromGPR_S> _Float_FromGPR_S(uint8_t DstElementSize, uint8_t SrcElementSize, OrderedNode *ssa0) {
return _Float_FromGPR_S(ssa0, SrcElementSize, DstElementSize);
}
IRPair<IROp_Float_FToF> _Float_FToF(uint8_t DstElementSize, uint8_t SrcElementSize, OrderedNode *ssa0) {
return _Float_FToF(ssa0, SrcElementSize, DstElementSize);
}
IRPair<IROp_VUMul> _VUMul(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUMul(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSMul> _VSMul(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSMul(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUMull> _VUMull(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUMull(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSMull> _VSMull(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSMull(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUMull2> _VUMull2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUMull2(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSMull2> _VSMull2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VSMull2(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VUABDL> _VUABDL(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VUABDL(ssa0, ssa1, RegisterSize, ElementSize);
}
IRPair<IROp_VSXTL> _VSXTL(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VSXTL(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VSXTL2> _VSXTL2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VSXTL2(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VUXTL> _VUXTL(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VUXTL(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VUXTL2> _VUXTL2(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0) {
return _VUXTL2(ssa0, RegisterSize, ElementSize);
}
IRPair<IROp_VTBL1> _VTBL1(uint8_t RegisterSize, OrderedNode *ssa0, OrderedNode *ssa1) {
return _VTBL1(ssa0, ssa1, RegisterSize);
}
IRPair<IROp_Jump> _Jump() {
return _Jump(InvalidNode);
}
IRPair<IROp_CondJump> _CondJump(OrderedNode *ssa0, CondClassType cond = {COND_NEQ}) {
return _CondJump(ssa0, _Constant(0), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
}
IRPair<IROp_CondJump> _CondJump(OrderedNode *ssa0, OrderedNode *ssa1, OrderedNode *ssa2, CondClassType cond = {COND_NEQ}) {
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
}
IRPair<IROp_Phi> _Phi() {
return _Phi(InvalidNode, InvalidNode, 0);
}
IRPair<IROp_PhiValue> _PhiValue(OrderedNode *Value, OrderedNode *Block) {
return _PhiValue(Value, Block, InvalidNode);
}
OrderedNode *Invalid() {
return InvalidNode;
}
+1
View File
@@ -4,6 +4,7 @@
#include <cstdint>
#include <functional>
#include <sys/types.h>
namespace FEXCore::Allocator {
using MMAP_Hook = void*(*)(void*, size_t, int, int, int, off_t);
+1 -1
Vendored Submodule
+1
Submodule External/robin-map added at a603419b9a.
+1 -1
@@ -0,0 +1,24 @@
#pragma once
#define FEX_DEF_ENUM_CLASS_BIN_OP(Enum, Op) \
[[maybe_unused]] \
static constexpr Enum operator Op(Enum lhs, Enum rhs) { \
using Type = std::underlying_type_t<Enum>; \
Type _lhs = static_cast<Type>(lhs); \
Type _rhs = static_cast<Type>(rhs); \
return static_cast<Enum>(_lhs Op _rhs); \
}
#define FEX_DEF_ENUM_CLASS_UNARY_OP(Enum, Op) \
[[maybe_unused]] \
static constexpr Enum operator Op(Enum rhs) { \
using Type = std::underlying_type_t<Enum>; \
Type _rhs = static_cast<Type>(rhs); \
return static_cast<Enum>(Op _rhs); \
}
#define FEX_DEF_NUM_OPS(Enum) \
FEX_DEF_ENUM_CLASS_BIN_OP(Enum, |) \
FEX_DEF_ENUM_CLASS_BIN_OP(Enum, &) \
FEX_DEF_ENUM_CLASS_BIN_OP(Enum, ^) \
FEX_DEF_ENUM_CLASS_UNARY_OP(Enum, ~)
+10
View File
@@ -28,6 +28,16 @@ namespace FHU::Syscalls {
#define CLONE_PIDFD 0x00001000
#endif
#if defined(__aarch64__) || defined(_M_ARM64)
#ifndef SYS_statx
#define SYS_statx 291
#endif
#elif defined(__x86_64__) || defined(_M_X64)
#ifndef SYS_statx
#define SYS_statx 332
#endif
#endif
inline int32_t getcpu(uint32_t *cpu, uint32_t *node) {
// Third argument is unused
#if defined(HAS_SYSCALL_GETCPU) && HAS_SYSCALL_GETCPU
+1 -1
View File
@@ -9,7 +9,7 @@ FEX is very much work in progress, so expect things to change.
### For Ubuntu 20.04, 21.04, 21.10, 22.04
Execute the following command in the terminal to install FEX through a PPA.
`curl --silent https://raw.githubusercontent.com/FEX-Emu/FEX/main/Scripts/InstallFEX.py | python3`
`curl --silent https://raw.githubusercontent.com/FEX-Emu/FEX/main/Scripts/InstallFEX.py --output /tmp/InstallFEX.py && python3 /tmp/InstallFEX.py && rm /tmp/InstallFEX.py`
This command will walk you through installing FEX through a PPA, and downloading a RootFS for use with FEX.
+14 -2
View File
@@ -2,7 +2,7 @@
import re
import sys
import subprocess
from distutils.version import LooseVersion, StrictVersion
from pkg_resources import parse_version
# Order this list from oldest to newest
# try not to list something newer than our minimum compiler supported version
@@ -16,6 +16,14 @@ BigCoreIDs = {
tuple([0x41, 0xd0d]): "cortex-a77",
tuple([0x41, 0xd41]): "cortex-a78",
tuple([0x41, 0xd44]): "cortex-x1",
tuple([0x41, 0xd47]):
[ ["cortex-a78", "0.0"],
["cortex-a710", "999.0"], # Doesn't exist in clang as of version 13
],
tuple([0x41, 0xd48]):
[ ["cortex-x1", "0.0"],
["cortex-x2", "999.0"], # Doesn't exist in clang as of version 13
],
tuple([0x41, 0xd0c]): "neoverse-n1",
tuple([0x41, 0xd49]): "neoverse-n2",
## Nvidia
@@ -36,6 +44,10 @@ LittleCoreIDs = {
tuple([0x41, 0xd04]): "cortex-a35",
tuple([0x41, 0xd03]): "cortex-a53",
tuple([0x41, 0xd05]): "cortex-a55",
tuple([0x41, 0xd46]):
[ ["cortex-a55", "0.0"],
["cortex-a510", "999.0"], # Doesn't exist in clang as of version 13
],
# Qualcomm
tuple([0x51, 0x801]): "cortex-a53", # Kryo 2xx Silver
@@ -68,7 +80,7 @@ for core in cpuinfo:
IDList = BigCoreIDs.get(core)
if type(IDList) is list:
for ID in IDList:
if StrictVersion(clang_version) >= StrictVersion(ID[1]):
if parse_version(clang_version) >= parse_version(ID[1]):
largest_big = ID[0]
else:
largest_big = BigCoreIDs.get(core)
+24 -2
View File
@@ -148,7 +148,7 @@ def parse_json(json_text, output_file):
OptionRegData = {}
OptionMemoryRegions = {}
OptionMemoryData = {}
OptionEnvironmentVariables = {}
json_object = json.loads(json_text)
json_object = {k.upper(): v for k, v in json_object.items()}
@@ -242,6 +242,14 @@ def parse_json(json_text, output_file):
length, byte_data = parse_hexstring(data_val)
OptionMemoryData[int(data_key, 0)] = (length, byte_data)
if ("ENV" in json_object):
data = json_object["ENV"]
if not (type(data) is dict):
sys.exit("Environment variables value must be list of key:value pairs")
for data_key, data_val in data.items():
OptionEnvironmentVariables[data_key] = data_val
# If Match option wasn't touched then set it to the default
if (OptionMatch == Regs.REG_INVALID):
OptionMatch = Regs.REG_NONE
@@ -250,6 +258,7 @@ def parse_json(json_text, output_file):
memRegions = bytes()
regData = bytes()
memData = bytes()
envData = bytes()
# Write memory regions
for key, val in OptionMemoryRegions.items():
@@ -271,6 +280,13 @@ def parse_json(json_text, output_file):
for byte in data:
memData += struct.pack('B', byte)
# Write environment variables
for key, val in OptionEnvironmentVariables.items():
envData += key.encode()
envData += struct.pack('B', 0)
envData += val.encode()
envData += struct.pack('B', 0)
config_file = open(output_file, "wb")
config_file.write(struct.pack('Q', OptionMatch.value))
config_file.write(struct.pack('Q', OptionIgnore.value))
@@ -280,7 +296,7 @@ def parse_json(json_text, output_file):
config_file.write(struct.pack('I', OptionMode.value))
# Total length of header, including offsets/counts below
headerLength = (8 * 4) + (4 * 2) + (4 * 6)
headerLength = (8 * 4) + (4 * 2) + (4 * 8)
offset = headerLength
# memory regions offset/count
@@ -298,10 +314,16 @@ def parse_json(json_text, output_file):
config_file.write(struct.pack('I', len(OptionMemoryData)))
offset += len(memData)
# environment data offset/count
config_file.write(struct.pack('I', offset))
config_file.write(struct.pack('I', len(OptionEnvironmentVariables)))
offset += len(envData)
# write out the actual data for memory regions, reg data and memory data
config_file.write(memRegions)
config_file.write(regData)
config_file.write(memData)
config_file.write(envData)
config_file.close()
+17
View File
@@ -213,6 +213,23 @@ std::string GetRootFSLockFile() {
return LockPath;
}
std::string GetRootFSPathIfExists() {
FEX_CONFIG_OPT(LDPath, ROOTFS);
if (FEX::FormatCheck::IsSquashFS(LDPath())) {
auto LockFile = FEX::RootFS::GetRootFSLockFile();
std::string MountPath;
if (FEX::RootFS::CheckLockExists(LockFile, &MountPath)) {
return MountPath;
}
// Wasn't mounted, don't return squashfs name
return {};
}
// Wasn't squashfs, return regular ROOTFS config
return LDPath();
}
enum class ErrorResult {
ERROR_SUCCESS,
ERROR_FAIL,
+7
View File
@@ -10,6 +10,13 @@ namespace FEX::RootFS {
std::string GetRootFSLockFile();
// Returns the socket file for a mount path
std::string GetRootFSSocketFile(std::string const &MountPath);
// Returns the string to the rootfs path if it exists
// If unconfigured: Return empty string
// If regular directory: Return directory
// If squashfs: Only returns tmpfs directory if mounted, otherwise empty string
std::string GetRootFSPathIfExists();
// Checks if the rootfs lock exists
bool CheckLockExists(std::string const &LockPath, std::string *MountPath = nullptr);
bool Setup(char **const envp, uint32_t TryCount = 0);
+7
View File
@@ -5,8 +5,10 @@
#include <FEXHeaderUtils/Syscalls.h>
#include <atomic>
#include <byteswap.h>
#include <fcntl.h>
#include <netdb.h>
#include <netinet/in.h>
#include <poll.h>
#include <sstream>
#include <string>
@@ -14,7 +16,12 @@
#include <sys/types.h>
#include <sys/un.h>
#include <unistd.h>
#include <vector>
// For older build environments
#ifndef POLLREMOVE
#define POLLREMOVE 0x1000
#endif
namespace FEX::SocketLogging {
namespace Common {
enum class PacketTypes : uint32_t {
+27 -11
View File
@@ -35,18 +35,34 @@ install(TARGETS FEXLoader
COMPONENT runtime
)
add_custom_target(FEXInterpreter ALL
COMMAND "ln" "-f" "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/FEXLoader" "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/FEXInterpreter"
DEPENDS FEXLoader
)
if(TERMUX_BUILD)
# Termux doesn't support hard links, just copy FEXLoader
add_custom_target(FEXInterpreter ALL
COMMAND "cp" "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/FEXLoader" "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/FEXInterpreter"
DEPENDS FEXLoader
)
install(
CODE "MESSAGE(\"-- Installing: ${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter\")"
CODE "
EXECUTE_PROCESS(COMMAND ln -f FEXLoader FEXInterpreter
WORKING_DIRECTORY ${CMAKE_INSTALL_PREFIX}/bin/
)"
)
install(
CODE "MESSAGE(\"-- Installing: ${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter\")"
CODE "
EXECUTE_PROCESS(COMMAND cp FEXLoader FEXInterpreter
WORKING_DIRECTORY ${CMAKE_INSTALL_PREFIX}/bin/
)"
)
else()
add_custom_target(FEXInterpreter ALL
COMMAND "ln" "-f" "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/FEXLoader" "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/FEXInterpreter"
DEPENDS FEXLoader
)
install(
CODE "MESSAGE(\"-- Installing: ${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter\")"
CODE "
EXECUTE_PROCESS(COMMAND ln -f FEXLoader FEXInterpreter
WORKING_DIRECTORY ${CMAKE_INSTALL_PREFIX}/bin/
)"
)
endif()
install(PROGRAMS "${PROJECT_SOURCE_DIR}/Scripts/FEXUpdateAOTIRCache.sh" DESTINATION bin RENAME FEXUpdateAOTIRCache)
+13 -3
View File
@@ -341,6 +341,16 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
// map stack here, so that nothing gets mapped there
// This works with both 64-bit and 32-bit. The mapper will only give us a function in the correct region
//
// MAP_GROWSDOWN is required here. The default stack pointer allocated by the kernel is mapped with it.
// Some libraries (like libfmod) will have a PT_GNU_STACK with executable stack bit set
// On dlopen glibc will check its current stack allocation permission bits (using internal expectations of allocation, not /proc/self/maps)
// If stack hasn't been allocated as executable then it will proceed to mprotect the range with the executable bit set
// Then it will mprotect the base stack page with `PROT_READ|PROT_WRITE|PROT_EXEC|PROT_GROWSDOWN`
// If the original stack memory region wasn't allocated with MAP_GROWSDOWN then the mprotect with PROT_GROWSDOWN will fail with EINVAL
//
// This is still technically a memory leak if the stack grows, but since the primary thread's stack only gets destroyed on process close, this is
// fine.
StackPointer = reinterpret_cast<uintptr_t>(Mapper(nullptr, StackSize(), PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN, -1, 0));
if (StackPointer == ~0ULL) {
@@ -382,7 +392,7 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
return false;
}
InterpeterElfBase = InterpLoadBase + InterpElf.phdrs.front().p_vaddr - MainElf.phdrs.front().p_offset;
InterpeterElfBase = InterpLoadBase + InterpElf.phdrs.front().p_vaddr - InterpElf.phdrs.front().p_offset;
Entrypoint = InterpLoadBase + InterpElf.ehdr.e_entry;
} else {
InterpeterElfBase = 0;
@@ -401,7 +411,7 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
AuxVariables.emplace_back(auxv_t{25, ~0ULL}); // AT_RANDOM
AuxVariables.emplace_back(auxv_t{23, 0}); // AT_SECURE
AuxVariables.emplace_back(auxv_t{8, 0}); // AT_FLAGS
AuxVariables.emplace_back(auxv_t{5, MainElf.phdrs.size()});
AuxVariables.emplace_back(auxv_t{5, MainElf.phdrs.size()}); // AT_PHNUM
if (Is64BitMode()) {
AuxVariables.emplace_back(auxv_t{4, 0x38}); // AT_PHENT
@@ -424,7 +434,7 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
//AuxVariables.emplace_back(auxv_t{33, 0}); // AT_SYSINFO_EHDR - Address of the start of VDSO
}
AuxVariables.emplace_back(auxv_t{3, MainElfBase + MainElf.ehdr.e_phoff}); // Program header
AuxVariables.emplace_back(auxv_t{7, InterpeterElfBase}); // Interpreter address
AuxVariables.emplace_back(auxv_t{7, InterpeterElfBase}); // AT_BASE - Interpreter address
AuxVariables.emplace_back(auxv_t{9, MainElfEntrypoint}); // AT_ENTRY
AuxVariables.emplace_back(auxv_t{0, 0}); // Null ender
-51
View File
@@ -108,47 +108,6 @@ void AssertHandler(char const *Message) {
fsync(OutputFD);
}
bool CheckMemMapping() {
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
std::string Line;
while (std::getline(fs, Line)) {
if (fs.eof()) {
break;
}
uint64_t Begin{};
uint64_t End{};
if (sscanf(Line.c_str(), "%lx-%lx", &Begin, &End) == 2) {
// If a memory range is living inside the 32bit memory space then we have a problem
if (Begin < 0x1'0000'0000) {
return false;
}
}
}
return true;
}
void PrintIntersectingMapping() {
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
std::string Line;
while (std::getline(fs, Line)) {
if (fs.eof()) {
break;
}
uint64_t Begin{};
uint64_t End{};
if (sscanf(Line.c_str(), "%lx-%lx", &Begin, &End) == 2) {
// If a memory range is living inside the 32bit memory space then we have a problem
if (Begin < 0x1'0000'0000) {
LogMan::Msg::DFmt("*** {}", Line);
}
else {
LogMan::Msg::DFmt(" {}", Line);
}
}
}
}
} // Anonymous namespace
void InterpreterHandler(std::string *Filename, std::string const &RootFS, std::vector<std::string> *args) {
@@ -234,16 +193,6 @@ int main(int argc, char **argv, char **const envp) {
LogMan::Throw::InstallHandler(AssertHandler);
LogMan::Msg::InstallHandler(MsgHandler);
#if !(defined(ENABLE_ASAN) && ENABLE_ASAN)
// LLVM ASAN maps things to the lower 32bits
// Valgrind also places us in the lower 32-bits
if (!getenv("VALGRIND_LAUNCHER") &&
!CheckMemMapping()) {
PrintIntersectingMapping();
LogMan::Msg::DFmt("[WARNING] FEX mapped to lower 32bits! 32-bit applications may have issues!");
}
#endif
auto Program = FEX::Config::LoadConfig(
IsInterpreter,
true,
+21 -7
View File
@@ -10,6 +10,7 @@
#include <cstring>
#include <fstream>
#include <sys/mman.h>
#include <sys/user.h>
#include <vector>
#include <FEXCore/Core/CodeLoader.h>
@@ -161,6 +162,21 @@ namespace FEX::HarnessHelper {
void Init(std::string const &ConfigFilename) {
ReadFile(ConfigFilename, &RawConfigFile);
memcpy(&BaseConfig, RawConfigFile.data(), sizeof(ConfigStructBase));
GetEnvironmentOptions();
}
std::vector<std::pair<std::string_view, std::string_view>> GetEnvironmentOptions() {
std::vector<std::pair<std::string_view, std::string_view>> Env{};
uintptr_t DataOffset = BaseConfig.OptionEnvOptionOffset;
for (unsigned i = 0; i < BaseConfig.OptionEnvOptionCount; ++i) {
// Environment variables are null terminated strings
std::string_view Key = RawConfigFile.data() + DataOffset;
std::string_view Value = RawConfigFile.data() + DataOffset + Key.size() + 1;
DataOffset += Key.size() + Value.size() + 2;
Env.emplace_back(Key, Value);
}
return Env;
}
bool CompareStates(FEXCore::Core::CPUState const* State1, FEXCore::Core::CPUState const* State2) {
@@ -316,6 +332,8 @@ namespace FEX::HarnessHelper {
uint32_t OptionRegDataCount;
uint32_t OptionMemDataOffset;
uint32_t OptionMemDataCount;
uint32_t OptionEnvOptionOffset;
uint32_t OptionEnvOptionCount;
uint8_t AdditionalData[];
} FEX_PACKED;
@@ -340,14 +358,7 @@ namespace FEX::HarnessHelper {
ConfigStructBase BaseConfig;
};
#ifdef PAGE_SIZE
static_assert(PAGE_SIZE == 4096, "FEX only supports 4k pages");
#undef PAGE_SIZE
#endif
class HarnessCodeLoader final : public FEXCore::CodeLoader {
static constexpr uint32_t PAGE_SIZE = 4096;
public:
HarnessCodeLoader(std::string const &Filename, const char *ConfigFilename) {
@@ -424,6 +435,9 @@ namespace FEX::HarnessHelper {
Config.LoadMemory();
}
std::vector<std::pair<std::string_view, std::string_view>> GetEnvironmentOptions() {
return Config.GetEnvironmentOptions();
}
bool CompareStates(FEXCore::Core::CPUState const* State1, FEXCore::Core::CPUState const* State2) {
return Config.CompareStates(State1, State2);
@@ -76,8 +76,8 @@ enum Syscalls_Arm64 {
SYSCALL_Arm64_write = 64,
SYSCALL_Arm64_readv = 65,
SYSCALL_Arm64_writev = 66,
SYSCALL_Arm64_pread64 = 67,
SYSCALL_Arm64_pwrite64 = 68,
SYSCALL_Arm64_pread_64 = 67,
SYSCALL_Arm64_pwrite_64 = 68,
SYSCALL_Arm64_preadv = 69,
SYSCALL_Arm64_pwritev = 70,
SYSCALL_Arm64_sendfile = 71,
@@ -255,7 +255,7 @@ enum Syscalls_Arm64 {
SYSCALL_Arm64_accept4 = 242,
SYSCALL_Arm64_recvmmsg = 243,
SYSCALL_Arm64_wait4 = 260,
SYSCALL_Arm64_prlimit64 = 261,
SYSCALL_Arm64_prlimit_64 = 261,
SYSCALL_Arm64_fanotify_init = 262,
SYSCALL_Arm64_fanotify_mark = 263,
SYSCALL_Arm64_name_to_handle_at = 264,
@@ -21,6 +21,7 @@ $end_info$
#include <filesystem>
#include <fstream>
#include <stdio.h>
#include <string.h>
#include <sys/stat.h>
#include <sys/statfs.h>
#include <syscall.h>
@@ -15,6 +15,7 @@ $end_info$
#include <stddef.h>
#include <string>
#include <sys/stat.h>
#include <vector>
#include <unordered_map>
#include <unordered_set>
+24 -12
View File
@@ -8,6 +8,7 @@
#include <map>
#include <linux/mman.h>
#include <unistd.h>
#include <sys/user.h>
#include <sys/mman.h>
#include <sys/shm.h>
@@ -18,9 +19,6 @@
namespace FEX::HLE {
class MemAllocator32Bit final : public FEX::HLE::MemAllocator {
private:
static constexpr uint64_t PAGE_SHIFT = 12;
static constexpr uint64_t PAGE_SIZE = 1 << PAGE_SHIFT;
static constexpr uint64_t PAGE_MASK = (1 << PAGE_SHIFT) - 1;
static constexpr uint64_t BASE_KEY = 16;
const uint64_t TOP_KEY = 0xFFFF'F000ULL >> PAGE_SHIFT;
const uint64_t TOP_KEY32BIT = 0x1F'F000ULL >> PAGE_SHIFT;
@@ -139,17 +137,19 @@ void *MemAllocator32Bit::mmap(void *addr, size_t length, int prot, int flags, in
uintptr_t Addr = reinterpret_cast<uintptr_t>(addr);
uintptr_t PageAddr = Addr >> PAGE_SHIFT;
// Define MAP_FIXED_NOREPLACE ourselves to ensure we always parse this flag
constexpr int FEX_MAP_FIXED_NOREPLACE = 0x100000;
bool Fixed = ((flags & MAP_FIXED) ||
(flags & MAP_FIXED_NOREPLACE));
(flags & FEX_MAP_FIXED_NOREPLACE));
// Both Addr and length must be page aligned
if (Addr & PAGE_MASK) {
if (Addr & ~PAGE_MASK) {
return reinterpret_cast<void*>(-EINVAL);
}
// If we do have an fd then offset must be page aligned
if (fd != -1 &&
offset & PAGE_MASK) {
offset & ~PAGE_MASK) {
return reinterpret_cast<void*>(-EINVAL);
}
@@ -170,8 +170,7 @@ void *MemAllocator32Bit::mmap(void *addr, size_t length, int prot, int flags, in
bool Map32Bit = flags & FEX::HLE::X86_64_MAP_32BIT;
// Find a region that fits our address
if (Addr == 0) {
auto AllocateNoHint = [&]() -> void*{
bool Wrapped = false;
uint64_t BottomPage = Map32Bit && (LastScanLocation >= LastKeyLocation32Bit) ? LastKeyLocation32Bit : LastScanLocation;
restart:
@@ -194,7 +193,7 @@ restart:
reinterpret_cast<void*>(LowerPage<< PAGE_SHIFT),
length,
prot,
flags | MAP_FIXED_NOREPLACE,
flags | FEX_MAP_FIXED_NOREPLACE,
fd,
offset);
@@ -244,6 +243,11 @@ restart:
}
}
}
};
// Find a region that fits our address
if (Addr == 0) {
return AllocateNoHint();
}
else {
void *MappedPtr = ::mmap(
@@ -254,7 +258,15 @@ restart:
fd,
offset);
if (MappedPtr != MAP_FAILED) {
if (MappedPtr >= reinterpret_cast<void*>(TOP_KEY << PAGE_SHIFT) &&
(flags & FEX_MAP_FIXED_NOREPLACE)) {
// Handles the case where MAP_FIXED_NOREPLACE isn't handled by the host system's
// kernel and returns the wrong pointer
// Make sure to munmap this so we don't leak memory
::munmap(MappedPtr, length);
return reinterpret_cast<void*>(-EEXIST);
}
else if (MappedPtr != MAP_FAILED) {
SetUsedPages(PageAddr, PagesLength);
return MappedPtr;
}
@@ -275,11 +287,11 @@ int MemAllocator32Bit::munmap(void *addr, size_t length) {
uintptr_t PageEnd = PageAddr + PagesLength;
// Both Addr and length must be page aligned
if (Addr & PAGE_MASK) {
if (Addr & ~PAGE_MASK) {
return -EINVAL;
}
if (length & PAGE_MASK) {
if (length & ~PAGE_MASK) {
return -EINVAL;
}
@@ -24,7 +24,6 @@ $end_info$
#include <exception>
#include <functional>
#include <linux/futex.h>
#include <bits/types/stack_t.h>
#include <signal.h>
#include <syscall.h>
#include <sys/mman.h>
@@ -32,6 +31,11 @@ $end_info$
#include <unistd.h>
#include <utility>
// For older build environments
#ifndef SS_AUTODISARM
#define SS_AUTODISARM (1U << 31)
#endif
namespace FEX::HLE {
#ifdef _M_X86_64
__attribute__((naked))
@@ -42,7 +46,6 @@ namespace FEX::HLE {
}
#endif
constexpr static uint32_t SS_AUTODISARM = (1U << 31);
constexpr static uint32_t X86_MINSIGSTKSZ = 0x2000U;
// We can only have one delegator per process
+1 -2
View File
@@ -9,8 +9,7 @@ $end_info$
#include <array>
#include <atomic>
#include <bits/types/siginfo_t.h>
#include <bits/types/stack_t.h>
#include <signal.h>
#include <stddef.h>
#include <stdint.h>
#include <mutex>
Loaded 100 of 243 files, more files were not shown because too many files have changed in this diff. Show more