Compare commits

...
151 Commits
Author SHA1 Message Date
Ryan Houdek fd3e988a20 Docs: Update for release FEX-2211 2022-11-02 23:25:10 -07:00
Ryan Houdek b1d98f4e58 Merge pull request #2134 from lioncash/temp
Arm64/ConversionOps: Eliminate use of temporary in Vector_FToF
2022-11-02 19:18:40 -07:00
Ryan Houdek 9e7daf61d0 Merge pull request #2133 from lioncash/inselem
IR: Handle 256-bit VInsElement
2022-11-02 18:59:18 -07:00
lioncash 6fbe25753b IR: Handle 256-bit VInsElement
Extends VInsElem to handle 256-bit vectors.
2022-11-03 01:43:42 +00:00
Ryan Houdek 03f0edc5b5 Merge pull request #2132 from lioncash/indexed
IR: Handle 256-bit LoadContextIndexed/StoreContextIndexed
2022-11-02 17:13:36 -07:00
lioncash 5536f1e835 Arm64/ConversionOps: Eliminate use of temporary in Vector_FToF
We can just use the destination register in this case.
2022-11-02 23:50:28 +00:00
lioncash 0de36706da IR.json: Expand allowed size in LoadContext and StoreContext IR ops
These can now handle 256-bit destinations
2022-11-02 16:19:05 +00:00
lioncash 17722dad6d IR: Handle 256-bit LoadContextIndexed
Extends LoadContextIndexed to handle 256-bit vectors.
2022-11-02 16:16:58 +00:00
lioncash 0371599996 IR: Handle 256-bit StoreContextIndexed
Extends StoreContextIndexed to handle 256-bit vectors.
2022-11-02 16:07:02 +00:00
Ryan Houdek 199649b30f Merge pull request #2131 from lioncash/simplify
Arm64/MemoryOps: Merge if statement into switch in ParanoidLoadMemTSO
2022-11-01 21:35:25 -07:00
lioncash 4ef35488db Arm64/MemoryOps: Merge if statement into switch in ParanoidLoadMemTSO
There's nothing preventing the OpSize == 1 case from being merged into
the switch, so we can do that to make things a little more consistent.
2022-11-02 03:45:17 +00:00
Ryan Houdek 70a91ee6ce Merge pull request #2130 from lioncash/memory
IR: Handle 256-bit StoreMem/StoreMemTSO/ParanoidStoreMemTSO
2022-11-01 20:41:34 -07:00
lioncash 418a27e47e IR: Handle 256-bit ParanoidStoreMemTSO
Extends ParanoidStoreMemTSO to handle 256-bit vectors.
2022-11-01 23:04:30 +00:00
lioncash 61c76d02cc IR: Handle 256-bit StoreMemTSO
Extends StoreMemTSO to handle 256-bit vectors.
2022-11-01 22:59:29 +00:00
lioncash d98641221d IR: Handle 256-bit StoreMem
Extends StoreMem to handle 256-bit vectors.
2022-11-01 21:39:52 +00:00
Ryan Houdek 8a14f87a44 Merge pull request #2129 from lioncash/memory
IR: Handle 256-bit LoadMem/LoadMemTSO/ParanoidLoadMemTSO
2022-11-01 14:23:19 -07:00
lioncash 02ce71734c IR: handle 256-bit ParanoidLoadTSO
Extends ParanoidLoadTSO to handle 256-bit vectors.
2022-11-01 21:02:42 +00:00
lioncash 96c2743280 IR: Handle 256-bit LoadMemTSO
Extends LoadMemTSO to handle 256-bit vectors.
2022-11-01 21:02:42 +00:00
lioncash 7bfc34b51c IR: Handle 256-bit LoadMem
Extends LoadMem to handle 256-bit vectors.
2022-11-01 21:02:38 +00:00
Ryan Houdek 40d820fd05 Merge pull request #2127 from lioncash/spill
Arm64/MemoryOps: Remove lingering unnecessary ptrue instances
2022-11-01 10:47:15 -07:00
lioncash d69287aaf7 Arm64/MemoryOps: Remove lingering unnecessary ptrue instances
Gets rid of some leftover bits from when we didn't have statically
allocated predicate registers.
2022-11-01 17:25:24 +00:00
Ryan Houdek d475b0ba9e Merge pull request #2126 from lioncash/ctx
IR: Handle 256-bit LoadContext/StoreContext
2022-11-01 10:22:26 -07:00
lioncash 1638b744b7 x86_64/MemoryOps: Ensure upper lane is cleared properly in FillRegister
Ensures that loaded values don't potentially have junk in the upper
lane. Will prevent potential wonky situations when implementing AVX
instructions.
2022-11-01 16:39:18 +00:00
lioncash 8b19894a06 IR: Handle 256-bit StoreContext
Extends StoreContext to handle 256-bit vectors.
2022-11-01 16:20:24 +00:00
Ryan Houdek d2e0dc99de Merge pull request #2125 from lioncash/unused
Interpreter/MiscOps: Remove unused StopThread() function
2022-11-01 09:13:31 -07:00
lioncash d04e40b5fd Interpreter/MiscOps: Remove unused StopThread() function
This has been unused since ff1d51c7bd

Silences a compiler warning.
2022-11-01 15:59:17 +00:00
lioncash 75d797b5cd IR: Handle 256-bit LoadContext
Extends LoadContext to handle 256-bit vectors.
2022-11-01 14:54:26 +00:00
Mai ecf4891087 Merge pull request #1668 from Sonicadvance1/wip_segment_register
Segment register index optimization
2022-11-01 02:54:44 +00:00
Ryan Houdek 0e1a418678 WIP: Segment register index optimization
Segment registers are indexed significantly more than they are changed.
Pay the cost of indexing during the set and store rather than the per
register index.

Should be a fairly significant performance improvement for 32-bit
applications. At least on hardware that doesn't have a data dependent
prefetcher.

Breaks Steam atm and isn't clean.
2022-10-31 19:42:30 -07:00
Mai 5bef13df94 Merge pull request #2124 from Sonicadvance1/gvisor_flakes
unittests/gvisor: Adds a bunch of tests to flakes
2022-10-31 21:03:38 +00:00
Ryan Houdek d8386121a8 Merge pull request #2115 from Sonicadvance1/fix_x11_thunk_recursion
Thunks/libX11: Fix recursive initialize
2022-10-31 13:41:07 -07:00
Ryan Houdek 000677abb6 Merge pull request #2078 from Sonicadvance1/fix_48bit_va_stack
Allocator: Expand stack space when stealing virtual address space
2022-10-31 13:11:20 -07:00
Ryan Houdek 64eb87e9b5 Merge pull request #2099 from Sonicadvance1/fix_infinite_loop
FEXServer: Be robust against invalid packets.
2022-10-31 13:11:13 -07:00
Ryan Houdek aa5e92bee2 Merge pull request #2083 from Sonicadvance1/fix_x87_flag_range
X87: Claim incoming float was in the range for trancendental ops
2022-10-31 13:10:42 -07:00
Ryan Houdek 0bf79dc5d6 unittests/gvisor: Adds a bunch of tests to flakes
These are getting annoying.
2022-10-31 12:53:04 -07:00
Ryan Houdek adb2171c0a Thunks/libX11: Fix recursive initialize
Fixes a crash that occurs due to `_XInitDisplayLock` due to the display
lock function being initialized to our own handler.

Once XInitThreads is called once then it becomes a no-op.

steamwebhelper was hitting this.
2022-10-31 12:38:12 -07:00
Ryan Houdek d6f8923f86 X87: Claim incoming float was in the range for trancendental ops
We don't detect the range of the long F80, so we need to set that the
source was in range to fix sin/cos/tan calculations.

If we don't set this flag to zero then glibc will do some additional
operations that causes the value to be incorrect.

Fixes the output of the test application in #2021, probably fixes some
camera orientation problems in games as well.
2022-10-31 12:36:49 -07:00
Ryan Houdek cf91ab9d5f Merge pull request #2123 from Sonicadvance1/fix_32bit_vdso
32bit: Fixes Debug build of VDSO
2022-10-31 12:17:10 -07:00
Ryan Houdek a0fb9531db FEXServer: Be robust against invalid packets.
Chrome seems to like sending us invalid packets of data sometimes. With
an invalid packet type just skip parsing the data entirely.

Fixes an infinite loop in Vampire Survivors.
2022-10-31 12:05:48 -07:00
Ryan Houdek eca9353b28 Merge pull request #2122 from Sonicadvance1/fix_rotate_right_of
OpcodeDispatcher: Fixes ROR imm OF calculation
2022-10-31 11:52:49 -07:00
Ryan Houdek a259730639 32bit: Fixes Debug build of VDSO
This was generating GOT prologues even on naked functions which was
breaking VDSO on 32-bit.

Fixes almost every 32-bit application when running with debug options.
2022-10-31 11:52:17 -07:00
Ryan Houdek 2e93d10eba OpcodeDispatcher: Fixes ROR imm OF calculation
Turns out this was calculating OF incorrectly, breaking Denuvo early in
its execution.

Changes the ROL imm OF calculation code as well to be more consistent
and not keep src1 alive longer than it needs to be.

Also adds two new unit tests to ensure this stays correct.
2022-10-31 10:28:47 -07:00
Mai 70a3ceb64e Merge pull request #2096 from Sonicadvance1/cleanup_64allocator
Utils/64BitAllocator: Minor cleanups and optimization for munmap
2022-10-31 16:47:53 +00:00
Mai b726f60afd Merge pull request #2098 from Sonicadvance1/fprem_tests
unittests/asm: Adds more extensive FPREM/FPREM1 tests
2022-10-31 16:38:42 +00:00
Mai 2fa1a64999 Merge pull request #2120 from Sonicadvance1/fix_proton_experimental_48bit
ELFCodeLoader: Fixes Proton Experimental on 48-bit VA systems
2022-10-31 16:38:07 +00:00
Ryan Houdek a42b659af9 ELFCodeLoader: Fixes Proton Experimental on 48-bit VA systems
This is a tricky situation that wine-preloader allocates the lower
32MB of stack space through fixed address mmap with MAP_FIXED.

They can't use mmap with an address hint nor MAP_FIXED_NOREPLACE because
it changes behaviour. mmap won't give you the allocation inside the
stack space even if you check `/proc/self/maps` that space isn't yet
allocated. The growable space of the stack blocks those allocations.

So the wine peeps might be SOL if they actually require this allocation
to exist.

To replicate this, allocate the application stack at the same location using an address hint.
This will give us the correct region on a 48-bit VA system, while also
letting it select a different region on a 36-bit VA system.
2022-10-28 02:30:18 -07:00
Ryan Houdek 004c3230a4 Merge pull request #2108 from Sonicadvance1/implement_thunk_disables
Thunks: Add support for disabling thunks in config
2022-10-26 23:38:55 -07:00
Ryan Houdek 2332c41510 Merge pull request #2119 from lioncash/tbl
IR: Handle 256-bit VTBL1
2022-10-26 20:44:23 -07:00
lioncash ec3039c5a2 IR: Handle 256-bit VTBL1
Extends VTBL1 to handle 256-bit vectors.
2022-10-27 00:35:45 +00:00
Ryan Houdek 639d6e6071 Merge pull request #2118 from lioncash/prfx
Arm64/VectorOps: Make use of MOVPRFX where applicable
2022-10-26 15:25:51 -07:00
lioncash cd518d4726 Arm64/VectorOps: Make use of MOVPRFX where applicable
Allows hardware to pack the move and following destructive operation
together into one constructive operation if possible.

e.g.

movprfx VTMP1.D, VectorLower.D
addp VTMP1.B, Pred, VTMP1.B, VectorUpper.B

is allowed to be merged as if it executed constructively like:

addp VTMP1.B, Pred, VectorLower.B, VectorUpper.B

if the hardware supports it. If it doesn't, then the instructions will
behave like a regular move and destructive addp operation separately.
2022-10-26 21:38:48 +00:00
Ryan Houdek b7d9c00dff Merge pull request #2117 from lioncash/ins
IR: Handle 256-bit VInsGPR
2022-10-26 13:22:35 -07:00
lioncash 4b17575f5a IR: Handle 256-bit VInsGPR
Extends VInsGPR to handle 256-bit vectors.
2022-10-26 19:45:27 +00:00
Ryan Houdek 5ba4bba138 Thunks: Add support for disabling thunks in config
Previously the config options could only have ever enabled thunks rather than
disable them.

Now sort the code so it can enable thunks, then following configs can
redisable them.  Allowing testing with global thunks enabled and
disabling problematic applications.

Also sorts the "ThunkConfigFile" config as lower priority than the
application configs. I wasn't thinking about ordering that hard for
these five configuration paths, but application configs should be higher
priority in this case.
2022-10-26 11:43:16 -07:00
Ryan Houdek b3ee5dba0f Merge pull request #2116 from lioncash/extract
IR: Handle 256-bit VExtractToGPR
2022-10-26 11:16:45 -07:00
lioncash d87ff5afa9 IR: Handle 256-bit VExtractToGPR
Extends VExtractToGPR to handle 256-bit vectors.
2022-10-26 17:43:31 +00:00
Ryan Houdek 62a24bd38f Merge pull request #2075 from Sonicadvance1/gpuvis_profiler
FEXCore: Adds support for a timeline profiler interface
2022-10-26 08:44:54 -07:00
Ryan Houdek 8d373c15b8 Merge pull request #2107 from Sonicadvance1/sort_and_upgrade_x11_thunk
Thunks/X11: Reorder and sort X11 interface by headers included.
2022-10-26 04:53:48 -07:00
Ryan Houdek 671f3e74a4 Merge pull request #2103 from Sonicadvance1/sse2_for_guest
Thunks/Guest: Enable SSE2 on thunks and set fpmath to sse
2022-10-26 04:51:35 -07:00
Ryan Houdek 7e810233d9 Merge pull request #2112 from lioncash/ftoi
IR: Handle 256-bit Vector_FToI
2022-10-26 00:07:04 -07:00
Ryan Houdek 4700dbd676 Thunks/Guest: Enable SSE2 on thunks and set fpmath to sse
Clang thunks already have these default enabled, but let's also enable
this on the GCC side.

sse2 will enable most things we care about, which matches ASIMD quite
closely.
fpmath=sse removes some x87 usage for 32-bit thunks specifically.

Should effectively be a non-functional-change
2022-10-26 00:05:16 -07:00
Ryan Houdek 74e18f4317 Merge pull request #2114 from lioncash/vec
Arm64/BranchOps: Remove unused std::vector
2022-10-25 22:02:09 -07:00
lioncash 1eea95cf18 Arm64/BranchOps: Remove unused std::vector
Removes a heap allocation for inline syscalls.
2022-10-26 04:34:57 +00:00
Ryan Houdek 7291b10727 Merge pull request #2113 from lioncash/scvtf
IR: Check for invalid conversion masks in Float_FromGPR_S
2022-10-25 21:32:43 -07:00
lioncash 6804916697 IR: Check for invalid conversion masks in Float_FromGPR_S
Previously this would silently ignore unhandled masks.
2022-10-26 03:59:25 +00:00
lioncash 819e61bf14 IR: Handle 256-bit Vector_FToI
Expands Vector_FToI to handle 256-bit vectors.
2022-10-26 03:42:23 +00:00
Ryan Houdek b8f7e4c8ec Merge pull request #2111 from lioncash/ftof
IR: Handle 256-bit Vector_FtoF
2022-10-25 20:17:26 -07:00
lioncash 17bcc0eed4 IR: Handle 256-bit Vector_FtoF
Extends Vector_FtoF to handle 256-bit vectors.
2022-10-26 02:58:06 +00:00
Ryan Houdek 13003da289 Merge pull request #2110 from lioncash/ftozs
IR: Handle 256-bit Vector_FToZS/Vector_FToS
2022-10-25 19:10:45 -07:00
lioncash 9273538955 IR: Handle 256-bit Vector_FToS
Extends Vector_FToS to handle 256-bit vectors.
2022-10-26 00:59:11 +00:00
lioncash 9750189def IR: Handle 256-bit Vector_FToZS
Extends Vector_FToZS to handle 256-bit vectors.
2022-10-26 00:53:53 +00:00
Ryan Houdek cb17ee9871 Merge pull request #2109 from lioncash/stof
IR: Handle 256-bit Vector_SToF
2022-10-25 17:14:16 -07:00
lioncash 4c3b78ba9a IR: Handle 256-bit Vector_SToF
Extends Vector_SToF to handle 256-bit vectors.
2022-10-25 23:54:41 +00:00
Ryan Houdek e00b6a401b Thunks/X11: Reorder and sort X11 interface by headers included.
Each one of these are sorted through the DefinitionExtracy.py script
running over a temporary header file for each set of includes.

eg:
```bash
$ cat test.h
 #include <X11/Xproto.h>
 #include <X11/XKBlib.h>
 #include <X11/Xlib.h>
 #include <X11/Xutil.h>
 #include <X11/Xresource.h>

 #include <X11/ImUtil.h>
$ ./Scripts/DefinitionExtract.h test.h > out.txt
```

Any custom defined types have been sorted appropriately.
A bunch of missing XKB definitions were missing and added in the
process.
I've had this stashed in my git stash for a while now, I just haven't
cleaned it up.

Fixes a bunch of thunks around X11 applications missing symbols.
2022-10-25 15:43:02 -07:00
Ryan Houdek ac0ab8a7b4 Thunks/X11: Ensure 11 headers are included with C linkage
Otherwise the compiler gets confused about some functions getting
declared with C++ linkage.
2022-10-25 15:32:40 -07:00
Ryan Houdek 0aff3941f4 Scripts/DefinitionExtract: Fixes some more function attributes
X11 has an attribute that was causing function declarations to be
missed.

These definitions exist in XLibint.h
eg:
```cpp
extern void _XEatData(
    Display*		/* dpy */,
    unsigned long	/* n */
) _X_COLD;
```

This `_X_COLD` attribute was causing these function definitions to get
missed.
2022-10-25 15:32:40 -07:00
Ryan Houdek 27b022d4d9 Merge pull request #2106 from lioncash/dup
IR: Handle 256-bit VDupElement
2022-10-25 13:49:29 -07:00
lioncash e188928742 IR: Handle 256-bit VDupElement
Extends VDupElement to handle 256-bit vectors.
2022-10-25 20:01:53 +00:00
Ryan Houdek 780e3c7fb7 Merge pull request #2105 from lioncash/unzip
IR: Handle 256-bit VUnZip/VUnZip2
2022-10-25 12:45:05 -07:00
lioncash 1b5146d3ac IR: Handle 256-bit VUnZip2
Extends VUnZip2 to handle 256-bit vectors.
2022-10-25 16:01:55 +00:00
lioncash 80cf3ca6b9 IR: Handle 256-bit VUnZip
Extends VUnZip to handle 256-bit vectors.
2022-10-25 15:49:02 +00:00
Ryan Houdek 2272b30a91 Merge pull request #2101 from Sonicadvance1/fix_thunk_versions
Thunks: Fixes missing thunk librarie so versions
2022-10-24 23:11:54 -07:00
Ryan Houdek b5fb1cb07c Merge pull request #2100 from Sonicadvance1/fix_thunk_loaded_check
ThunksDB: Fixes Thunks loaded boolean pointer check
2022-10-24 23:11:33 -07:00
Ryan Houdek 4ea34a9c22 Thunks: Fixes missing thunk librarie so versions
Some libraries were missing these version defines, which was causing
dlopen to fail.

This was causing thunks to break in pressure-vessel.
2022-10-24 20:59:54 -07:00
Ryan Houdek 48e7de9f9e ThunksDB: Fixes Thunks loaded boolean pointer check
Need to dereference the boolean to ensure we only load the thunksDB
files once.
2022-10-24 20:55:58 -07:00
Ryan Houdek 76dd2369a7 unittests/asm: Adds more extensive FPREM/FPREM1 tests
unit tests that show the difference of output between FPREM and FPREM1.
Setup as known failures on everything except for host since we don't
implement fprem correctly.

An incorrect fix to FPREM is as follows:
```diff
--- a/External/FEXCore/Source/Common/SoftFloat.h
+++ b/External/FEXCore/Source/Common/SoftFloat.h
@@ -158,6 +158,10 @@ struct X80SoftFloat {

     return Result;
 #else
+    BIGFLOAT lhs_ = lhs;
+    BIGFLOAT rhs_ = rhs;
+    BIGFLOAT Result = fmodl(lhs_, rhs_);
+    return Result;
     return extF80_rem(lhs, rhs);
 #endif
   }
```

But we shouldn't implement this fix. We should instead implement a new `extF80_mod`
function that handles the rounding differences between FPREM and FPREM1.

Fixes #2097.
Doesn't attempt to resolve #1538
2022-10-22 20:47:24 -07:00
Ryan Houdek 78e0cd6e77 Scripts: Updates testharness_runner to support runner specific known failures 2022-10-22 20:29:51 -07:00
Ryan Houdek 5514a04cb4 Utils/64BitAllocator: Minor cleanups and optimization for munmap
- Some minor cleanups in the VMARegion struct type.
- Switches over to using the FlexBitSet range scanning, based off this
implementation.
- Move memory region allocation to its own function instead of
  constructor
  - This will be used by a new constructor later for 48-bit host-side
    allocations
- Minor optimization to keep track of Munmap.
  - We were burning a bunch of time on backward scanning for free
    regions even though we never did a munmap to free anything.
  - Now only do backward scanning if a munmap occured.
  - Saves a bunch of CPU time
2022-10-21 21:07:09 -07:00
Ryan Houdek 99ca78b235 Utils/FlexBitSet: Adds range scanning functions
These were currently living in the 64BitAllocator class but can be moved
directly to the FlexBitSet.

Ideally in the future these routines can be optimized so our allocator
is faster but for now these are just moved.
2022-10-21 20:58:26 -07:00
Ryan Houdek ab45db1665 Merge pull request #2094 from lioncash/zip
IR: Handle 256-bit VZip/VZip2
2022-10-20 19:02:29 -07:00
Ryan Houdek bb38bcb67d Merge pull request #2093 from lioncash/shrn
IR: Handle 256-bit VUShrNI/VUShrNI2
2022-10-20 19:00:51 -07:00
Ryan Houdek 6cc2912542 Merge pull request #2092 from lioncash/vsqxtun
IR: Handle 256-bit VSQXTUN/VSQXTUN2
2022-10-20 18:57:24 -07:00
lioncash 340b2ca624 IR: Handle 256-bit VZip2
Extends VZip2 to handle 256-bit vectors.
2022-10-20 21:15:18 +00:00
lioncash 5baa15de03 IR: Handle 256-bit VZip
Extends VZip to handle 256-bit vectors.
2022-10-20 19:22:07 +00:00
lioncash 6ddca804d1 IR: Handle 256-bit VUShrNI2
Extends VUShrNI2 to handle 256-bit vectors.
2022-10-20 18:31:36 +00:00
lioncash f9831a85fb IR: Handle 256-bit VUShrNI
Extends VUShrNI to handle 256-bit vectors.
2022-10-20 17:51:47 +00:00
lioncash 7261033b7f IR: Handle 256-bit VSQXTUN2
Extends VSQXTUN2 to handle 256-bit vectors.
2022-10-20 17:08:00 +00:00
lioncash 3ad6866198 IR: Handle 256-bit VSQXTUN
Extends VSQXTUN to handle 256-bit vectors.
2022-10-20 16:51:53 +00:00
Ryan Houdek 1c7d4165ab Merge pull request #2091 from lioncash/vsqxtn
IR: Handle 256-bit VSQXTN/VSQXTN2
2022-10-19 20:46:53 -07:00
Ryan Houdek 3e48b1a8ac FEXCore: Adds support for a timeline profiler interface
This creates a generic interface that FEXCore can use for timeline
profiling. This allows us to create a generic interface which the
backend details are hidden so we can support multiple timeline profile
APIs.

The only API supported right now is ftrace/gpuvis. Which is extremely
lightweight of an interface with minimal overhead.

We must be careful here since in most cases will will have dozens of
FEX instances running at any given time. So a timeline profiler like
Microprofiler can have major issues since that only ever expects a
single process at a time.

Not enabled by default but just needs the `ENABLE_FEXCORE_PROFILER`
cmake option set to enable.
2022-10-19 19:56:35 -07:00
lioncash 07be100daf IR: Handle 256-bit VSQXTN2
Extends VSQXTN2 to handle 256-bit vectors.
2022-10-20 02:41:45 +00:00
lioncash de9351eefb IR: Handle 256-bit VSQXTN
Expands VSQXTN to handle 256-bit vectors.
2022-10-20 02:04:05 +00:00
Ryan Houdek 2c44b5b3a1 Allocator: Expand stack space when stealing virtual address space
If we take all of the stack space then the auto expanding stack doesn't
work and we get stuck with a small stack that breaks thunks.
2022-10-19 19:02:41 -07:00
Ryan Houdek 136f1e2fc7 Merge pull request #2090 from lioncash/sxtl
Arm64/VectorOps: Simplify SVE VSXTL/VSXTL2/VUXTL/VUXTL2 implementations
2022-10-19 17:02:17 -07:00
lioncash ad39add55f Arm64/VectorOps: Simplify VUXTL2 SVE implementation
Turns out there's an instruction that does what we need, but isn't named
similarly to UXTL2 at all.
2022-10-19 22:24:07 +00:00
lioncash f3c301e359 Arm64/VectorOps: Simplify VUXTL SVE implementation
Turns out there's an instruction that does what we need, but has a name
not similar to UXTL
2022-10-19 22:22:23 +00:00
lioncash 1c37a1b4d6 Arm64/VectorOps: Simplify VSXTL2 SVE implementation
Turns out there's a built-in instruction that does exactly what we want,
but just has a different name from SXTL2
2022-10-19 22:16:33 +00:00
lioncash 7222529904 Arm64/VectorOps: Simplify VSXTL SVE implementation
Was reading the ARM ARM and realized there's an instruction that does
exactly what we need right out of the box.
2022-10-19 22:12:22 +00:00
Ryan Houdek f26eccd00f Merge pull request #2089 from wannacu/main
Implements DAA, DAS, AAA, AAS, AAM and AAD instruction
2022-10-19 03:57:13 -07:00
wannacu 73375a76ac unittests: Adds DAA, DAS, AAA, AAS, AAM and AAD unit test 2022-10-19 13:56:21 +08:00
wannacu d4416d200e OpcodeDispatcher: Implements DAA, DAS, AAA, AAS, AAM and AAD instruction 2022-10-19 13:56:06 +08:00
Mai d1b235dd83 Merge pull request #2080 from Sonicadvance1/fix_64bit_syscall_mman
Syscalls: Fixes 64-bit mmap and munmap
2022-10-19 01:26:15 +00:00
Ryan Houdek 3ac5e0423a Merge pull request #2088 from lioncash/vsmull
IR: Handle 256-bit VSMull/VSMull2
2022-10-18 16:11:03 -07:00
Ryan Houdek fc6de5f3c0 Merge pull request #2087 from lioncash/vixl-narrow
External: Update vixl submodule
2022-10-18 16:08:31 -07:00
Mai b1e475d81d Merge pull request #2081 from Sonicadvance1/fix_rotate_flags
OpcodeDispatcher: Fixes flag calculation on ROR and ROL by immediate
2022-10-18 22:33:47 +00:00
Mai 4a09a4324f Merge pull request #2082 from Sonicadvance1/fix_c2_fprem1
OpcodeDispatcher: Fixes FPREM1 C2 flag calculation
2022-10-18 22:33:29 +00:00
lioncash 47f94327c5 IR: Amend x86_64 32->64 case for VUMull2
Realized I forgot to amend the registers used in the final multiply.
2022-10-18 16:23:16 +00:00
lioncash a009ed0b6b IR: Handle 256-bit VSMull2
Extends VSMull2 to handle 256-bit vectors.
2022-10-18 16:23:14 +00:00
lioncash 2476a686e7 IR: Handle 256-bit VSMull
Extends VSMull to handle 256-bit vectors.
2022-10-18 16:22:44 +00:00
lioncash f0db93773f unittests: Re-enable narrowing and widening tests
Now that the bug in vixl's simulator is fixed, we can enable these tests
again.
2022-10-18 15:18:14 +00:00
lioncash fabe824c8b Externals: Update vixl submodule
Includes fixes for the narrowing instructions.
2022-10-18 15:15:54 +00:00
Ryan Houdek 78a077397e Merge pull request #2085 from lioncash/vmull
IR: Handle 256-bit VUMull/VUMull2
2022-10-17 18:10:48 -07:00
lioncash 0c4b456aaa IR: Handle 256-bit VUMull2
Extends VUMull2 to handle 256-bit vectors.
2022-10-17 19:03:26 +00:00
lioncash 09185167bc OpcodeDispatcher/Vector: Amend and simplify PMULLOp
Allows PMULLOp to function correctly with the amended VPSHUFD entries.

Since this is only used to perform expanded multiplication from 32-bit
entries to 64-bit entries, we can simplify things a little bit.

All we need to do is yank the third 32-bit word down into the second
32-bit word's spot in the vector and let the VUMull/VSMull IR ops handle
it.
2022-10-17 19:03:26 +00:00
lioncash 5d78c3203c IR: Handle 256-bit VUMull
Extends VUMull to handle 256-bit vectors.

While we're at it, we can fix a typo in the VPSHUFD called for the
32->64-bit case.

To mirror UMULL, we need to replicate element 0 and 1, not 0 and 2

While we're at it, we can fix this with VSMull as well.
2022-10-17 19:03:23 +00:00
Ryan Houdek fa5322d3f9 Merge pull request #2084 from Sonicadvance1/more_auxv
ELFCodeLoader: Implement four more auxv values
2022-10-17 09:48:18 -07:00
Ryan Houdek d21aa5cac2 ELFCodeLoader: Implement four more auxv values
Implements AT_PLATFORM: Ends up being `i686` or `x86_64` depending on
ELF arch

Implements AT_HWCAP and AT_HWCAP2
AT_HWCAP is just CPUID function 01h EDX result
AT_HWCAP2 only has two defined bits in it, which we don't support
either.

Implements AT_RANDOM
Previously we were just sticking hardcoded values in to this.
Now we pass along the host's AT_RANDOM, or we generate our own if that
doesn't exist

Fixes #788
2022-10-16 19:45:19 -07:00
Ryan Houdek 102d5c57cb OpcodeDispatcher: Fixes FPREM1 C2 flag calculation
Accidentally didn't implement this for FPREM1 but it /was/ implemented
for FPREM. Fixes an infinite loop in cossin implementations.

Test code from the application returns an incorrect result, but it isn't
due to FPREM1.

```
$ `which wine` ./hello.exe
Sin 1.22460635382238E-16
Cos -1
$ FEXInterpreter `which wine` ./hello.exe
Sin 1.22460635382238E-16
Cos 0.54030230586814
```

Fixes #2021
2022-10-16 15:24:46 -07:00
Ryan Houdek ffb4de9fd9 Merge pull request #2079 from Sonicadvance1/ensure_armemitter_uses_allocator
Ensure Arm64Emitter uses FEX allocator
2022-10-15 15:37:23 -07:00
Ryan Houdek 6b3d8886e5 Merge pull request #2077 from Sonicadvance1/fix_thunks_with_lots_args
Thunks: Fixes indirect thunks with 8+ arguments
2022-10-15 15:37:09 -07:00
Ryan Houdek ddc10272a0 Merge pull request #2076 from Sonicadvance1/update_vulkan
Thunks: Update Vulkan thunk to v1.3.231
2022-10-15 15:14:30 -07:00
Ryan Houdek 0b5ef00165 Thunks: Fixes indirect thunks with 8+ arguments
Due to how we use a modified ABI for these indirect functions, we don't
have a clean way to say that the host_addr lives in a side-argument.

The previous inline asm that moved the value from r11 in to a variable
worked up until you hit functions with 8 or more arguments. At that
point the compiler was generating code before our inline assembly and
using r11 as a temporary, thus destroying our value.
Then a crash would occur and it was very hard to determine why. It would
end up calling some random function (0x1 in this case) from an indirect
call.

This made it /look/ like it was calling an invalid function returned
from the loader but in reality it was a corrupt register loading bad
data.

To work around this case, we can use an inline asm register variable and
a volatile asm block that "sets" the variable. In this case GCC and
Clang both seem to extend the live range of the register from the start
of the function to the use of the variable.

This resolves the issue for now, and I tested quite a large number of
function signatures to see if it would break in the future.

Theoretically our functional testing should catch this, but we don't
currently have something that abuses all the functions like this
currently.
2022-10-15 15:13:40 -07:00
Ryan Houdek 2a50416fc3 unittests: Ensures overloaded shifts don't result in JIT failure 2022-10-14 23:38:35 -07:00
Ryan Houdek ce514d9f83 unittests: Adds ROL and ROR CF flag calculation tests
This would have failed prior to the last commit
2022-10-14 23:37:46 -07:00
Ryan Houdek 9b77e7fd13 OpcodeDispatcher: Fixes flag calculation on ROR and ROL by immediate
These were being calculated incorrectly in the case of rotating with
values larger than 8-bit or 16-bit
2022-10-14 23:36:48 -07:00
Ryan Houdek abb44d3327 Merge pull request #2069 from wannacu/main
Flags: Refine _Bfe's shift
2022-10-14 22:53:05 -07:00
Ryan Houdek 11eaf3d48a Ensure Arm64Emitter uses FEX allocator
Otherwise we will end up allocating code buffers in the lower 32-bits,
consuming precious virtual address space.
2022-10-14 22:04:07 -07:00
Ryan Houdek 76c2cc2c3e Syscalls: Fixes 64-bit mmap and munmap
These should be using the real syscalls, not our provided allocators.

While not a problem currently since these redirect to host mmap and
munmap, it will become an issue once we have an allocator that lives
outside of x86-64 space.
2022-10-14 21:43:04 -07:00
Ryan Houdek 0e6c8bd12e Thunks: Update Vulkan thunk to v1.3.231
Only missing a few function definitions, resorted to match order of
definitions in the headers so future changes don't mix up as much
2022-10-14 01:51:35 -07:00
Ryan Houdek a9fb008317 External: Update Vulkan-Headers to v1.3.231 2022-10-14 01:50:49 -07:00
Ryan Houdek e9f3a5b3e4 Merge pull request #2074 from lioncash/vuxtl
IR: Handle 256-bit VUXTL/VUXTL2
2022-10-13 13:15:47 -07:00
Ryan Houdek ebc45dff45 Merge pull request #2070 from lioncash/vuabdl
IR: Handle 256-bit VUABDL
2022-10-13 12:17:47 -07:00
lioncash fc4a5ebfd3 IR: Handle 256-bit VUXTL2
Extends VUXTL2 to handle 256-bit vectors.
2022-10-13 19:15:05 +00:00
lioncash 1d7b688c55 IR: Handle 256-bit VUXTL
Extends VUXTL to handle 256-bit values.
2022-10-13 19:07:57 +00:00
Ryan Houdek f14a5ffbbf Merge pull request #2073 from lioncash/vsxtl
IR: Handle 256-bit VSXTL/VSXTL2
2022-10-13 11:53:12 -07:00
Ryan Houdek cada0d593c Merge pull request #2072 from lioncash/test
unittests: Amend mm register usage in H0F38/66_04.asm test
2022-10-13 11:28:27 -07:00
lioncash 0436540791 IR: Handle 256-bit VSXTL2
Extends VSXTL2 to handle 256-bit vectors.
2022-10-13 18:21:25 +00:00
lioncash a87ac86e18 IR: Handle 256-bit VSXTL
Extends VSXTL to handle 256-bit vectors.
2022-10-13 18:21:22 +00:00
lioncash 1278b23150 unittests: Amend mm register usage in H0F38/66_04.asm test
This should be using xmm2 rather than mm2.
2022-10-13 17:09:40 +00:00
lioncash 02f5ea4b9d IR: Handle 256-bit VUABDL
Extends VUABDL to handle 256-bit vectors.
2022-10-13 15:37:10 +00:00
wannacu 2e14e613d0 Flags: Refine _Bfe's shift 2022-10-13 16:41:04 +08:00
106 changed files with 7719 additions and 3092 deletions

No files matched your search

+13
View File
@@ -29,11 +29,24 @@ option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
if (ENABLE_FEXCORE_PROFILER)
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
else()
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
endif()
endif()
# uninstall target
if(NOT TARGET uninstall)
configure_file(
+1
View File
@@ -143,6 +143,7 @@ set (SRCS
Utils/NetStream.cpp
Utils/Telemetry.cpp
Utils/Threads.cpp
Utils/Profiler.cpp
)
if (ENABLE_INTERPRETER)
@@ -20,7 +20,9 @@ namespace FEXCore::CPU {
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
: vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode)
: vixl::aarch64::Assembler(size ? (byte*)FEXCore::Allocator::mmap(nullptr, size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0) : reinterpret_cast<byte*>(~0ULL),
size,
vixl::aarch64::PositionDependentCode)
, EmitterCTX {ctx} {
CPU.SetUp();
@@ -37,6 +39,13 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
SetCPUFeatures(Features);
}
Arm64Emitter::~Arm64Emitter() {
auto CodeBuffer = GetBuffer();
if (CodeBuffer->GetCapacity()) {
FEXCore::Allocator::munmap(CodeBuffer->GetStartAddress<void*>(), CodeBuffer->GetCapacity());
}
}
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad) {
bool Is64Bit = Reg.IsX();
int Segments = Is64Bit ? 4 : 2;
@@ -88,6 +88,7 @@ const std::array<aarch64::VRegister, 12> RAFPR = {
class Arm64Emitter : public vixl::aarch64::Assembler {
protected:
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
~Arm64Emitter();
FEXCore::Context::Context *EmitterCTX;
vixl::aarch64::CPU CPU;
+6
View File
@@ -44,6 +44,7 @@ $end_info$
#include <FEXCore/Utils/Event.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Threads.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXHeaderUtils/Syscalls.h>
#include <FEXHeaderUtils/TodoDefines.h>
@@ -674,6 +675,8 @@ namespace FEXCore::Context {
}
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) {
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
{
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
// Use the thread's object cache ref counter for this
@@ -741,6 +744,8 @@ namespace FEXCore::Context {
}
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
FEXCORE_PROFILE_SCOPED("GenerateIR");
Thread->OpDispatcher->ReownOrClaimBuffer();
Thread->OpDispatcher->ResetWorkingList();
@@ -1011,6 +1016,7 @@ namespace FEXCore::Context {
}
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
FEXCORE_PROFILE_SCOPED("CompileBlock");
auto Thread = Frame->Thread;
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
@@ -212,12 +212,20 @@ void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread,
Frame->State.flags[9] = 1;
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
Frame->State.cs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
Frame->State.ds = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
Frame->State.es = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
Frame->State.fs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
Frame->State.gs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
Frame->State.ss = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
Frame->State.cs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
Frame->State.ds_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
Frame->State.es_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
Frame->State.fs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
Frame->State.gs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
Frame->State.ss_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
Frame->State.cs_cached = Frame->State.gdt[Frame->State.cs_idx >> 3].base;
Frame->State.ds_cached = Frame->State.gdt[Frame->State.ds_idx >> 3].base;
Frame->State.es_cached = Frame->State.gdt[Frame->State.es_idx >> 3].base;
Frame->State.fs_cached = Frame->State.gdt[Frame->State.fs_idx >> 3].base;
Frame->State.gs_cached = Frame->State.gdt[Frame->State.gs_idx >> 3].base;
Frame->State.ss_cached = Frame->State.gdt[Frame->State.ss_idx >> 3].base;
#define COPY_REG(x) \
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
COPY_REG(RDI);
@@ -565,10 +573,13 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
SetXStateInfo(xstate, IsAVXEnabled);
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs_idx;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss_idx;
if (ContextBackup->FaultToTopAndGeneratedException) {
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
@@ -581,10 +592,8 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
}
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = 0;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss;
#define COPY_REG(x) \
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
+2 -1
View File
@@ -18,6 +18,7 @@ $end_info$
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <set>
@@ -1132,6 +1133,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
}
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
Blocks.clear();
BlocksToDecode.clear();
HasBlocks.clear();
@@ -1166,7 +1168,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
std::set<uint64_t> CodePages = { CurrentCodePage };
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
while (!BlocksToDecode.empty()) {
auto BlockDecodeIt = BlocksToDecode.begin();
+35 -17
View File
@@ -894,33 +894,51 @@ DEF_OP(Select) {
}
DEF_OP(VExtractToGPR) {
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
const auto OpSize = IROp->Size;
constexpr auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
constexpr auto SSEBitSize = SSERegSize * 8;
const auto ElementSize = Op->Header.ElementSize;
const auto ElementSizeBits = ElementSize * 8;
const auto Shift = ElementSizeBits * Op->Index;
const uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Vector);
LOGMAN_THROW_AA_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
LOGMAN_THROW_AA_FMT(OpSize <= AVXRegSize,
"OpSize is too large for VExtractToGPR: {}", OpSize);
if (SourceSize == 16) {
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
if (Op->Header.ElementSize == 8)
if (SourceSize >= SSERegSize) {
__uint128_t SourceMask = (1ULL << ElementSizeBits) - 1;
if (ElementSize == 8) {
SourceMask = ~0ULL;
}
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
Src >>= Shift;
Src &= SourceMask;
memcpy(GDP, &Src, Op->Header.ElementSize);
const auto Src = *GetSrc<InterpVector256*>(Data->SSAData, Op->Vector);
const auto GetResult = [&] {
if (Shift >= SSEBitSize) {
const auto NormalizedShift = Shift - SSEBitSize;
return (Src.Upper >> NormalizedShift) & SourceMask;
} else {
return (Src.Lower >> Shift) & SourceMask;
}
};
const auto Result = GetResult();
memcpy(GDP, &Result, ElementSize);
}
else {
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
if (Op->Header.ElementSize == 8)
uint64_t SourceMask = (1ULL << ElementSizeBits) - 1;
if (ElementSize == 8) {
SourceMask = ~0ULL;
}
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
Src >>= Shift;
Src &= SourceMask;
GD = Src;
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
const uint64_t Result = (Src >> Shift) & SourceMask;
GD = Result;
}
}
@@ -13,22 +13,46 @@ $end_info$
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(VInsGPR) {
auto Op = IROp->C<IR::IROp_VInsGPR>();
const uint8_t OpSize = IROp->Size;
const auto Op = IROp->C<IR::IROp_VInsGPR>();
const auto OpSize = IROp->Size;
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->DestVector);
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
const auto ElementSize = Op->Header.ElementSize;
const auto ElementSizeBits = ElementSize * 8;
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
if (Op->Header.ElementSize == 8) {
const uint64_t Offset = Op->DestIdx * ElementSizeBits;
const auto InUpperLane = Offset >= SSEBitSize;
__uint128_t Mask = (1ULL << ElementSizeBits) - 1;
if (ElementSize == 8) {
Mask = ~0ULL;
}
Src2 = Src2 & Mask;
Mask <<= Offset;
const auto Src1 = *GetSrc<InterpVector256*>(Data->SSAData, Op->DestVector);
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
const auto Scalar = Src2 & Mask;
const auto ScaledOffset = InUpperLane ? Offset - SSEBitSize
: Offset;
// Now shift into place and set all bits but
// the ones where we're going to insert our value.
Mask <<= ScaledOffset;
Mask = ~Mask;
__uint128_t Dst = Src1 & Mask;
Dst |= Src2 << Offset;
const auto Dst = [&] {
if (InUpperLane) {
return InterpVector256{
.Lower = Src1.Lower,
.Upper = (Src1.Upper & Mask) | (Scalar << ScaledOffset),
};
} else {
return InterpVector256{
.Lower = (Src1.Lower & Mask) | (Scalar << ScaledOffset),
.Upper = Src1.Upper,
};
}
}();
memcpy(GDP, &Dst, OpSize);
}
@@ -89,63 +113,73 @@ DEF_OP(Vector_SToF) {
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToZS) {
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToS) {
auto Op = IROp->C<IR::IROp_Vector_FToS>();
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToF) {
auto Op = IROp->C<IR::IROp_Vector_FToF>();
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
const uint16_t ElementSize = Op->Header.ElementSize;
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (Conv) {
@@ -165,19 +199,22 @@ DEF_OP(Vector_FToF) {
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv); break;
default:
LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(Vector_FToI) {
auto Op = IROp->C<IR::IROp_Vector_FToI>();
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func_Nearest = [](auto a) { return std::rint(a); };
const auto Func_Neg = [](auto a) { return std::floor(a); };
const auto Func_Pos = [](auto a) { return std::ceil(a); };
@@ -186,31 +223,31 @@ DEF_OP(Vector_FToI) {
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
}
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
}
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
}
break;
case FEXCore::IR::Round_Towards_Zero.Val:
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
}
break;
case FEXCore::IR::Round_Host.Val:
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_OP(4, float, Func_Host)
DO_VECTOR_1SRC_OP(8, double, Func_Host)
}
@@ -25,40 +25,45 @@ static inline void CacheLineFlush(char *Addr) {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(LoadContext) {
auto Op = IROp->C<IR::IROp_LoadContext>();
uint8_t OpSize = IROp->Size;
const auto Op = IROp->C<IR::IROp_LoadContext>();
const auto OpSize = IROp->Size;
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Src = ContextPtr + Op->Offset;
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += Op->Offset;
#define LOAD_CTX(x, y) \
case x: { \
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
y const *MemData = reinterpret_cast<y const*>(Src); \
GD = *MemData; \
break; \
}
switch (OpSize) {
LOAD_CTX(1, uint8_t)
LOAD_CTX(2, uint16_t)
LOAD_CTX(4, uint32_t)
LOAD_CTX(8, uint64_t)
case 16: {
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
case 16:
case 32: {
void const *MemData = reinterpret_cast<void const*>(Src);
memcpy(GDP, MemData, OpSize);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
break;
}
#undef LOAD_CTX
}
DEF_OP(StoreContext) {
auto Op = IROp->C<IR::IROp_StoreContext>();
uint8_t OpSize = IROp->Size;
const auto Op = IROp->C<IR::IROp_StoreContext>();
const auto OpSize = IROp->Size;
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += Op->Offset;
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Dst = ContextPtr + Op->Offset;
void *MemData = reinterpret_cast<void*>(ContextPtr);
void *MemData = reinterpret_cast<void*>(Dst);
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
memcpy(MemData, Src, OpSize);
}
@@ -72,46 +77,51 @@ DEF_OP(StoreRegister) {
}
DEF_OP(LoadContextIndexed) {
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
const auto OpSize = IROp->Size;
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
ContextPtr += Op->BaseOffset;
ContextPtr += Index * Op->Stride;
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Src = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
#define LOAD_CTX(x, y) \
case x: { \
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
y const *MemData = reinterpret_cast<y const*>(Src); \
GD = *MemData; \
break; \
}
switch (IROp->Size) {
switch (OpSize) {
LOAD_CTX(1, uint8_t)
LOAD_CTX(2, uint16_t)
LOAD_CTX(4, uint32_t)
LOAD_CTX(8, uint64_t)
case 16: {
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
memcpy(GDP, MemData, IROp->Size);
case 16:
case 32: {
void const *MemData = reinterpret_cast<void const*>(Src);
memcpy(GDP, MemData, OpSize);
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
break;
}
#undef LOAD_CTX
}
DEF_OP(StoreContextIndexed) {
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
const auto OpSize = IROp->Size;
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
ContextPtr += Op->BaseOffset;
ContextPtr += Index * Op->Stride;
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
void *MemData = reinterpret_cast<void*>(ContextPtr);
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
const auto Dst = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
void *MemData = reinterpret_cast<void*>(Dst);
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
memcpy(MemData, Src, IROp->Size);
memcpy(MemData, Src, OpSize);
}
DEF_OP(SpillRegister) {
@@ -144,8 +154,8 @@ DEF_OP(StoreFlag) {
}
DEF_OP(LoadMem) {
auto Op = IROp->C<IR::IROp_LoadMem>();
uint8_t OpSize = IROp->Size;
const auto Op = IROp->C<IR::IROp_LoadMem>();
const auto OpSize = IROp->Size;
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
@@ -158,7 +168,8 @@ DEF_OP(LoadMem) {
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
}
}
memset(GDP, 0, 16);
memset(GDP, 0, Core::CPUState::XMM_AVX_REG_SIZE);
switch (OpSize) {
case 1: {
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
@@ -180,16 +191,15 @@ DEF_OP(LoadMem) {
GD = D->load();
break;
}
default:
memcpy(GDP, MemData, IROp->Size);
memcpy(GDP, MemData, OpSize);
break;
}
}
DEF_OP(StoreMem) {
auto Op = IROp->C<IR::IROp_StoreMem>();
uint8_t OpSize = IROp->Size;
const auto Op = IROp->C<IR::IROp_StoreMem>();
const auto OpSize = IROp->Size;
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
@@ -221,7 +231,7 @@ DEF_OP(StoreMem) {
}
default:
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), IROp->Size);
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
break;
}
}
@@ -19,13 +19,6 @@ $end_info$
#include <sys/random.h>
namespace FEXCore::CPU {
[[noreturn]]
static void StopThread(FEXCore::Core::InternalThreadState *Thread) {
Thread->CTX->StopThread(Thread);
LOGMAN_MSG_A_FMT("unreachable");
FEX_UNREACHABLE;
}
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(Fence) {
+206 -121
View File
@@ -838,17 +838,18 @@ DEF_OP(VSMax) {
}
DEF_OP(VZip) {
auto Op = IROp->C<IR::IROp_VZip>();
const auto Op = IROp->C<IR::IROp_VZip>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
uint8_t Tmp[16];
uint8_t Elements = OpSize / Op->Header.ElementSize;
uint8_t BaseOffset = IROp->Op == IR::OP_VZIP2 ? (Elements / 2) : 0;
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
const uint8_t ElementSize = Op->Header.ElementSize;
uint8_t Elements = OpSize / ElementSize;
const uint8_t BaseOffset = IROp->Op == IR::OP_VZIP2 ? (Elements / 2) : 0;
Elements >>= 1;
switch (Op->Header.ElementSize) {
switch (ElementSize) {
case 1: {
auto *Dst_d = reinterpret_cast<uint8_t*>(Tmp);
auto *Src1_d = reinterpret_cast<uint8_t*>(Src1);
@@ -889,24 +890,27 @@ DEF_OP(VZip) {
}
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VUnZip) {
auto Op = IROp->C<IR::IROp_VUnZip>();
const auto Op = IROp->C<IR::IROp_VUnZip>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
uint8_t Tmp[16];
uint8_t Elements = OpSize / Op->Header.ElementSize;
unsigned Start = IROp->Op == IR::OP_VUNZIP ? 0 : 1;
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
const uint8_t ElementSize = Op->Header.ElementSize;
uint8_t Elements = OpSize / ElementSize;
const unsigned Start = IROp->Op == IR::OP_VUNZIP ? 0 : 1;
Elements >>= 1;
switch (Op->Header.ElementSize) {
switch (ElementSize) {
case 1: {
auto *Dst_d = reinterpret_cast<uint8_t*>(Tmp);
auto *Src1_d = reinterpret_cast<uint8_t*>(Src1);
@@ -947,7 +951,9 @@ DEF_OP(VUnZip) {
}
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
@@ -1503,16 +1509,18 @@ DEF_OP(VSShrS) {
}
DEF_OP(VInsElement) {
auto Op = IROp->C<IR::IROp_VInsElement>();
const uint8_t OpSize = IROp->Size;
const auto Op = IROp->C<IR::IROp_VInsElement>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->DestVector);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->SrcVector);
uint8_t Tmp[16];
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
// Copy src1 in to dest
memcpy(Tmp, Src1, OpSize);
switch (Op->Header.ElementSize) {
switch (ElementSize) {
case 1: {
auto *Dst_d = reinterpret_cast<uint8_t*>(Tmp);
auto *Src2_d = reinterpret_cast<uint8_t*>(Src2);
@@ -1537,43 +1545,75 @@ DEF_OP(VInsElement) {
Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx];
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
};
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VDupElement) {
auto Op = IROp->C<IR::IROp_VDupElement>();
const auto Op = IROp->C<IR::IROp_VDupElement>();
const uint8_t OpSize = IROp->Size;
const uint8_t Elements = OpSize / Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(OpSize <= 16, "OpSize is too large for VDupElement: {}", OpSize);
if (OpSize == 16) {
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
if (Op->Header.ElementSize == 8)
const uint64_t ElementSize = Op->Header.ElementSize;
const uint64_t ElementSizeBits = ElementSize * 8;
const uint8_t Elements = OpSize / ElementSize;
constexpr auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
constexpr auto SSEBitSize = SSERegSize * 8;
const auto Is128BitElement = ElementSizeBits == SSEBitSize;
const auto Is256Bit = OpSize == AVXRegSize;
LOGMAN_THROW_AA_FMT(OpSize <= AVXRegSize,
"OpSize is too large for VDupElement: {}", OpSize);
if (OpSize >= SSERegSize) {
__uint128_t SourceMask = (1ULL << ElementSizeBits) - 1;
if (ElementSize == 8) {
SourceMask = ~0ULL;
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
Src >>= Shift;
Src &= SourceMask;
for (size_t i = 0; i < Elements; ++i) {
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * i)),
&Src, Op->Header.ElementSize);
}
}
else {
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
if (Op->Header.ElementSize == 8)
SourceMask = ~0ULL;
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
Src >>= Shift;
Src &= SourceMask;
const auto GetResult = [&]() -> __uint128_t {
const auto Src = *GetSrc<InterpVector256*>(Data->SSAData, Op->Vector);
uint64_t Shift = ElementSizeBits * Op->Index;
if (Is128BitElement) {
if (Shift == 0) {
return Src.Lower;
} else {
return Src.Upper;
}
} else {
// Normalize shift to act on upper uint128_t
if (Is256Bit && Shift >= SSEBitSize) {
Shift -= SSEBitSize;
return (Src.Upper >> Shift) & SourceMask;
} else {
return (Src.Lower >> Shift) & SourceMask;
}
}
};
const __uint128_t Result = GetResult();
for (size_t i = 0; i < Elements; ++i) {
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * i)),
&Src, Op->Header.ElementSize);
auto* Dst = static_cast<uint8_t*>(GDP) + (ElementSize * i);
memcpy(Dst, &Result, ElementSize);
}
} else {
const uint64_t Shift = ElementSizeBits * Op->Index;
uint64_t SourceMask = (1ULL << ElementSizeBits) - 1;
if (ElementSize == 8) {
SourceMask = ~0ULL;
}
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
const uint64_t Result = (Src >> Shift) & SourceMask;
for (size_t i = 0; i < Elements; ++i) {
auto* Dst = static_cast<uint8_t*>(GDP) + (ElementSize * i);
memcpy(Dst, &Result, ElementSize);
}
}
}
@@ -1695,205 +1735,235 @@ DEF_OP(VShlI) {
}
DEF_OP(VUShrNI) {
auto Op = IROp->C<IR::IROp_VUShrNI>();
const auto Op = IROp->C<IR::IROp_VUShrNI>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t BitShift = Op->BitShift;
uint8_t Tmp[16]{};
const uint8_t BitShift = Op->BitShift;
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / (ElementSize << 1);
const auto Func = [BitShift](auto a, auto min, auto max) {
return BitShift >= (sizeof(a) * 8) ? 0 : a >> BitShift;
};
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(1, uint8_t, uint16_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(2, uint16_t, uint32_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(4, uint32_t, uint64_t, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VUShrNI2) {
auto Op = IROp->C<IR::IROp_VUShrNI2>();
const auto Op = IROp->C<IR::IROp_VUShrNI2>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
uint8_t BitShift = Op->BitShift;
uint8_t Tmp[16];
const uint8_t BitShift = Op->BitShift;
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / (ElementSize << 1);
const auto Func = [BitShift](auto a, auto min, auto max) {
return BitShift >= (sizeof(a) * 8) ? 0 : a >> BitShift;
};
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP_TOP(1, uint8_t, uint16_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP_TOP(2, uint16_t, uint32_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP_TOP(4, uint32_t, uint64_t, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VSXTL) {
auto Op = IROp->C<IR::IROp_VSXTL>();
const auto Op = IROp->C<IR::IROp_VSXTL>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(2, int16_t, int8_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, int16_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, int32_t, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VSXTL2) {
auto Op = IROp->C<IR::IROp_VSXTL2>();
const auto Op = IROp->C<IR::IROp_VSXTL2>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16];
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(2, int16_t, int8_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(4, int32_t, int16_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(8, int64_t, int32_t, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VUXTL) {
auto Op = IROp->C<IR::IROp_VUXTL>();
const auto Op = IROp->C<IR::IROp_VUXTL>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(2, uint16_t, uint8_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(4, uint32_t, uint16_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP(8, uint64_t, uint32_t, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VUXTL2) {
auto Op = IROp->C<IR::IROp_VUXTL2>();
const auto Op = IROp->C<IR::IROp_VUXTL2>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16];
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto min, auto max) { return a; };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(2, uint16_t, uint8_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(4, uint32_t, uint16_t, Func, 0, 0)
DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(8, uint64_t, uint32_t, Func, 0, 0)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VSQXTN) {
auto Op = IROp->C<IR::IROp_VSQXTN>();
const auto Op = IROp->C<IR::IROp_VSQXTN>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / (ElementSize << 1);
const auto Func = [](auto a, auto min, auto max) {
return std::max(std::min(a, (decltype(a))max), (decltype(a))min);
};
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(1, int8_t, int16_t, Func, std::numeric_limits<int8_t>::min(), std::numeric_limits<int8_t>::max())
DO_VECTOR_1SRC_2TYPE_OP(2, int16_t, int32_t, Func, std::numeric_limits<int16_t>::min(), std::numeric_limits<int16_t>::max())
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VSQXTN2) {
auto Op = IROp->C<IR::IROp_VSQXTN2>();
const auto Op = IROp->C<IR::IROp_VSQXTN2>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / (ElementSize << 1);
const auto Func = [](auto a, auto min, auto max) {
return std::max(std::min(a, (decltype(a))max), (decltype(a))min);
};
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP_TOP(1, int8_t, int16_t, Func, std::numeric_limits<int8_t>::min(), std::numeric_limits<int8_t>::max())
DO_VECTOR_1SRC_2TYPE_OP_TOP(2, int16_t, int32_t, Func, std::numeric_limits<int16_t>::min(), std::numeric_limits<int16_t>::max())
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VSQXTUN) {
auto Op = IROp->C<IR::IROp_VSQXTUN>();
const auto Op = IROp->C<IR::IROp_VSQXTUN>();
const uint8_t OpSize = IROp->Size;
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / (ElementSize << 1);
const auto Func = [](auto a, auto min, auto max) {
return std::max(std::min(a, (decltype(a))max), (decltype(a))min);
};
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP(1, uint8_t, int16_t, Func, 0, (1 << 8) - 1)
DO_VECTOR_1SRC_2TYPE_OP(2, uint16_t, int32_t, Func, 0, (1 << 16) - 1)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VSQXTUN2) {
auto Op = IROp->C<IR::IROp_VSQXTUN2>();
const auto Op = IROp->C<IR::IROp_VSQXTUN2>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->VectorLower);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->VectorUpper);
uint8_t Tmp[16]{};
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
const uint8_t Elements = OpSize / (Op->Header.ElementSize << 1);
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / (ElementSize << 1);
const auto Func = [](auto a, auto min, auto max) {
return std::max(std::min(a, (decltype(a))max), (decltype(a))min);
};
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_1SRC_2TYPE_OP_TOP(1, uint8_t, int16_t, Func, 0, (1 << 8) - 1)
DO_VECTOR_1SRC_2TYPE_OP_TOP(2, uint16_t, int32_t, Func, 0, (1 << 16) - 1)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
@@ -1923,22 +1993,25 @@ DEF_OP(VUMul) {
}
DEF_OP(VUMull) {
auto Op = IROp->C<IR::IROp_VUMull>();
const auto Op = IROp->C<IR::IROp_VUMull>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
uint8_t Tmp[16];
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto b) { return a * b; };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_2SRC_2TYPE_OP(2, uint16_t, uint8_t, Func)
DO_VECTOR_2SRC_2TYPE_OP(4, uint32_t, uint16_t, Func)
DO_VECTOR_2SRC_2TYPE_OP(8, uint64_t, uint32_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
@@ -1968,100 +2041,112 @@ DEF_OP(VSMul) {
}
DEF_OP(VSMull) {
auto Op = IROp->C<IR::IROp_VSMull>();
const auto Op = IROp->C<IR::IROp_VSMull>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
uint8_t Tmp[16];
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto b) { return a * b; };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_2SRC_2TYPE_OP(2, int16_t, int8_t, Func)
DO_VECTOR_2SRC_2TYPE_OP(4, int32_t, int16_t, Func)
DO_VECTOR_2SRC_2TYPE_OP(8, int64_t, int32_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VUMull2) {
auto Op = IROp->C<IR::IROp_VUMull2>();
const auto Op = IROp->C<IR::IROp_VUMull2>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
uint8_t Tmp[16];
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto b) { return a * b; };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(2, uint16_t, uint8_t, Func)
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(4, uint32_t, uint16_t, Func)
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(8, uint64_t, uint32_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VSMull2) {
auto Op = IROp->C<IR::IROp_VSMull2>();
const auto Op = IROp->C<IR::IROp_VSMull2>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
uint8_t Tmp[16];
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func = [](auto a, auto b) { return a * b; };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(2, int16_t, int8_t, Func)
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(4, int32_t, int16_t, Func)
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(8, int64_t, int32_t, Func)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, Op->Header.Size);
}
DEF_OP(VUABDL) {
auto Op = IROp->C<IR::IROp_VUABDL>();
const auto Op = IROp->C<IR::IROp_VUABDL>();
const uint8_t OpSize = IROp->Size;
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
uint8_t Tmp[16];
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
const uint8_t Elements = OpSize / Op->Header.ElementSize;
const uint8_t ElementSize = Op->Header.ElementSize;
const uint8_t Elements = OpSize / ElementSize;
const auto Func8 = [](auto a, auto b) { return std::abs((int16_t)a - (int16_t)b); };
const auto Func16 = [](auto a, auto b) { return std::abs((int32_t)a - (int32_t)b); };
const auto Func32 = [](auto a, auto b) { return std::abs((int64_t)a - (int64_t)b); };
switch (Op->Header.ElementSize) {
switch (ElementSize) {
DO_VECTOR_2SRC_2TYPE_OP(2, uint16_t, uint8_t, Func8)
DO_VECTOR_2SRC_2TYPE_OP(4, uint32_t, uint16_t, Func16)
DO_VECTOR_2SRC_2TYPE_OP(8, uint64_t, uint32_t, Func32)
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
memcpy(GDP, Tmp, OpSize);
}
DEF_OP(VTBL1) {
auto Op = IROp->C<IR::IROp_VTBL1>();
const auto Op = IROp->C<IR::IROp_VTBL1>();
const uint8_t OpSize = IROp->Size;
const auto *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->VectorTable);
const auto *Src2 = GetSrc<uint8_t*>(Data->SSAData, Op->VectorIndices);
uint8_t Tmp[16];
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
for (size_t i = 0; i < OpSize; ++i) {
const uint8_t Index = Src2[i];
+72 -18
View File
@@ -1168,25 +1168,79 @@ DEF_OP(Select) {
}
DEF_OP(VExtractToGPR) {
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
const uint8_t OpSize = IROp->Size;
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
const auto OpSize = IROp->Size;
switch (OpSize) {
case 1:
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V16B(), Op->Index);
break;
case 2:
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V8H(), Op->Index);
break;
case 4:
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V4S(), Op->Index);
break;
case 8:
umov(GetReg<RA_64>(Node), GetSrc(Op->Vector.ID()).V2D(), Op->Index);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
break;
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
const auto ElementSizeBits = Op->Header.ElementSize * 8;
const auto Offset = ElementSizeBits * Op->Index;
const auto Is256Bit = Offset >= SSERegBitSize;
const auto Vector = GetSrc(Op->Vector.ID());
const auto PerformMove = [&](const aarch64::VRegister& reg, int index) {
switch (OpSize) {
case 1:
umov(GetReg<RA_32>(Node), reg.V16B(), index);
break;
case 2:
umov(GetReg<RA_32>(Node), reg.V8H(), index);
break;
case 4:
umov(GetReg<RA_32>(Node), reg.V4S(), index);
break;
case 8:
umov(GetReg<RA_64>(Node), reg.V2D(), index);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
break;
}
};
if (Offset < SSERegBitSize) {
// Desired data lies within the lower 128-bit lane, so we
// can treat the operation as a 128-bit operation, even
// when acting on larger register sizes.
PerformMove(Vector, Op->Index);
} else {
LOGMAN_THROW_AA_FMT(HostSupportsSVE,
"Host doesn't support SVE. Cannot perform 256-bit operation.");
LOGMAN_THROW_AA_FMT(Is256Bit,
"Can't perform 256-bit extraction with op side: {}", OpSize);
LOGMAN_THROW_AA_FMT(Offset < AVXRegBitSize,
"Trying to extract element outside bounds of register. Offset={}, Index={}",
Offset, Op->Index);
// We need to use the upper 128-bit lane, so lets move it down.
// Inverting our dedicated predicate for 128-bit operations selects
// all of the top lanes. We can then compact those into a temporary.
const auto CompactPred = p0;
not_(CompactPred.VnB(), PRED_TMP_32B.Zeroing(), PRED_TMP_16B.VnB());
compact(VTMP1.Z().VnD(), CompactPred, Vector.Z().VnD());
// Sanitize the zero-based index to work on the now-moved
// upper half of the vector.
const auto SanitizedIndex = [OpSize, Op] {
switch (OpSize) {
case 1:
return Op->Index - 16;
case 2:
return Op->Index - 8;
case 4:
return Op->Index - 4;
case 8:
return Op->Index - 2;
default:
LOGMAN_MSG_A_FMT("Unhandled OpSize: {}", OpSize);
return 0;
}
}();
// Move the value from the now-low-lane data.
PerformMove(VTMP1, SanitizedIndex);
}
}
@@ -243,7 +243,6 @@ DEF_OP(InlineSyscall) {
bool Intersects{};
// We always need to spill x8 since we can't know if it is live at this SSA location
uint32_t SpillMask = 1U << 8;
std::vector<vixl::aarch64::Register> IntersectRegs(FEXCore::HLE::SyscallArguments::MAX_ARGS);
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
if (Op->Header.Args[i].IsInvalid()) break;
@@ -12,26 +12,115 @@ using namespace vixl;
using namespace vixl::aarch64;
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
DEF_OP(VInsGPR) {
auto Op = IROp->C<IR::IROp_VInsGPR>();
mov(GetDst(Node), GetSrc(Op->DestVector.ID()));
switch (Op->Header.ElementSize) {
case 1: {
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
break;
const auto Op = IROp->C<IR::IROp_VInsGPR>();
const auto OpSize = IROp->Size;
const auto DestIdx = Op->DestIdx;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto DestVector = GetSrc(Op->DestVector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto ElementSizeBits = ElementSize * 8;
const auto Offset = ElementSizeBits * DestIdx;
const auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
const auto InUpperLane = Offset >= SSEBitSize;
// This is going to be a little gross. Pls forgive me.
// Since SVE has the whole vector length agnostic programming
// thing going on, we can't exactly freely insert entries into
// arbitrary locations in the vector.
//
// SVE *does* have INSR, however this only shifts the entire
// vector to the left by an element size and inserts a value
// at the beginning of the vector. Not *quite* what we need.
// (though INSR *is* very useful for other things).
//
// The idea is (in the case of the upper lane), move the upper
// lane down, insert into it and recombine with the lower lane.
//
// In the case of the lower lane, insert and then recombine with
// the upper lane.
if (InUpperLane) {
// Move the upper lane down for the insertion.
const auto CompactPred = p0;
not_(CompactPred.VnB(), PRED_TMP_32B.Zeroing(), PRED_TMP_16B.VnB());
compact(VTMP1.Z().VnD(), CompactPred, DestVector.Z().VnD());
}
case 2: {
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
break;
// Put data in place for destructive SPLICE below.
mov(Dst.Z().VnD(), DestVector.Z().VnD());
// Inserts the GPR value into the given V register.
// Also automatically adjusts the index in the case of using the
// moved upper lane.
const auto Insert = [&](const aarch64::VRegister& reg, int index) {
switch (ElementSize) {
case 1:
if (InUpperLane) {
index -= 16;
}
ins(reg.V16B(), index, GetReg<RA_32>(Op->Src.ID()));
break;
case 2:
if (InUpperLane) {
index -= 8;
}
ins(reg.V8H(), index, GetReg<RA_32>(Op->Src.ID()));
break;
case 4:
if (InUpperLane) {
index -= 4;
}
ins(reg.V4S(), index, GetReg<RA_32>(Op->Src.ID()));
break;
case 8:
if (InUpperLane) {
index -= 2;
}
ins(reg.V2D(), index, GetReg<RA_64>(Op->Src.ID()));
break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
};
if (InUpperLane) {
Insert(VTMP1, DestIdx);
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), VTMP1.Z().VnD());
} else {
Insert(Dst, DestIdx);
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), DestVector.Z().VnD());
}
case 4: {
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
break;
} else {
mov(Dst, DestVector);
switch (ElementSize) {
case 1: {
ins(Dst.V16B(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
break;
}
case 2: {
ins(Dst.V8H(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
break;
}
case 4: {
ins(Dst.V4S(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
break;
}
case 8: {
ins(Dst.V2D(), DestIdx, GetReg<RA_64>(Op->Src.ID()));
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
case 8: {
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Src.ID()));
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
}
}
@@ -57,8 +146,11 @@ DEF_OP(VCastFromGPR) {
}
DEF_OP(Float_FromGPR_S) {
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
const uint16_t ElementSize = Op->Header.ElementSize;
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0404: { // Float <- int32_t
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
@@ -76,6 +168,10 @@ DEF_OP(Float_FromGPR_S) {
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
Conv, ElementSize, Op->SrcElementSize);
break;
}
}
@@ -96,116 +192,379 @@ DEF_OP(Float_FToF) {
}
DEF_OP(Vector_SToF) {
auto Op = IROp->C<IR::IROp_Vector_SToF>();
switch (Op->Header.ElementSize) {
case 4:
scvtf(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
break;
case 8:
scvtf(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
switch (ElementSize) {
case 2:
scvtf(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
scvtf(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
scvtf(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
break;
}
} else {
switch (ElementSize) {
case 2:
scvtf(Dst.V8H(), Vector.V8H());
break;
case 4:
scvtf(Dst.V4S(), Vector.V4S());
break;
case 8:
scvtf(Dst.V2D(), Vector.V2D());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
break;
}
}
}
DEF_OP(Vector_FToZS) {
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
switch (Op->Header.ElementSize) {
case 4:
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
break;
case 8:
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
switch (ElementSize) {
case 2:
fcvtzs(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
fcvtzs(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
fcvtzs(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
break;
}
} else {
switch (ElementSize) {
case 2:
fcvtzs(Dst.V8H(), Vector.V8H());
break;
case 4:
fcvtzs(Dst.V4S(), Vector.V4S());
break;
case 8:
fcvtzs(Dst.V2D(), Vector.V2D());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
break;
}
}
}
DEF_OP(Vector_FToS) {
auto Op = IROp->C<IR::IROp_Vector_FToS>();
switch (Op->Header.ElementSize) {
case 4:
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
break;
case 8:
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
switch (ElementSize) {
case 2:
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
fcvtzs(Dst.Z().VnH(), Mask, Dst.Z().VnH());
break;
case 4:
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
fcvtzs(Dst.Z().VnS(), Mask, Dst.Z().VnS());
break;
case 8:
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
fcvtzs(Dst.Z().VnD(), Mask, Dst.Z().VnD());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
break;
}
} else {
switch (ElementSize) {
case 2:
frinti(Dst.V8H(), Vector.V8H());
fcvtzs(Dst.V8H(), Dst.V8H());
break;
case 4:
frinti(Dst.V4S(), Vector.V4S());
fcvtzs(Dst.V4S(), Dst.V4S());
break;
case 8:
frinti(Dst.V2D(), Vector.V2D());
fcvtzs(Dst.V2D(), Dst.V2D());
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
break;
}
}
}
DEF_OP(Vector_FToF) {
auto Op = IROp->C<IR::IROp_Vector_FToF>();
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
const auto OpSize = IROp->Size;
switch (Conv) {
case 0x0804: { // Double <- Float
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2S());
break;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
// Curiously, FCVTLT and FCVTNT have no bottom variants,
// and also interesting is that FCVTLT will iterate the
// source vector by accessing each odd element and storing
// them consecutively in the destination.
//
// FCVTNT is somewhat like the opposite. It will read each
// consecutive element, but store each result into every odd
// element in the destination vector.
//
// We need to undo the behavior of FCVTNT with UZP2. In the case
// of FCVTLT, we instead need to set the vector up with ZIP1, so
// that the elements will be processed correctly.
const auto Mask = PRED_TMP_32B.Merging();
switch (Conv) {
case 0x0402: { // Float <- Half
zip1(Dst.Z().VnH(), Vector.Z().VnH(), Vector.Z().VnH());
fcvtlt(Dst.Z().VnS(), Mask, Dst.Z().VnH());
break;
}
case 0x0804: { // Double <- Float
zip1(Dst.Z().VnS(), Vector.Z().VnS(), Vector.Z().VnS());
fcvtlt(Dst.Z().VnD(), Mask, Dst.Z().VnS());
break;
}
case 0x0204: { // Half <- Float
fcvtnt(Dst.Z().VnH(), Mask, Vector.Z().VnS());
uzp2(Dst.Z().VnH(), Dst.Z().VnH(), Dst.Z().VnH());
break;
}
case 0x0408: { // Float <- Double
fcvtnt(Dst.Z().VnS(), Mask, Vector.Z().VnD());
uzp2(Dst.Z().VnS(), Dst.Z().VnS(), Dst.Z().VnS());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
break;
}
case 0x0408: { // Float <- Double
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Vector.ID()).V2D());
break;
} else {
switch (Conv) {
case 0x0402: { // Float <- Half
fcvtl(Dst.V4S(), Vector.V4H());
break;
}
case 0x0804: { // Double <- Float
fcvtl(Dst.V2D(), Vector.V2S());
break;
}
case 0x0204: { // Half <- Float
fcvtn(Dst.V4H(), Vector.V4S());
break;
}
case 0x0408: { // Float <- Double
fcvtn(Dst.V2S(), Vector.V2D());
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
}
}
DEF_OP(Vector_FToI) {
auto Op = IROp->C<IR::IROp_Vector_FToI>();
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (Op->Header.ElementSize) {
case 4:
frintn(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
if (HostSupportsSVE && Is256Bit) {
const auto Mask = PRED_TMP_32B.Merging();
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (ElementSize) {
case 2:
frintn(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frintn(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frintn(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
case 8:
frintn(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (ElementSize) {
case 2:
frintm(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frintm(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frintm(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
}
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (Op->Header.ElementSize) {
case 4:
frintm(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (ElementSize) {
case 2:
frintp(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frintp(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frintp(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
case 8:
frintm(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
case FEXCore::IR::Round_Towards_Zero.Val:
switch (ElementSize) {
case 2:
frintz(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frintz(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frintz(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
}
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (Op->Header.ElementSize) {
case 4:
frintp(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
case FEXCore::IR::Round_Host.Val:
switch (ElementSize) {
case 2:
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
break;
case 4:
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
break;
case 8:
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
break;
}
break;
case 8:
frintp(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
}
} else {
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
switch (ElementSize) {
case 2:
frintn(Dst.V8H(), Vector.V8H());
break;
case 4:
frintn(Dst.V4S(), Vector.V4S());
break;
case 8:
frintn(Dst.V2D(), Vector.V2D());
break;
}
break;
}
break;
case FEXCore::IR::Round_Towards_Zero.Val:
switch (Op->Header.ElementSize) {
case 4:
frintz(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
case FEXCore::IR::Round_Negative_Infinity.Val:
switch (ElementSize) {
case 2:
frintm(Dst.V8H(), Vector.V8H());
break;
case 4:
frintm(Dst.V4S(), Vector.V4S());
break;
case 8:
frintm(Dst.V2D(), Vector.V2D());
break;
}
break;
case 8:
frintz(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
case FEXCore::IR::Round_Positive_Infinity.Val:
switch (ElementSize) {
case 2:
frintp(Dst.V8H(), Vector.V8H());
break;
case 4:
frintp(Dst.V4S(), Vector.V4S());
break;
case 8:
frintp(Dst.V2D(), Vector.V2D());
break;
}
break;
}
break;
case FEXCore::IR::Round_Host.Val:
switch (Op->Header.ElementSize) {
case 4:
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
case FEXCore::IR::Round_Towards_Zero.Val:
switch (ElementSize) {
case 2:
frintz(Dst.V8H(), Vector.V8H());
break;
case 4:
frintz(Dst.V4S(), Vector.V4S());
break;
case 8:
frintz(Dst.V2D(), Vector.V2D());
break;
}
break;
case 8:
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
case FEXCore::IR::Round_Host.Val:
switch (ElementSize) {
case 2:
frinti(Dst.V8H(), Vector.V8H());
break;
case 4:
frinti(Dst.V4S(), Vector.V4S());
break;
case 8:
frinti(Dst.V2D(), Vector.V2D());
break;
}
break;
}
break;
}
}
}
@@ -28,6 +28,7 @@ $end_info$
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/Profiler.h>
#include "Interface/Core/Interpreter/InterpreterOps.h"
@@ -731,6 +732,8 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData,
bool GDBEnabled) {
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
using namespace aarch64;
JumpTargets.clear();
uint32_t SSACount = IR->GetSSACount();
@@ -118,6 +118,17 @@ private:
IR::MemOffsetType OffsetType,
uint8_t OffsetScale);
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
// the limits of the scalar plus immediate variant of SVE load/stores.
//
// TMP1 is safe to use again once this memory operand is used with its
// equivalent loads or stores that this was called for.
[[nodiscard]] SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize,
aarch64::Register Base,
IR::OrderedNodeWrapper Offset,
IR::MemOffsetType OffsetType,
uint8_t OffsetScale);
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
+44 -11
View File
@@ -1143,26 +1143,59 @@ DEF_OP(Select) {
}
DEF_OP(VExtractToGPR) {
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
switch (Op->Header.ElementSize) {
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
constexpr auto SSEBitSize = SSERegSize * 8;
const auto ElementSize = Op->Header.ElementSize;
const auto ElementSizeBits = ElementSize * 8;
const auto Offset = ElementSizeBits * Op->Index;
const auto Is256Bit = Offset >= SSEBitSize;
const auto Vector = GetSrc(Op->Vector.ID());
switch (ElementSize) {
case 1: {
pextrb(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
break;
if (Is256Bit) {
vextracti128(xmm15, ToYMM(Vector), 1);
pextrb(GetDst<RA_32>(Node), xmm15, Op->Index - 16);
} else {
pextrb(GetDst<RA_32>(Node), Vector, Op->Index);
}
break;
}
case 2: {
pextrw(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
break;
if (Is256Bit) {
vextracti128(xmm15, ToYMM(Vector), 1);
pextrw(GetDst<RA_32>(Node), xmm15, Op->Index - 8);
} else {
pextrw(GetDst<RA_32>(Node), Vector, Op->Index);
}
break;
}
case 4: {
pextrd(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
break;
if (Is256Bit) {
vextracti128(xmm15, ToYMM(Vector), 1);
pextrd(GetDst<RA_32>(Node), xmm15, Op->Index - 4);
} else {
pextrd(GetDst<RA_32>(Node), Vector, Op->Index);
}
break;
}
case 8: {
pextrq(GetDst<RA_64>(Node), GetSrc(Op->Vector.ID()), Op->Index);
break;
if (Is256Bit) {
vextracti128(xmm15, ToYMM(Vector), 1);
pextrq(GetDst<RA_64>(Node), xmm15, Op->Index - 2);
} else {
pextrq(GetDst<RA_64>(Node), Vector, Op->Index);
}
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
}
@@ -17,27 +17,76 @@ namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
DEF_OP(VInsGPR) {
auto Op = IROp->C<IR::IROp_VInsGPR>();
movapd(GetDst(Node), GetSrc(Op->DestVector.ID()));
const auto Op = IROp->C<IR::IROp_VInsGPR>();
const auto OpSize = IROp->Size;
switch (Op->Header.ElementSize) {
case 1: {
pinsrb(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
break;
const auto Dst = GetDst(Node);
const auto DestVector = GetSrc(Op->DestVector.ID());
const auto DestIdx = Op->DestIdx;
const auto ElementSize = Op->Header.ElementSize;
const auto ElementSizeBits = ElementSize * 8;
const auto Offset = ElementSizeBits * DestIdx;
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto InUpperLane = Offset >= SSEBitSize;
if (InUpperLane && !Is256Bit) {
LOGMAN_MSG_A_FMT("Attempt to access upper 128-bit lane in 128-bit operation! Offset={}",
Offset);
return;
}
if (Is256Bit) {
vmovapd(ToYMM(Dst), ToYMM(DestVector));
} else {
vmovapd(Dst, DestVector);
}
const auto Insert = [&](const Xbyak::Xmm& reg, int index) {
switch (ElementSize) {
case 1: {
if (InUpperLane) {
index -= 16;
}
pinsrb(reg, GetSrc<RA_32>(Op->Src.ID()), index);
break;
}
case 2: {
if (InUpperLane) {
index -= 8;
}
pinsrw(reg, GetSrc<RA_32>(Op->Src.ID()), index);
break;
}
case 4: {
if (InUpperLane) {
index -= 4;
}
pinsrd(reg, GetSrc<RA_32>(Op->Src.ID()), index);
break;
}
case 8: {
if (InUpperLane) {
index -= 2;
}
pinsrq(reg, GetSrc<RA_64>(Op->Src.ID()), index);
break;
}
default:
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
break;
}
case 2: {
pinsrw(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
break;
}
case 4: {
pinsrd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
break;
}
case 8: {
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()), Op->DestIdx);
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
};
if (InUpperLane) {
vextracti128(xmm15, ToYMM(Dst), 1);
Insert(xmm15, DestIdx);
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
} else {
Insert(Dst, DestIdx);
}
}
@@ -63,8 +112,10 @@ DEF_OP(VCastFromGPR) {
}
DEF_OP(Float_FromGPR_S) {
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
const uint16_t ElementSize = Op->Header.ElementSize;
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
switch (Conv) {
case 0x0404: { // Float <- int32_t
@@ -83,6 +134,10 @@ DEF_OP(Float_FromGPR_S) {
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
Conv, ElementSize, Op->SrcElementSize);
break;
}
}
@@ -104,99 +159,194 @@ DEF_OP(Float_FToF) {
}
DEF_OP(Vector_SToF) {
auto Op = IROp->C<IR::IROp_Vector_SToF>();
switch (Op->Header.ElementSize) {
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
switch (ElementSize) {
case 4:
cvtdq2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
break;
if (Is256Bit) {
vcvtdq2ps(ToYMM(Dst), ToYMM(Vector));
} else {
vcvtdq2ps(Dst, Vector);
}
break;
case 8:
// This operation is a bit disgusting in x86
// There is no vector form of this instruction until AVX512VL + AVX512DQ (vcvtqq2pd)
// 1) First extract the top 64bits
// 2) Do a scalar conversion on each
// 3) Make sure to merge them together at the end
pextrq(rax, GetSrc(Op->Vector.ID()), 1);
pextrq(rcx, GetSrc(Op->Vector.ID()), 0);
cvtsi2sd(GetDst(Node), rcx);
pextrq(rax, Vector, 1);
pextrq(rcx, Vector, 0);
cvtsi2sd(Dst, rcx);
cvtsi2sd(xmm15, rax);
movlhps(GetDst(Node), xmm15);
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
vmovlhps(Dst, Dst, xmm15);
if (Is256Bit) {
vextracti128(xmm15, ToYMM(Vector), 1);
pextrq(rax, xmm15, 1);
pextrq(rcx, xmm15, 0);
cvtsi2sd(xmm15, rcx);
cvtsi2sd(xmm14, rax);
movlhps(xmm15, xmm14);
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
}
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
break;
}
}
DEF_OP(Vector_FToZS) {
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
switch (Op->Header.ElementSize) {
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
switch (ElementSize) {
case 4:
cvttps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
break;
if (Is256Bit) {
vcvttps2dq(ToYMM(Dst), ToYMM(Vector));
} else {
vcvttps2dq(Dst, Vector);
}
break;
case 8:
cvttpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
if (Is256Bit) {
vcvttpd2dq(ToYMM(Dst), ToYMM(Vector));
} else {
vcvttpd2dq(Dst, Vector);
}
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
break;
}
}
DEF_OP(Vector_FToS) {
auto Op = IROp->C<IR::IROp_Vector_FToS>();
switch (Op->Header.ElementSize) {
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
switch (ElementSize) {
case 4:
cvtps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
break;
if (Is256Bit) {
vcvtps2dq(ToYMM(Dst), ToYMM(Vector));
} else {
vcvtps2dq(Dst, Vector);
}
break;
case 8:
cvtpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
if (Is256Bit) {
vcvtpd2dq(ToYMM(Dst), ToYMM(Vector));
} else {
vcvtpd2dq(Dst, Vector);
}
break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
break;
}
}
DEF_OP(Vector_FToF) {
auto Op = IROp->C<IR::IROp_Vector_FToF>();
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
const auto OpSize = IROp->Size;
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
switch (Conv) {
case 0x0804: { // Double <- Float
cvtps2pd(GetDst(Node), GetSrc(Op->Vector.ID()));
if (Is256Bit) {
vcvtps2pd(ToYMM(Dst), Vector);
} else {
vcvtps2pd(Dst, Vector);
}
break;
}
case 0x0408: { // Float <- Double
cvtpd2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
if (Is256Bit) {
vcvtpd2ps(Dst, ToYMM(Vector));
} else {
vcvtpd2ps(Dst, Vector);
}
break;
}
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv); break;
default:
LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv);
break;
}
}
DEF_OP(Vector_FToI) {
auto Op = IROp->C<IR::IROp_Vector_FToI>();
uint8_t RoundMode{};
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
const auto OpSize = IROp->Size;
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
RoundMode = 0b0000'0'0'00;
break;
case FEXCore::IR::Round_Negative_Infinity.Val:
RoundMode = 0b0000'0'0'01;
break;
case FEXCore::IR::Round_Positive_Infinity.Val:
RoundMode = 0b0000'0'0'10;
break;
case FEXCore::IR::Round_Towards_Zero.Val:
RoundMode = 0b0000'0'0'11;
break;
case FEXCore::IR::Round_Host.Val:
RoundMode = 0b0000'0'1'00;
break;
}
const uint8_t RoundMode = [Op] {
switch (Op->Round) {
case FEXCore::IR::Round_Nearest.Val:
return 0b0000'0'0'00;
case FEXCore::IR::Round_Negative_Infinity.Val:
return 0b0000'0'0'01;
case FEXCore::IR::Round_Positive_Infinity.Val:
return 0b0000'0'0'10;
case FEXCore::IR::Round_Towards_Zero.Val:
return 0b0000'0'0'11;
case FEXCore::IR::Round_Host.Val:
return 0b0000'0'1'00;
default:
LOGMAN_MSG_A_FMT("Unhandled rounding mode");
return 0;
}
}();
switch (Op->Header.ElementSize) {
const auto ElementSize = Op->Header.ElementSize;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Dst = GetDst(Node);
const auto Vector = GetSrc(Op->Vector.ID());
switch (ElementSize) {
case 4:
roundps(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
break;
if (Is256Bit) {
vroundps(ToYMM(Dst), ToYMM(Vector), RoundMode);
} else {
vroundps(Dst, Vector, RoundMode);
}
break;
case 8:
roundpd(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
break;
if (Is256Bit) {
vroundpd(ToYMM(Dst), ToYMM(Vector), RoundMode);
} else {
vroundpd(Dst, Vector, RoundMode);
}
break;
default:
LOGMAN_MSG_A_FMT("Unhandled element size: {}", ElementSize);
break;
}
}
+4 -1
View File
@@ -27,6 +27,7 @@ $end_info$
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/EnumUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <algorithm>
#include <array>
@@ -370,7 +371,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
{
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
@@ -582,6 +583,8 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
}
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
FEXCORE_PROFILE_SCOPED("x86::CompileCode");
JumpTargets.clear();
uint32_t SSACount = IR->GetSSACount();
+232 -162
View File
@@ -21,130 +21,160 @@ namespace FEXCore::CPU {
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
DEF_OP(LoadContext) {
auto Op = IROp->C<IR::IROp_LoadContext>();
uint8_t OpSize = IROp->Size;
const auto Op = IROp->C<IR::IROp_LoadContext>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::GPRClass) {
switch (OpSize) {
case 1: {
movzx(GetDst<RA_32>(Node), byte [STATE + Op->Offset]);
break;
}
break;
case 2: {
movzx(GetDst<RA_32>(Node), word [STATE + Op->Offset]);
break;
}
break;
case 4: {
mov(GetDst<RA_32>(Node), dword [STATE + Op->Offset]);
break;
}
break;
case 8: {
mov(GetDst<RA_64>(Node), qword [STATE + Op->Offset]);
break;
}
break;
case 16: {
LOGMAN_MSG_A_FMT("Invalid GPR load of size 16");
break;
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
break;
}
}
else {
const auto Dst = GetDst(Node);
switch (OpSize) {
case 1: {
movzx(rax, byte [STATE + Op->Offset]);
vmovq(GetDst(Node), rax);
vmovq(Dst, rax);
break;
}
break;
case 2: {
movzx(rax, word [STATE + Op->Offset]);
vmovq(GetDst(Node), rax);
vmovq(Dst, rax);
break;
}
break;
case 4: {
vmovd(GetDst(Node), dword [STATE + Op->Offset]);
vmovd(Dst, dword [STATE + Op->Offset]);
break;
}
break;
case 8: {
vmovq(GetDst(Node), qword [STATE + Op->Offset]);
vmovq(Dst, qword [STATE + Op->Offset]);
break;
}
break;
case 16: {
if (Op->Offset % 16 == 0)
movaps(GetDst(Node), xword [STATE + Op->Offset]);
else
movups(GetDst(Node), xword [STATE + Op->Offset]);
if (Op->Offset % 16 == 0) {
vmovaps(Dst, xword [STATE + Op->Offset]);
} else {
vmovups(Dst, xword [STATE + Op->Offset]);
}
break;
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
case 32: {
if (Op->Offset % 32 == 0) {
vmovaps(ToYMM(Dst), yword [STATE + Op->Offset]);
} else {
vmovups(ToYMM(Dst), yword [STATE + Op->Offset]);
}
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
break;
}
}
}
DEF_OP(StoreContext) {
auto Op = IROp->C<IR::IROp_StoreContext>();
uint8_t OpSize = IROp->Size;
const auto Op = IROp->C<IR::IROp_StoreContext>();
const auto OpSize = IROp->Size;
if (Op->Class == IR::GPRClass) {
switch (OpSize) {
case 1: {
mov(byte [STATE + Op->Offset], GetSrc<RA_8>(Op->Value.ID()));
break;
}
break;
case 2: {
mov(word [STATE + Op->Offset], GetSrc<RA_16>(Op->Value.ID()));
break;
}
break;
case 4: {
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
break;
}
break;
case 8: {
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
break;
}
break;
case 16:
LogMan::Msg::DFmt("Invalid store size of 16");
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
case 16: {
LOGMAN_MSG_A_FMT("Invalid store size of 16");
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
break;
}
}
else {
const auto Value = GetSrc(Op->Value.ID());
switch (OpSize) {
case 1: {
pextrb(byte [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
pextrb(byte [STATE + Op->Offset], Value, 0);
break;
}
break;
case 2: {
pextrw(word [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
pextrw(word [STATE + Op->Offset], Value, 0);
break;
}
break;
case 4: {
vmovd(dword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
vmovd(dword [STATE + Op->Offset], Value);
break;
}
break;
case 8: {
vmovq(qword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
vmovq(qword [STATE + Op->Offset], Value);
break;
}
break;
case 16: {
if (Op->Offset % 16 == 0)
movaps(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
else
movups(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
if (Op->Offset % 16 == 0) {
vmovaps(xword [STATE + Op->Offset], Value);
} else {
vmovups(xword [STATE + Op->Offset], Value);
}
break;
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
case 32: {
if (Op->Offset % 32 == 0) {
vmovaps(yword [STATE + Op->Offset], ToYMM(Value));
} else {
vmovups(yword [STATE + Op->Offset], ToYMM(Value));
}
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
break;
}
}
}
DEF_OP(LoadContextIndexed) {
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
size_t size = IROp->Size;
Reg index = GetSrc<RA_64>(Op->Index.ID());
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
const auto OpSize = IROp->Size;
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
if (Op->Class == IR::GPRClass) {
switch (Op->Stride) {
@@ -153,21 +183,21 @@ DEF_OP(LoadContextIndexed) {
case 4:
case 8: {
lea(rax, dword [STATE + Op->BaseOffset]);
switch (size) {
switch (OpSize) {
case 1:
movzx(GetDst<RA_32>(Node), byte [rax + index * Op->Stride]);
movzx(GetDst<RA_32>(Node), byte [rax + Index * Op->Stride]);
break;
case 2:
movzx(GetDst<RA_32>(Node), word [rax + index * Op->Stride]);
movzx(GetDst<RA_32>(Node), word [rax + Index * Op->Stride]);
break;
case 4:
mov(GetDst<RA_32>(Node), dword [rax + index * Op->Stride]);
mov(GetDst<RA_32>(Node), dword [rax + Index * Op->Stride]);
break;
case 8:
mov(GetDst<RA_64>(Node), qword [rax + index * Op->Stride]);
mov(GetDst<RA_64>(Node), qword [rax + Index * Op->Stride]);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
break;
}
break;
@@ -186,53 +216,67 @@ DEF_OP(LoadContextIndexed) {
case 2:
case 4:
case 8: {
const auto Dst = GetDst(Node);
lea(rax, dword [STATE + Op->BaseOffset]);
switch (size) {
switch (OpSize) {
case 1:
movzx(eax, byte [rax + index * Op->Stride]);
vmovd(GetDst(Node), eax);
movzx(eax, byte [rax + Index * Op->Stride]);
vmovd(Dst, eax);
break;
case 2:
movzx(eax, word [rax + index * Op->Stride]);
vmovd(GetDst(Node), eax);
movzx(eax, word [rax + Index * Op->Stride]);
vmovd(Dst, eax);
break;
case 4:
vmovd(GetDst(Node), dword [rax + index * Op->Stride]);
vmovd(Dst, dword [rax + Index * Op->Stride]);
break;
case 8:
vmovq(GetDst(Node), qword [rax + index * Op->Stride]);
vmovq(Dst, qword [rax + Index * Op->Stride]);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
break;
}
break;
}
case 16: {
mov(rax, index);
shl(rax, 4);
case 16:
case 32: {
const auto Dst = GetDst(Node);
const auto Shift = Op->Stride == 16 ? 4 : 5;
mov(rax, Index);
shl(rax, Shift);
lea(rax, dword [rax + Op->BaseOffset]);
switch (size) {
switch (OpSize) {
case 1:
pinsrb(GetDst(Node), byte [STATE + rax], 0);
pinsrb(Dst, byte [STATE + rax], 0);
break;
case 2:
pinsrw(GetDst(Node), word [STATE + rax], 0);
pinsrw(Dst, word [STATE + rax], 0);
break;
case 4:
vmovd(GetDst(Node), dword [STATE + rax]);
vmovd(Dst, dword [STATE + rax]);
break;
case 8:
vmovq(GetDst(Node), qword [STATE + rax]);
vmovq(Dst, qword [STATE + rax]);
break;
case 16:
if (Op->BaseOffset % 16 == 0)
movaps(GetDst(Node), xword [STATE + rax]);
else
movups(GetDst(Node), xword [STATE + rax]);
if (Op->BaseOffset % 16 == 0) {
vmovaps(Dst, xword [STATE + rax]);
} else {
vmovups(Dst, xword [STATE + rax]);
}
break;
case 32:
if (Op->BaseOffset % 32 == 0) {
vmovaps(ToYMM(Dst), yword [STATE + rax]);
} else {
vmovups(ToYMM(Dst), yword [STATE + rax]);
}
break;
default:
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
break;
}
break;
@@ -245,12 +289,13 @@ DEF_OP(LoadContextIndexed) {
}
DEF_OP(StoreContextIndexed) {
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
Reg index = GetSrc<RA_64>(Op->Index.ID());
size_t size = IROp->Size;
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
const auto OpSize = IROp->Size;
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
if (Op->Class == IR::GPRClass) {
auto value = GetSrc<RA_64>(Op->Value.ID());
const auto Value = GetSrc<RA_64>(Op->Value.ID());
lea(rax, dword [STATE + Op->BaseOffset]);
switch (Op->Stride) {
@@ -258,10 +303,10 @@ DEF_OP(StoreContextIndexed) {
case 2:
case 4:
case 8: {
if (!(size == 1 || size == 2 || size == 4 || size == 8)) {
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
if (!(OpSize == 1 || OpSize == 2 || OpSize == 4 || OpSize == 8)) {
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
}
mov(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
mov(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
break;
}
default:
@@ -270,57 +315,68 @@ DEF_OP(StoreContextIndexed) {
}
}
else {
auto value = GetSrc(Op->Value.ID());
const auto Value = GetSrc(Op->Value.ID());
switch (Op->Stride) {
case 1:
case 2:
case 4:
case 8: {
lea(rax, dword [STATE + Op->BaseOffset]);
switch (size) {
switch (OpSize) {
case 1:
pextrb(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
pextrb(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
break;
case 2:
pextrw(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
pextrw(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
break;
case 4:
vmovd(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
vmovd(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
break;
case 8:
vmovq(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
vmovq(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
break;
}
break;
}
case 16: {
mov(rax, index);
shl(rax, 4);
case 16:
case 32: {
const auto Shift = Op->Stride == 16 ? 4 : 5;
mov(rax, Index);
shl(rax, Shift);
lea(rax, dword [rax + Op->BaseOffset]);
switch (size) {
switch (OpSize) {
case 1:
pextrb(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
pextrb(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
break;
case 2:
pextrw(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
pextrw(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
break;
case 4:
vmovd(AddressFrame(IROp->Size * 8) [STATE + rax], value);
vmovd(AddressFrame(OpSize * 8) [STATE + rax], Value);
break;
case 8:
vmovq(AddressFrame(IROp->Size * 8) [STATE + rax], value);
vmovq(AddressFrame(OpSize * 8) [STATE + rax], Value);
break;
case 16:
if (Op->BaseOffset % 16 == 0)
movaps(xword [STATE + rax], value);
else
movups(xword [STATE + rax], value);
if (Op->BaseOffset % 16 == 0) {
vmovaps(xword [STATE + rax], Value);
} else {
vmovups(xword [STATE + rax], Value);
}
break;
case 32:
if (Op->BaseOffset % 32 == 0) {
vmovaps(yword [STATE + rax], ToYMM(Value));
} else {
vmovups(yword [STATE + rax], ToYMM(Value));
}
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
break;
}
break;
@@ -420,15 +476,15 @@ DEF_OP(FillRegister) {
switch (OpSize) {
case 4: {
movss(Dst, dword [rsp + SlotOffset]);
vmovss(Dst, dword [rsp + SlotOffset]);
break;
}
case 8: {
movsd(Dst, qword [rsp + SlotOffset]);
vmovsd(Dst, qword [rsp + SlotOffset]);
break;
}
case 16: {
movaps(Dst, xword [rsp + SlotOffset]);
vmovaps(Dst, xword [rsp + SlotOffset]);
break;
}
case 32: {
@@ -482,118 +538,132 @@ Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper
}
DEF_OP(LoadMem) {
auto Op = IROp->C<IR::IROp_LoadMem>();
const auto Op = IROp->C<IR::IROp_LoadMem>();
const auto OpSize = IROp->Size;
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class == IR::GPRClass) {
auto Dst = GetDst<RA_64>(Node);
const auto Dst = GetDst<RA_64>(Node);
switch (IROp->Size) {
switch (OpSize) {
case 1: {
movzx (Dst, byte [MemPtr]);
movzx(Dst, byte [MemPtr]);
break;
}
break;
case 2: {
movzx (Dst, word [MemPtr]);
movzx(Dst, word [MemPtr]);
break;
}
break;
case 4: {
mov(Dst.cvt32(), dword [MemPtr]);
break;
}
break;
case 8: {
mov(Dst, qword [MemPtr]);
break;
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
default:
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
break;
}
}
else
{
auto Dst = GetDst(Node);
const auto Dst = GetDst(Node);
switch (IROp->Size) {
switch (OpSize) {
case 1: {
movzx(eax, byte [MemPtr]);
vmovd(Dst, eax);
break;
}
break;
case 2: {
movzx(eax, word [MemPtr]);
vmovd(Dst, eax);
break;
}
break;
case 4: {
vmovd(Dst, dword [MemPtr]);
break;
}
break;
case 8: {
vmovq(Dst, qword [MemPtr]);
break;
}
break;
case 16: {
if (IROp->Size == Op->Align)
movups(GetDst(Node), xword [MemPtr]);
else
movups(GetDst(Node), xword [MemPtr]);
if (MemoryDebug) {
movq(rcx, GetDst(Node));
}
}
break;
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
vmovups(Dst, xword [MemPtr]);
if (MemoryDebug) {
movq(rcx, Dst);
}
break;
}
case 32: {
vmovups(ToYMM(Dst), yword [MemPtr]);
if (MemoryDebug) {
movq(rcx, Dst);
}
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
break;
}
}
}
DEF_OP(StoreMem) {
auto Op = IROp->C<IR::IROp_StoreMem>();
const auto Op = IROp->C<IR::IROp_StoreMem>();
const auto OpSize = IROp->Size;
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
if (Op->Class == IR::GPRClass) {
switch (IROp->Size) {
switch (OpSize) {
case 1:
mov(byte [MemPtr], GetSrc<RA_8>(Op->Value.ID()));
break;
break;
case 2:
mov(word [MemPtr], GetSrc<RA_16>(Op->Value.ID()));
break;
break;
case 4:
mov(dword [MemPtr], GetSrc<RA_32>(Op->Value.ID()));
break;
break;
case 8:
mov(qword [MemPtr], GetSrc<RA_64>(Op->Value.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
break;
}
}
else {
switch (IROp->Size) {
const auto Value = GetSrc(Op->Value.ID());
switch (OpSize) {
case 1:
pextrb(byte [MemPtr], GetSrc(Op->Value.ID()), 0);
break;
pextrb(byte [MemPtr], Value, 0);
break;
case 2:
pextrw(word [MemPtr], GetSrc(Op->Value.ID()), 0);
break;
pextrw(word [MemPtr], Value, 0);
break;
case 4:
vmovd(dword [MemPtr], GetSrc(Op->Value.ID()));
break;
vmovd(dword [MemPtr], Value);
break;
case 8:
vmovq(qword [MemPtr], GetSrc(Op->Value.ID()));
break;
vmovq(qword [MemPtr], Value);
break;
case 16:
if (IROp->Size == Op->Align)
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
else
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
break;
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
vmovups(xword [MemPtr], Value);
break;
case 32:
vmovups(yword [MemPtr], ToYMM(Value));
break;
default:
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
break;
}
}
}
File diff suppressed because it is too large. Load diff
+345 -56
View File
@@ -240,8 +240,11 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) {
// RIP (64/32/16 bits)
auto NewRIP = _LoadMem(GPRClass, GPRSize, SP, GPRSize);
SP = _Add(SP, Constant);
//CS (lower 16 used)
_StoreContext(2, GPRClass, _LoadMem(GPRClass, GPRSize, SP, GPRSize), offsetof(FEXCore::Core::CPUState, cs));
// CS (lower 16 used)
auto NewSegmentCS = _LoadMem(GPRClass, GPRSize, SP, GPRSize);
_StoreContext(2, GPRClass, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx));
UpdatePrefixFromSegment(NewSegmentCS, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX);
SP = _Add(SP, Constant);
//eflags (lower 16 used)
auto eflags = _LoadMem(GPRClass, GPRSize, SP, GPRSize);
@@ -253,8 +256,11 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) {
// FEX doesn't support a CPL mode switch, so don't need to worry about this on 32-bit
_StoreContext(GPRSize, GPRClass, _LoadMem(GPRClass, GPRSize, SP, GPRSize), RSPOffset);
SP = _Add(SP, Constant);
//ss
_StoreContext(2, GPRClass, _LoadMem(GPRClass, GPRSize, SP, GPRSize), offsetof(FEXCore::Core::CPUState, ss));
// ss
auto NewSegmentSS = _LoadMem(GPRClass, GPRSize, SP, GPRSize);
_StoreContext(2, GPRClass, NewSegmentSS, offsetof(FEXCore::Core::CPUState, ss_idx));
UpdatePrefixFromSegment(NewSegmentSS, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX);
SP = _Add(SP, Constant);
}
else {
@@ -572,26 +578,51 @@ void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs) {
_StoreContext(GPRSize, GPRClass, NewSP, RSPOffset);
OrderedNode *Src{};
switch (SegmentReg) {
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, es));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs));
break;
default: break; // Do nothing
if (!CTX->Config.Is64BitMode()) {
switch (SegmentReg) {
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_idx));
break;
default: break; // Do nothing
}
}
else {
switch (SegmentReg) {
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
break;
default: break; // Do nothing
}
}
// Store our value to the new stack location
@@ -689,25 +720,27 @@ void OpDispatchBuilder::POPSegmentOp(OpcodeArgs) {
switch (SegmentReg) {
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, es));
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, es_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs));
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss));
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ds));
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ds_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, fs));
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, fs_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, gs));
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, gs_idx));
break;
default: break; // Do nothing
}
UpdatePrefixFromSegment(NewSegment, SegmentReg);
}
void OpDispatchBuilder::LEAVEOp(OpcodeArgs) {
@@ -1591,11 +1624,13 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
switch (Op->Dest.Data.GPR.GPR) {
case 0: // ES
case FEXCore::X86State::REG_R8: // ES
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, es));
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, es_idx));
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX);
break;
case 1: // DS
case FEXCore::X86State::REG_R11: // DS
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ds));
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ds_idx));
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX);
break;
case 2: // CS
case FEXCore::X86State::REG_R9: // CS
@@ -1609,12 +1644,14 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
break;
case 3: // SS
case FEXCore::X86State::REG_R10: // SS
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ss));
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ss_idx));
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX);
break;
case 6: // GS
case FEXCore::X86State::REG_R13: // GS
if (!CTX->Config.Is64BitMode) {
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs));
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs_idx));
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX);
} else {
LogMan::Msg::EFmt("We don't support modifying GS selector in 64bit mode!");
DecodeFailure = true;
@@ -1623,7 +1660,8 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
case 7: // FS
case FEXCore::X86State::REG_R12: // FS
if (!CTX->Config.Is64BitMode) {
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs));
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs_idx));
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX);
} else {
LogMan::Msg::EFmt("We don't support modifying FS selector in 64bit mode!");
DecodeFailure = true;
@@ -1641,19 +1679,19 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
switch (Op->Src[0].Data.GPR.GPR) {
case 0: // ES
case FEXCore::X86State::REG_R8: // ES
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, es));
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, es_idx));
break;
case 1: // DS
case FEXCore::X86State::REG_R11: // DS
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ds));
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ds_idx));
break;
case 2: // CS
case FEXCore::X86State::REG_R9: // CS
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, cs));
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, cs_idx));
break;
case 3: // SS
case FEXCore::X86State::REG_R10: // SS
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ss));
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ss_idx));
break;
case 6: // GS
case FEXCore::X86State::REG_R13: // GS
@@ -1661,7 +1699,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
Segment = _Constant(0);
}
else {
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, gs));
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, gs_idx));
}
break;
case 7: // FS
@@ -1670,7 +1708,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
Segment = _Constant(0);
}
else {
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, fs));
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, fs_idx));
}
break;
default:
@@ -3342,6 +3380,222 @@ void OpDispatchBuilder::PopcountOp(OpcodeArgs) {
GenerateFlags_POPCOUNT(Op, Src);
}
void OpDispatchBuilder::DAAOp(OpcodeArgs) {
CalculateDeferredFlags();
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xF)), _Constant(9), _Constant(1), _Constant(0)));
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
_CondJump(Cond, TrueBlock, FalseBlock);
SetCurrentCodeBlock(FalseBlock);
{
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
_Jump(EndBlock);
}
SetCurrentCodeBlock(TrueBlock);
{
auto NewAL = _Add(AL, _Constant(0x6));
_StoreContext(1, GPRClass, NewAL, GPROffset(X86State::REG_RAX));
CalculateDeferredFlags();
auto NewCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Or(CF, NewCF));
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
_Jump(EndBlock);
}
SetCurrentCodeBlock(EndBlock);
Cond = _Or(CF, _Select(FEXCore::IR::COND_UGT, AL, _Constant(0x99), _Constant(1), _Constant(0)));
FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
EndBlock = CreateNewCodeBlockAfter(TrueBlock);
_CondJump(Cond, TrueBlock, FalseBlock);
SetCurrentCodeBlock(FalseBlock);
{
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
_Jump(EndBlock);
}
SetCurrentCodeBlock(TrueBlock);
{
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
auto NewAL = _Add(AL, _Constant(0x60));
_StoreContext(1, GPRClass, NewAL, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
_Jump(EndBlock);
}
SetCurrentCodeBlock(EndBlock);
// Update Flags
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
auto EightBitMask = _Constant(0xFF);
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
auto XorOp = _Xor(PopCountOp, _Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
}
void OpDispatchBuilder::DASOp(OpcodeArgs) {
CalculateDeferredFlags();
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xf)), _Constant(9), _Constant(1), _Constant(0)));
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
_CondJump(Cond, TrueBlock, FalseBlock);
SetCurrentCodeBlock(FalseBlock);
{
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
_Jump(EndBlock);
}
SetCurrentCodeBlock(TrueBlock);
{
auto NewAL = _Sub(AL, _Constant(0x6));
_StoreContext(1, GPRClass, NewAL, GPROffset(X86State::REG_RAX));
CalculateDeferredFlags();
auto NewCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Or(CF, NewCF));
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
_Jump(EndBlock);
}
SetCurrentCodeBlock(EndBlock);
Cond = _Or(CF, _Select(FEXCore::IR::COND_UGT, AL, _Constant(0x99), _Constant(1), _Constant(0)));
FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
EndBlock = CreateNewCodeBlockAfter(TrueBlock);
_CondJump(Cond, TrueBlock, FalseBlock);
SetCurrentCodeBlock(FalseBlock);
{
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
_Jump(EndBlock);
}
SetCurrentCodeBlock(TrueBlock);
{
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
auto NewAL = _Sub(AL, _Constant(0x60));
_StoreContext(1, GPRClass, NewAL, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
_Jump(EndBlock);
}
SetCurrentCodeBlock(EndBlock);
// Update Flags
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
auto EightBitMask = _Constant(0xFF);
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
auto XorOp = _Xor(PopCountOp, _Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
}
void OpDispatchBuilder::AAAOp(OpcodeArgs) {
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
auto AX = _LoadContext(2, GPRClass, GPROffset(X86State::REG_RAX));
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xF)), _Constant(9), _Constant(1), _Constant(0)));
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
_CondJump(Cond, TrueBlock, FalseBlock);
SetCurrentCodeBlock(FalseBlock);
{
auto NewAX = _And(AX, _Constant(0xFF0F));
_StoreContext(2, GPRClass, NewAX, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
_Jump(EndBlock);
}
SetCurrentCodeBlock(TrueBlock);
{
auto NewAX = _Add(AX, _Constant(0x106));
auto Result = _And(NewAX, _Constant(0xFF0F));
_StoreContext(2, GPRClass, Result, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
_Jump(EndBlock);
}
SetCurrentCodeBlock(EndBlock);
}
void OpDispatchBuilder::AASOp(OpcodeArgs) {
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
auto AX = _LoadContext(2, GPRClass, GPROffset(X86State::REG_RAX));
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xF)), _Constant(9), _Constant(1), _Constant(0)));
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
_CondJump(Cond, TrueBlock, FalseBlock);
SetCurrentCodeBlock(FalseBlock);
{
auto NewAX = _And(AX, _Constant(0xFF0F));
_StoreContext(2, GPRClass, NewAX, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
_Jump(EndBlock);
}
SetCurrentCodeBlock(TrueBlock);
{
auto NewAX = _Sub(AX, _Constant(6));
NewAX = _Sub(NewAX, _Constant(0x100));
auto Result = _And(NewAX, _Constant(0xFF0F));
_StoreContext(2, GPRClass, Result, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
_Jump(EndBlock);
}
SetCurrentCodeBlock(EndBlock);
}
void OpDispatchBuilder::AAMOp(OpcodeArgs) {
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
auto Imm8 = _Constant(Op->Src[0].Data.Literal.Value & 0xFF);
auto UDivOp = _UDiv(AL, Imm8);
auto URemOp = _URem(AL, Imm8);
auto AH = _Lshl(UDivOp, _Constant(8));
auto AX = _Add(AH, URemOp);
_StoreContext(2, GPRClass, AX, GPROffset(X86State::REG_RAX));
// Update Flags
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
auto EightBitMask = _Constant(0xFF);
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
auto XorOp = _Xor(PopCountOp, _Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
}
void OpDispatchBuilder::AADOp(OpcodeArgs) {
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
auto AH = _Lshr(_LoadContext(2, GPRClass, GPROffset(X86State::REG_RAX)), _Constant(8));
auto Imm8 = _Constant(Op->Src[0].Data.Literal.Value & 0xFF);
auto NewAL = _Add(AL, _Mul(AH, Imm8));
auto Result = _And(NewAL, _Constant(0xFF));
_StoreContext(2, GPRClass, Result, GPROffset(X86State::REG_RAX));
// Update Flags
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
auto EightBitMask = _Constant(0xFF);
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
auto XorOp = _Xor(PopCountOp, _Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
}
void OpDispatchBuilder::XLATOp(OpcodeArgs) {
const uint32_t RAXOffset = GPROffset(X86State::REG_RAX);
const uint32_t RBXOffset = GPROffset(X86State::REG_RBX);
@@ -3360,13 +3614,15 @@ void OpDispatchBuilder::XLATOp(OpcodeArgs) {
template<OpDispatchBuilder::Segment Seg>
void OpDispatchBuilder::ReadSegmentReg(OpcodeArgs) {
// 64-bit only
// Doesn't hit the segment register optimization
auto Size = GetSrcSize(Op);
OrderedNode *Src{};
if constexpr (Seg == Segment::FS) {
Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, fs));
Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached));
}
else {
Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, gs));
Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
}
StoreResult(GPRClass, Op, Src, -1);
@@ -3379,10 +3635,10 @@ void OpDispatchBuilder::WriteSegmentReg(OpcodeArgs) {
auto Size = GetDstSize(Op);
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
if constexpr (Seg == Segment::FS) {
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs));
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs_cached));
}
else {
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs));
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs_cached));
}
}
@@ -4580,10 +4836,10 @@ OrderedNode *OpDispatchBuilder::AppendSegmentOffset(OrderedNode *Value, uint32_t
if (CTX->Config.Is64BitMode) {
if (Flags & FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX) {
Value = _Add(Value, _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs)));
Value = _Add(Value, _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached)));
}
else if (Flags & FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX) {
Value = _Add(Value, _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs)));
Value = _Add(Value, _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached)));
}
// If there was any other segment in 64bit then it is ignored
}
@@ -4595,38 +4851,65 @@ OrderedNode *OpDispatchBuilder::AppendSegmentOffset(OrderedNode *Value, uint32_t
// Or the argument only uses a specific prefix (with override set)
Prefix = DefaultPrefix;
}
// With the segment register optimization we store the GDT bases directly in the segment register to remove indexed loads
switch (Prefix) {
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, es));
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, cs));
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ss));
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ds));
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, fs));
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, gs));
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
break;
default: break; // Do nothing
}
if (Segment) {
Segment = _Lshr(Segment, _Constant(3));
auto data = _LoadContextIndexed(Segment, 4, offsetof(FEXCore::Core::CPUState, gdt[0]), 4, GPRClass);
Value = _Add(Value, data);
Value = _Add(Value, Segment);
}
}
return Value;
}
void OpDispatchBuilder::UpdatePrefixFromSegment(OrderedNode *Segment, uint32_t SegmentReg) {
// Use BFE to extract the selector index in bits [15,3] of the segment register.
// In some cases the upper 16-bits of the 32-bit GPR contain garbage to ignore.
Segment = _Bfe(4, 16 - 3, 3, Segment);
auto NewSegment = _LoadContextIndexed(Segment, 4, offsetof(FEXCore::Core::CPUState, gdt[0]), 4, GPRClass);
switch (SegmentReg) {
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, es_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ds_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, fs_cached));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, gs_cached));
break;
default: break; // Do nothing
}
}
OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad, MemoryAccessType AccessType) {
LOGMAN_THROW_A_FMT(Operand.IsGPR() ||
Operand.IsLiteral() ||
@@ -5505,12 +5788,18 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
{0x17, 1, &OpDispatchBuilder::POPSegmentOp<FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
{0x1E, 1, &OpDispatchBuilder::PUSHSegmentOp<FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
{0x1F, 1, &OpDispatchBuilder::POPSegmentOp<FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
{0x27, 1, &OpDispatchBuilder::DAAOp},
{0x2F, 1, &OpDispatchBuilder::DASOp},
{0x37, 1, &OpDispatchBuilder::AAAOp},
{0x3F, 1, &OpDispatchBuilder::AASOp},
{0x40, 8, &OpDispatchBuilder::INCOp},
{0x48, 8, &OpDispatchBuilder::DECOp},
{0x60, 1, &OpDispatchBuilder::PUSHAOp},
{0x61, 1, &OpDispatchBuilder::POPAOp},
{0xCE, 1, &OpDispatchBuilder::INTOp},
{0xD4, 1, &OpDispatchBuilder::AAMOp},
{0xD5, 1, &OpDispatchBuilder::AADOp},
};
constexpr std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> BaseOpTable_64[] = {
@@ -278,6 +278,12 @@ public:
void NOTOp(OpcodeArgs);
void XADDOp(OpcodeArgs);
void PopcountOp(OpcodeArgs);
void DAAOp(OpcodeArgs);
void DASOp(OpcodeArgs);
void AAAOp(OpcodeArgs);
void AASOp(OpcodeArgs);
void AAMOp(OpcodeArgs);
void AADOp(OpcodeArgs);
void XLATOp(OpcodeArgs);
template<bool Reseed>
void RDRANDOp(OpcodeArgs);
@@ -646,6 +652,7 @@ private:
OrderedNode *Current_HeaderNode{};
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
void UpdatePrefixFromSegment(OrderedNode *Segment, uint32_t SegmentReg);
enum class MemoryAccessType {
// Choose TSO or Non-TSO depending on access type
@@ -769,7 +769,11 @@ void OpDispatchBuilder::CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, Orde
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, SrcSize * 8 - Shift, Src1));
auto OpSize = SrcSize * 8;
if (OpSize < Shift) {
Shift &= (OpSize - 1);
}
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, OpSize - Shift, Src1));
}
// PF
@@ -934,6 +938,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
// OF is set to the XOR of the new CF bit and the most significant bit of the result
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
// If shift == 0, don't update flags
@@ -963,7 +968,9 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode
// OF
{
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
// OF is set to the XOR of the new CF bit and the most significant bit of the result
// OF is the LSB and MSB XOR'd together.
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
auto OF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldOF, NewOF);
@@ -977,8 +984,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
if (Shift == 0) return;
auto OpSize = SrcSize * 8;
auto NewCF = _Bfe(1, OpSize - Shift, Src1);
auto NewCF = _Bfe(1, OpSize - 1, Res);
// CF
{
@@ -989,8 +995,10 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
// OF
{
if (Shift == 1) {
// OF is set to the XOR of the new CF bit and the most significant bit of the result
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Res), NewCF));
// OF is the top two MSBs XOR'd together
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
}
}
}
@@ -1000,17 +1008,22 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, Ord
auto OpSize = SrcSize * 8;
auto NewCF = _Bfe(1, 0, Res);
// CF
{
// Extract the last bit shifted in to CF
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, Shift, Src1));
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(NewCF);
}
// OF
{
if (Shift == 1) {
// OF is the top two MSBs XOR'd together
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Src1), _Bfe(1, OpSize - 2, Src1)));
// OF is the LSB and MSB XOR'd together.
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
}
}
}
@@ -1563,6 +1563,9 @@ void OpDispatchBuilder::PACKSSOp<4>(OpcodeArgs);
template<size_t ElementSize, bool Signed>
void OpDispatchBuilder::PMULLOp(OpcodeArgs) {
static_assert(ElementSize == sizeof(uint32_t),
"Currently only handles 32-bit -> 64-bit");
auto Size = GetSrcSize(Op);
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
@@ -1579,17 +1582,8 @@ void OpDispatchBuilder::PMULLOp(OpcodeArgs) {
}
}
else {
OrderedNode* Srcs1[2]{};
OrderedNode* Srcs2[2]{};
Srcs1[0] = _VExtr(Size, ElementSize, Src1, Src1, 0);
Srcs1[1] = _VExtr(Size, ElementSize, Src1, Src1, 2);
Srcs2[0] = _VExtr(Size, ElementSize, Src2, Src2, 0);
Srcs2[1] = _VExtr(Size, ElementSize, Src2, Src2, 2);
Src1 = _VInsElement(Size, ElementSize, 1, 0, Srcs1[0], Srcs1[1]);
Src2 = _VInsElement(Size, ElementSize, 1, 0, Srcs2[0], Srcs2[1]);
Src1 = _VInsElement(Size, ElementSize, 1, 2, Src1, Src1);
Src2 = _VInsElement(Size, ElementSize, 1, 2, Src2, Src2);
if constexpr (Signed) {
Res = _VSMull(Size, ElementSize, Src1, Src2);
@@ -782,6 +782,12 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F80SIN ||
IROp == IR::OP_F80COS) {
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
}
// Write to ST[TOP]
_StoreContextIndexed(result, top, 16, MMBaseOffset(), 16, FPRClass);
}
@@ -809,7 +815,8 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F80FPREM) {
if constexpr (IROp == IR::OP_F80FPREM ||
IROp == IR::OP_F80FPREM1) {
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
}
@@ -854,6 +861,9 @@ void OpDispatchBuilder::X87SinCos(OpcodeArgs) {
auto sin = _F80SIN(a);
auto cos = _F80COS(a);
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
// Write to ST[TOP]
_StoreContextIndexed(sin, orig_top, 16, MMBaseOffset(), 16, FPRClass);
_StoreContextIndexed(cos, top, 16, MMBaseOffset(), 16, FPRClass);
@@ -900,6 +910,9 @@ void OpDispatchBuilder::X87TAN(OpcodeArgs) {
OrderedNode *data = _VCastFromGPR(16, 8, low);
data = _VInsGPR(16, 8, 1, data, high);
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
// Write to ST[TOP]
_StoreContextIndexed(result, orig_top, 16, MMBaseOffset(), 16, FPRClass);
_StoreContextIndexed(data, top, 16, MMBaseOffset(), 16, FPRClass);
@@ -778,6 +778,12 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F64SIN ||
IROp == IR::OP_F64COS) {
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
}
// Write to ST[TOP]
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
}
@@ -804,7 +810,8 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
// Overwrite the op
result.first->Header.Op = IROp;
if constexpr (IROp == IR::OP_F64FPREM) {
if constexpr (IROp == IR::OP_F80FPREM ||
IROp == IR::OP_F80FPREM1) {
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
}
@@ -831,6 +838,9 @@ void OpDispatchBuilder::X87SinCosF64(OpcodeArgs) {
auto sin = _F64SIN(a);
auto cos = _F64COS(a);
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
// Write to ST[TOP]
_StoreContextIndexed(sin, orig_top, 8, MMBaseOffset(), 16, FPRClass);
_StoreContextIndexed(cos, top, 8, MMBaseOffset(), 16, FPRClass);
@@ -871,6 +881,9 @@ void OpDispatchBuilder::X87TANF64(OpcodeArgs) {
auto one = _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000));
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
// Write to ST[TOP]
_StoreContextIndexed(result, orig_top, 8, MMBaseOffset(), 16, FPRClass);
_StoreContextIndexed(one, top, 8, MMBaseOffset(), 16, FPRClass);
@@ -266,10 +266,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
{0x17, 1, X86InstInfo{"POP SS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
{0x1E, 1, X86InstInfo{"PUSH DS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
{0x1F, 1, X86InstInfo{"POP DS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
{0x40, 8, X86InstInfo{"INC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
{0x48, 8, X86InstInfo{"DEC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
@@ -283,8 +283,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
{0xA1, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
{0xA3, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
{0xCE, 1, X86InstInfo{"INTO", TYPE_INST, FLAGS_NONE, 0, nullptr}},
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, FLAGS_NONE, 1, nullptr}},
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, FLAGS_NONE, 1, nullptr}},
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
};
+6 -6
View File
@@ -355,7 +355,7 @@
"DestSize": "ByteSize",
"EmitValidation": [
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass"
]
},
@@ -370,7 +370,7 @@
"EmitValidation": [
"WalkFindRegClass($Value) == $Class",
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass"
]
},
@@ -381,7 +381,7 @@
"DestSize": "ByteSize",
"EmitValidation": [
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass"
]
},
"StoreContextIndexed SSA:$Value, GPR:$Index, u8:#ByteSize, u32:$BaseOffset, u32:$Stride, RegisterClass:$Class": {
@@ -393,7 +393,7 @@
"EmitValidation": [
"WalkFindRegClass($Value) == $Class",
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass"
]
},
@@ -1065,7 +1065,7 @@
},
"FPR = VSXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
"Desc": ["Sign extends elements from the source element size to the next size up",
"Source elements come from the upper 64bits of the register"
"Source elements come from the upper half of the register"
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / (ElementSize << 1)"
@@ -1077,7 +1077,7 @@
},
"FPR = VUXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
"Desc": ["Zero extends elements from the source element size to the next size up",
"Source elements come from the upper 64bits of the register"
"Source elements come from the upper half of the register"
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / (ElementSize << 1)"
+3
View File
@@ -12,6 +12,7 @@ $end_info$
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include <FEXCore/Config/Config.h>
#include <FEXCore/Utils/Profiler.h>
namespace FEXCore::IR {
class IREmitter;
@@ -66,6 +67,8 @@ void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAV
}
bool PassManager::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::Run");
bool Changed = false;
for (auto const &Pass : Passes) {
Changed |= Pass->Run(IREmit);
@@ -20,6 +20,7 @@ $end_info$
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <bit>
#include <cstdint>
@@ -1028,6 +1029,8 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
}
bool ConstProp::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::ConstProp");
bool Changed = false;
auto CurrentIR = IREmit->ViewIR();
auto OriginalWriteCursor = IREmit->GetWriteCursor();
@@ -9,6 +9,7 @@ $end_info$
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
@@ -22,6 +23,7 @@ private:
};
bool DeadCodeElimination::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::DCE");
auto CurrentIR = IREmit->ViewIR();
int NumRemoved = 0;
@@ -13,6 +13,7 @@ $end_info$
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <array>
#include <memory>
@@ -76,24 +77,6 @@ namespace {
std::vector<ContextMemberInfo> ClassificationInfo;
};
constexpr static std::array<LastAccessType, 15> DefaultAccess = {
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
ACCESS_INVALID, // SSE padding in non-AVX case
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
ACCESS_NONE,
};
static void ClassifyContextStruct(ContextInfo *ContextClassificationInfo, bool SupportsAVX) {
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
@@ -102,7 +85,7 @@ namespace {
offsetof(FEXCore::Core::CPUState, rip),
sizeof(FEXCore::Core::CPUState::rip),
},
DefaultAccess[0],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
@@ -112,62 +95,134 @@ namespace {
offsetof(FEXCore::Core::CPUState, gregs[0]) + sizeof(FEXCore::Core::CPUState::gregs[0]) * i,
FEXCore::Core::CPUState::GPR_REG_SIZE,
},
DefaultAccess[1],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
}
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, es),
sizeof(FEXCore::Core::CPUState::es),
offsetof(FEXCore::Core::CPUState, es_idx),
sizeof(FEXCore::Core::CPUState::es_idx),
},
DefaultAccess[2],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, cs),
sizeof(FEXCore::Core::CPUState::cs),
offsetof(FEXCore::Core::CPUState, cs_idx),
sizeof(FEXCore::Core::CPUState::cs_idx),
},
DefaultAccess[3],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, ss),
sizeof(FEXCore::Core::CPUState::ss),
offsetof(FEXCore::Core::CPUState, ss_idx),
sizeof(FEXCore::Core::CPUState::ss_idx),
},
DefaultAccess[4],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, ds),
sizeof(FEXCore::Core::CPUState::ds),
offsetof(FEXCore::Core::CPUState, ds_idx),
sizeof(FEXCore::Core::CPUState::ds_idx),
},
DefaultAccess[5],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, gs),
sizeof(FEXCore::Core::CPUState::gs),
offsetof(FEXCore::Core::CPUState, gs_idx),
sizeof(FEXCore::Core::CPUState::gs_idx),
},
DefaultAccess[6],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, fs),
sizeof(FEXCore::Core::CPUState::fs),
offsetof(FEXCore::Core::CPUState, fs_idx),
sizeof(FEXCore::Core::CPUState::fs_idx),
},
DefaultAccess[7],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, _pad),
sizeof(FEXCore::Core::CPUState::_pad),
},
ACCESS_INVALID,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, es_cached),
sizeof(FEXCore::Core::CPUState::es_cached),
},
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, cs_cached),
sizeof(FEXCore::Core::CPUState::cs_cached),
},
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, ss_cached),
sizeof(FEXCore::Core::CPUState::ss_cached),
},
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, ds_cached),
sizeof(FEXCore::Core::CPUState::ds_cached),
},
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, gs_cached),
sizeof(FEXCore::Core::CPUState::gs_cached),
},
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, fs_cached),
sizeof(FEXCore::Core::CPUState::fs_cached),
},
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo{
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, _pad2),
sizeof(FEXCore::Core::CPUState::_pad2),
},
ACCESS_INVALID,
FEXCore::IR::InvalidClass,
});
@@ -178,7 +233,7 @@ namespace {
offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) + FEXCore::Core::CPUState::XMM_AVX_REG_SIZE * i,
FEXCore::Core::CPUState::XMM_AVX_REG_SIZE,
},
DefaultAccess[8],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
}
@@ -189,7 +244,7 @@ namespace {
offsetof(FEXCore::Core::CPUState, xmm.sse.data[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
},
DefaultAccess[8],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
}
@@ -199,7 +254,7 @@ namespace {
offsetof(FEXCore::Core::CPUState, xmm.sse.pad[0][0]),
static_cast<uint16_t>(FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * FEXCore::Core::CPUState::NUM_XMMS),
},
DefaultAccess[9],
ACCESS_INVALID,
FEXCore::IR::InvalidClass,
});
}
@@ -210,7 +265,7 @@ namespace {
offsetof(FEXCore::Core::CPUState, flags[0]) + sizeof(FEXCore::Core::CPUState::flags[0]) * i,
FEXCore::Core::CPUState::FLAG_SIZE,
},
DefaultAccess[10],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
}
@@ -221,7 +276,7 @@ namespace {
offsetof(FEXCore::Core::CPUState, mm[0][0]) + sizeof(FEXCore::Core::CPUState::mm[0]) * i,
FEXCore::Core::CPUState::MM_REG_SIZE
},
DefaultAccess[11],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
}
@@ -233,7 +288,7 @@ namespace {
offsetof(FEXCore::Core::CPUState, gdt[0]) + sizeof(FEXCore::Core::CPUState::gdt[0]) * i,
sizeof(FEXCore::Core::CPUState::gdt[0]),
},
DefaultAccess[12],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
}
@@ -244,7 +299,7 @@ namespace {
offsetof(FEXCore::Core::CPUState, FCW),
sizeof(FEXCore::Core::CPUState::FCW),
},
DefaultAccess[13],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
@@ -254,7 +309,7 @@ namespace {
offsetof(FEXCore::Core::CPUState, FTW),
sizeof(FEXCore::Core::CPUState::FTW),
},
DefaultAccess[14],
ACCESS_NONE,
FEXCore::IR::InvalidClass,
});
@@ -288,40 +343,55 @@ namespace {
ContextClassification->at(Offset).StoreNode = nullptr;
};
size_t Offset = 0;
SetAccess(Offset++, DefaultAccess[0]);
SetAccess(Offset++, ACCESS_NONE);
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
SetAccess(Offset++, DefaultAccess[1]);
SetAccess(Offset++, ACCESS_NONE);
}
SetAccess(Offset++, DefaultAccess[2]);
SetAccess(Offset++, DefaultAccess[3]);
SetAccess(Offset++, DefaultAccess[4]);
SetAccess(Offset++, DefaultAccess[5]);
SetAccess(Offset++, DefaultAccess[6]);
SetAccess(Offset++, DefaultAccess[7]);
// Segment indexes
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
// Pad
SetAccess(Offset++, ACCESS_INVALID);
// Segments
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
// Pad2
SetAccess(Offset++, ACCESS_INVALID);
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
SetAccess(Offset++, DefaultAccess[8]);
SetAccess(Offset++, ACCESS_NONE);
}
if (!SupportsAVX) {
SetAccess(Offset++, DefaultAccess[9]);
SetAccess(Offset++, ACCESS_NONE);
}
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
SetAccess(Offset++, DefaultAccess[10]);
SetAccess(Offset++, ACCESS_NONE);
}
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
SetAccess(Offset++, DefaultAccess[11]);
SetAccess(Offset++, ACCESS_NONE);
}
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
SetAccess(Offset++, DefaultAccess[12]);
SetAccess(Offset++, ACCESS_NONE);
}
SetAccess(Offset++, DefaultAccess[13]);
SetAccess(Offset++, DefaultAccess[14]);
SetAccess(Offset++, ACCESS_NONE);
SetAccess(Offset++, ACCESS_NONE);
}
struct BlockInfo {
@@ -695,6 +765,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
}
bool RCLSE::Run(FEXCore::IR::IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::RCLSE");
// XXX: We don't do cross-block optimizations yet
//CalculateControlFlowInfo(IREmit);
bool Changed = false;
@@ -12,6 +12,7 @@ $end_info$
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
#include <stddef.h>
@@ -154,6 +155,8 @@ struct Info {
*
*/
bool DeadStoreElimination::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::DSE");
std::unordered_map<OrderedNode*, Info> InfoMap;
bool Changed = false;
@@ -13,6 +13,7 @@ $end_info$
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/Profiler.h>
#include <algorithm>
#include <cstdint>
@@ -52,6 +53,8 @@ IRCompaction::IRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator)
}
bool IRCompaction::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::IRCompaction");
LocalBuilder.ReownOrClaimBuffer();
auto CurrentIR = IREmit->ViewIR();
@@ -14,6 +14,7 @@ $end_info$
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <cstdint>
#include <memory>
@@ -32,6 +33,8 @@ IRValidation::~IRValidation() {
}
bool IRValidation::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::IRValidation");
bool HadError = false;
bool HadWarning = false;
@@ -9,6 +9,7 @@ $end_info$
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
#include <stdint.h>
@@ -53,6 +54,8 @@ bool LongDivideEliminationPass::IsSextOp(IREmitter *IREmit, OrderedNodeWrapper L
}
bool LongDivideEliminationPass::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::LDE");
bool Changed = false;
auto CurrentIR = IREmit->ViewIR();
auto OriginalWriteCursor = IREmit->GetWriteCursor();
@@ -9,6 +9,7 @@ $end_info$
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include "Interface/IR/PassManager.h"
@@ -24,6 +25,8 @@ public:
};
bool PhiValidation::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::PHIValidation");
bool HadError = false;
auto CurrentIR = IREmit->ViewIR();
@@ -6,7 +6,7 @@
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/IR/RegisterAllocationData.h>
#include <FEXCore/Utils/Profiler.h>
#include <algorithm>
#include <deque>
@@ -191,6 +191,8 @@ private:
bool RAValidation::Run(IREmitter *IREmit) {
if (!Manager->HasPass("RA")) return false;
FEXCORE_PROFILE_SCOPED("PassManager::RAValidation");
IR::RegisterAllocationData* RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
BlockExitState.clear();
// BlocksToVisit will already be empty
@@ -8,6 +8,8 @@ $end_info$
#include <FEXCore/IR/IR.h>
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/Profiler.h>
#include "Interface/IR/PassManager.h"
#include <array>
@@ -32,6 +34,8 @@ public:
*
*/
bool DeadFlagCalculationEliminination::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
std::array<OrderedNode*, 32> LastValidFlagStores{};
bool Changed = false;
@@ -15,6 +15,8 @@ $end_info$
#include <FEXCore/Utils/BucketList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXHeaderUtils/TypeDefines.h>
#include <algorithm>
@@ -1527,6 +1529,7 @@ namespace {
}
bool ConstrainedRAPass::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::RA");
bool Changed = false;
auto IR = IREmit->ViewIR();
@@ -11,6 +11,7 @@ $end_info$
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
#include <stddef.h>
@@ -76,6 +77,8 @@ private:
*
*/
bool StaticRegisterAllocationPass::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::SRA");
auto CurrentIR = IREmit->ViewIR();
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
@@ -11,6 +11,7 @@ $end_info$
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
#include <stdint.h>
@@ -23,6 +24,8 @@ public:
};
bool SyscallOptimization::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::SyscallOpt");
bool Changed = false;
auto CurrentIR = IREmit->ViewIR();
@@ -11,6 +11,7 @@ $end_info$
#include <FEXCore/IR/IREmitter.h>
#include <FEXCore/IR/IntrusiveIRList.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <functional>
#include <memory>
@@ -36,6 +37,8 @@ public:
};
bool ValueDominanceValidation::Run(IREmitter *IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::ValueDominanceValidation");
bool HadError = false;
auto CurrentIR = IREmit->ViewIR();
+31 -7
View File
@@ -145,8 +145,10 @@ namespace FEXCore::Allocator {
#define STEAL_LOG(...) // fprintf(stderr, __VA_ARGS__)
std::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
void * const StackLocation = alloca(0);
const uintptr_t StackLocation_u64 = reinterpret_cast<uintptr_t>(StackLocation);
std::vector<MemoryRegion> Regions;
int MapsFD = open("/proc/self/maps", O_RDONLY);
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
@@ -155,6 +157,8 @@ namespace FEXCore::Allocator {
uintptr_t RegionBegin = 0;
uintptr_t RegionEnd = 0;
uintptr_t PreviousMapEnd = 0;
char Buffer[2048];
const char *Cursor;
ssize_t Remaining = 0;
@@ -162,7 +166,7 @@ namespace FEXCore::Allocator {
for(;;) {
if (Remaining == 0) {
do {
do {
Remaining = read(MapsFD, Buffer, sizeof(Buffer));
} while ( Remaining == -1 && errno == EAGAIN);
@@ -172,8 +176,8 @@ namespace FEXCore::Allocator {
if (Remaining == 0 && State == ParseBegin) {
STEAL_LOG("[%d] EndOfFile; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
auto MapBegin = std::max(RegionEnd, Begin);
auto MapEnd = End;
const auto MapBegin = std::max(RegionEnd, Begin);
const auto MapEnd = End;
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
@@ -209,9 +213,12 @@ namespace FEXCore::Allocator {
if (c == '-') {
STEAL_LOG("[%d] ParseBegin; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
auto MapBegin = std::max(RegionEnd, Begin);
auto MapEnd = std::min(RegionBegin, End);
const auto MapBegin = std::max(RegionEnd, Begin);
const auto MapEnd = std::min(RegionBegin, End);
// Store the location we are going to map.
PreviousMapEnd = MapEnd;
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
if (MapEnd > MapBegin) {
@@ -225,6 +232,7 @@ namespace FEXCore::Allocator {
Regions.push_back({(void*)MapBegin, MapSize});
}
RegionBegin = 0;
RegionEnd = 0;
State = ParseEnd;
@@ -240,6 +248,22 @@ namespace FEXCore::Allocator {
STEAL_LOG("[%d] ParseEnd; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
State = ScanEnd;
// If the previous map's ending and the region we just parsed overlap the stack then we need to save the stack mapping.
// Otherwise we will have severely limited stack size which crashes quickly.
if (PreviousMapEnd <= StackLocation_u64 && RegionEnd > StackLocation_u64) {
auto BelowStackRegion = Regions.back();
LOGMAN_THROW_AA_FMT(reinterpret_cast<uint64_t>(BelowStackRegion.Ptr) + BelowStackRegion.Size == PreviousMapEnd,
"This needs to match");
// Allocate the region under the stack as READ | WRITE so the stack can still grow
auto Alloc = mmap(BelowStackRegion.Ptr, BelowStackRegion.Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", BelowStackRegion.Ptr, BelowStackRegion.Size);
LogMan::Throw::AFmt(Alloc == BelowStackRegion.Ptr, "mmap({},{:x}) returned {} instead of {:x}", Alloc, BelowStackRegion.Ptr);
Regions.pop_back();
}
continue;
} else {
LogMan::Throw::AFmt(std::isalpha(c) || std::isdigit(c), "Unexpected char '{}' in ParseEnd", c);
+83 -102
View File
@@ -69,11 +69,14 @@ namespace Alloc::OSAllocator {
struct LiveVMARegion {
ReservedVMARegion *SlabInfo;
uint64_t FreeSpace{};
uint64_t NumManagedPages{};
uint32_t LastPageAllocation{};
bool HadMunmap{};
// Align UsedPages so it pads to the next page.
// Necessary to take advantage of madvise zero page pooling.
alignas(4096) FEXCore::FlexBitSet<uint64_t> UsedPages;
using FlexBitElementType = uint64_t;
alignas(4096) FEXCore::FlexBitSet<FlexBitElementType> UsedPages;
// This returns the size of the LiveVMARegion in addition to the flex set that tracks the used data
// The LiveVMARegion lives at the start of the VMA region which means on initialization we need to set that
@@ -85,8 +88,8 @@ namespace Alloc::OSAllocator {
// 0x100'0000 Pages
// 1 bit per page for tracking means 0x20'0000 (Pages / 8) bytes of flex space
// Which is 2MB of tracking
uint64_t NumElements = (Size >> FHU::FEX_PAGE_SHIFT) * sizeof(uint64_t);
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<uint64_t>::Size(NumElements);
uint64_t NumElements = (Size >> FHU::FEX_PAGE_SHIFT) * sizeof(FlexBitElementType);
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<FlexBitElementType>::Size(NumElements);
}
static void InitializeVMARegionUsed(LiveVMARegion *Region, size_t AdditionalSize) {
@@ -95,19 +98,21 @@ namespace Alloc::OSAllocator {
Region->FreeSpace = Region->SlabInfo->RegionSize - SizePlusManagedData;
size_t NumPages = SizePlusManagedData >> FHU::FEX_PAGE_SHIFT;
size_t NumManagedPages = SizePlusManagedData >> FHU::FEX_PAGE_SHIFT;
size_t ManagedSize = NumManagedPages << FHU::FEX_PAGE_SHIFT;
// Use madvise to set the full tracking region to zero.
// This ensures unused pages are zero, while not having the backing pages consuming memory.
::madvise(Region->UsedPages.Memory + (NumPages * 4096), (Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT) - (NumPages * 4096), MADV_DONTNEED);
::madvise(Region->UsedPages.Memory + ManagedSize, (Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT) - ManagedSize, MADV_DONTNEED);
// Use madvise to claim WILLNEED on the beginning pages for initial state tracking.
// Improves performance of the following MemClear by not doing a page level fault dance for data necessary to track >170TB of used pages.
::madvise(Region->UsedPages.Memory, NumPages * 4096, MADV_WILLNEED);
::madvise(Region->UsedPages.Memory, ManagedSize, MADV_WILLNEED);
// Set our reserved pages
Region->UsedPages.MemSet(NumPages);
Region->LastPageAllocation = NumPages;
Region->UsedPages.MemSet(NumManagedPages);
Region->LastPageAllocation = NumManagedPages;
Region->NumManagedPages = NumManagedPages;
}
};
@@ -129,6 +134,7 @@ namespace Alloc::OSAllocator {
ReservedVMARegion *ReservedRegion = *ReservedIterator;
ReservedRegions->erase(ReservedIterator);
// mprotect the new region we've allocated
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(ReservedRegion->RegionSize), FHU::FEX_PAGE_SIZE);
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
@@ -152,6 +158,9 @@ namespace Alloc::OSAllocator {
// 32-bit old kernel workarounds
std::vector<FEXCore::Allocator::MemoryRegion> Steal32BitIfOldKernel();
void AllocateMemoryRegions(std::vector<FEXCore::Allocator::MemoryRegion> const &Ranges);
LiveVMARegion *FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd);
};
void OSAllocator_64Bit::DetermineVASize() {
@@ -167,6 +176,42 @@ void OSAllocator_64Bit::DetermineVASize() {
UPPER_BOUND_PAGE = UPPER_BOUND / FHU::FEX_PAGE_SIZE;
}
OSAllocator_64Bit::LiveVMARegion *OSAllocator_64Bit::FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd) {
LiveVMARegion *LiveRegion{};
// Check active slabs to see if we can fit this
for (auto it = LiveRegions->begin(); it != LiveRegions->end(); ++it) {
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
if (Addr >= RegionBegin &&
Addr < RegionEnd) {
LiveRegion = *it;
// Leave our loop
break;
}
}
// Couldn't find an active region that fit
// Check reserved regions
if (!LiveRegion) {
// Didn't have a slab that fit this range
// Check our reserved regions to see if we have one that fits
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
ReservedVMARegion *ReservedRegion = *it;
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
if (Addr >= ReservedRegion->Base &&
AddrEnd < RegionEnd) {
// Found one, let's make it active
LiveRegion = MakeRegionActive(it, 0);
break;
}
}
}
return LiveRegion;
}
void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) {
if (addr != 0 &&
addr < reinterpret_cast<void*>(LOWER_BOUND)) {
@@ -205,41 +250,13 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
LiveVMARegion *LiveRegion{};
if (Fixed || Addr != 0) {
// Check active slabs to see if we can fit this
for (auto it = LiveRegions->begin(); it != LiveRegions->end(); ++it) {
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
if (Addr >= RegionBegin &&
Addr < RegionEnd) {
LiveRegion = *it;
// Leave our loop
break;
}
}
// Couldn't find an active region that fit
// Check reserved regions
if (!LiveRegion) {
// Didn't have a slab that fit this range
// Check our reserved regions to see if we have one that fits
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
ReservedVMARegion *ReservedRegion = *it;
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
if (Addr >= ReservedRegion->Base &&
AddrEnd < RegionEnd) {
// Found one, let's make it active
LiveRegion = MakeRegionActive(it, 0);
break;
}
}
}
LiveRegion = FindLiveRegionForAddress(Addr, AddrEnd);
}
again:
auto CheckIfRangeFits = [&AllocatedOffset](LiveVMARegion *Region, uint64_t length, int prot, int flags, int fd, off_t offset, uint64_t StartingPosition = 0) -> std::pair<LiveVMARegion*, void*> {
uint64_t AllocatedPage{};
uint64_t AllocatedPage{~0ULL};
uint64_t NumberOfPages = length >> FHU::FEX_PAGE_SHIFT;
if (Region->FreeSpace >= length) {
@@ -249,72 +266,29 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
: Region->LastPageAllocation;
size_t RegionNumberOfPages = Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT;
// Backward scan
// We need to do a backward scan first to fill any holes
// Otherwise we will very quickly run out of VMA regions (65k maximum)
for (size_t CurrentPage = LastAllocation;
CurrentPage >= NumberOfPages;) {
size_t Remaining = NumberOfPages;
assert(Remaining <= CurrentPage);
while (Remaining) {
if (Region->UsedPages[CurrentPage - Remaining]) {
// Has an intersecting range
break;
}
--Remaining;
}
if (Region->HadMunmap) {
// Backward scan
// We need to do a backward scan first to fill any holes
// Otherwise we will very quickly run out of VMA regions (65k maximum)
auto SearchResult = Region->UsedPages.BackwardScanForRange<true>(LastAllocation, NumberOfPages, Region->NumManagedPages);
if (Remaining) {
// Didn't find a slab range
CurrentPage -= Remaining;
}
else {
// We have a slab range
CurrentPage -= NumberOfPages;
AllocatedPage = SearchResult.FoundElement;
// Keep scanning backwards to not introduce ANOTHER gap
while (CurrentPage >= 1) {
if (Region->UsedPages[CurrentPage - 1]) {
// Found a used page, we can leave now
break;
}
--CurrentPage;
}
AllocatedPage = CurrentPage;
break;
// If we didn't even have a one page free in the backward search, then unclaim HadMunmap.
// Switching over to default forward search.
if (SearchResult.FoundElement == ~0ULL && !SearchResult.FoundHole) {
Region->HadMunmap = false;
}
}
// Foward Scan
if (AllocatedPage == 0) {
for (size_t CurrentPage = LastAllocation;
CurrentPage < (RegionNumberOfPages - NumberOfPages);) {
// If we have enough free space, check if we have enough free pages that are contiguous
size_t Remaining = NumberOfPages;
assert((CurrentPage + Remaining - 1) < RegionNumberOfPages);
while (Remaining) {
if (Region->UsedPages[CurrentPage + Remaining - 1]) {
// Has an intersecting range
break;
}
--Remaining;
}
if (Remaining) {
// Didn't find a slab range
CurrentPage += Remaining;
}
else {
// We have a slab range
AllocatedPage = CurrentPage;
break;
}
}
if (AllocatedPage == ~0ULL) {
auto SearchResult = Region->UsedPages.ForwardScanForRange<true>(LastAllocation, NumberOfPages, RegionNumberOfPages);
AllocatedPage = SearchResult.FoundElement;
}
if (AllocatedPage) {
if (AllocatedPage != ~0ULL) {
AllocatedOffset = Region->SlabInfo->Base + AllocatedPage * FHU::FEX_PAGE_SIZE;
// We need to setup protections for this
@@ -497,6 +471,8 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
// This will let us more quickly fill holes
(*it)->LastPageAllocation = std::min((*it)->LastPageAllocation, SlabPageBegin);
(*it)->HadMunmap = true;
// XXX: Move region back to reserved list
return 0;
}
@@ -537,12 +513,7 @@ std::vector<FEXCore::Allocator::MemoryRegion> OSAllocator_64Bit::Steal32BitIfOld
return FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND_32, UPPER_BOUND_32);
}
OSAllocator_64Bit::OSAllocator_64Bit() {
DetermineVASize();
auto LowMem = Steal32BitIfOldKernel();
auto Ranges = FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND, UPPER_BOUND);
void OSAllocator_64Bit::AllocateMemoryRegions(std::vector<FEXCore::Allocator::MemoryRegion> const &Ranges) {
for (auto [Ptr, AllocationSize]: Ranges) {
if (!ObjectAlloc) {
auto MaxSize = std::min(size_t(64) * 1024 * 1024, AllocationSize);
@@ -564,12 +535,22 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
continue;
}
}
ReservedVMARegion *Region = ObjectAlloc->new_construct<ReservedVMARegion>();
Region->Base = reinterpret_cast<uint64_t>(Ptr);
Region->RegionSize = AllocationSize;
ReservedRegions->emplace_back(Region);
}
}
OSAllocator_64Bit::OSAllocator_64Bit() {
DetermineVASize();
auto LowMem = Steal32BitIfOldKernel();
auto Ranges = FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND, UPPER_BOUND);
AllocateMemoryRegions(Ranges);
FEXCore::Allocator::ReclaimMemoryRegion(LowMem);
}
+104
View File
@@ -1,6 +1,7 @@
#pragma once
#include <FEXCore/Utils/MathUtils.h>
#include <FEXCore/Utils/LogManager.h>
#include <cstddef>
#include <cstdint>
@@ -38,6 +39,109 @@ struct FlexBitSet final {
memset(Memory, 0xFF, FEXCore::AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
}
// Range scanning results
struct BitsetScanResults {
// Which element was found. ~0ULL if not found.
size_t FoundElement;
// During the scan, found a hole in the allocations that didn't fit.
bool FoundHole;
};
// TODO: Make {Forward,Backward}ScanForRange faster
// Currently these functions test a single bit at a time, which is fairly costly.
// The compiler emits a full element load per iteration, wasting a bunch of time on loads.
// If we change these functions to have a pre-amble and post-amble to align the primary loop to the element size then this can go significantly
// faster.
//
// Once the element scanning is aligned to the element size, we can then use native count leading zero(CLZ) and count trailing zero(CTZ)
// instructions on a full element to scan uint64_t elements per loop iteration.
// Implementation details:
// Template argument WantUnset
// Used to determine if the desired range is for set or unset ranges.
// Typically `WantUnset` should be true. Used for finding a unset range inside of a range will set elements.
//
// @param BeginningElement - The first element in the set to start scanning from.
// @param ElementCount - How many elements to find a range for fitting.
// @param MinimumElement - Minimum element in the set to search to
//
// @return The scan results
template<bool WantUnset>
BitsetScanResults BackwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t MinimumElement) {
bool FoundHole {};
for (size_t CurrentPage = BeginningElement;
CurrentPage >= (MinimumElement + ElementCount);) {
size_t Remaining = ElementCount;
LOGMAN_THROW_AA_FMT(Remaining <= CurrentPage, "Scanning less than available range");
while (Remaining) {
if (this->Get(CurrentPage - Remaining) == WantUnset) {
// Has an intersecting range
break;
}
--Remaining;
}
if (Remaining) {
// If we found at least one Element hole then track that
if (Remaining != ElementCount) {
FoundHole = true;
}
// Didn't find a slab range
CurrentPage -= Remaining;
}
else {
// We have a slab range
return BitsetScanResults{CurrentPage - ElementCount, FoundHole};
}
}
return BitsetScanResults {~0ULL, FoundHole};
}
// @param BeginningElement - The first element in the set to start scanning from.
// @param ElementCount - How many elements to find a range for fitting.
// @param ElementsInSet - How many elements are in the full set.
//
// @return The scan results
template<bool WantUnset>
BitsetScanResults ForwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t ElementsInSet) {
bool FoundHole {};
for (size_t CurrentElement = BeginningElement;
CurrentElement < (ElementsInSet - ElementCount);) {
// If we have enough free space, check if we have enough free pages that are contiguous
size_t Remaining = ElementCount;
LOGMAN_THROW_AA_FMT((CurrentElement + Remaining - 1) < ElementsInSet, "Scanning less than available range");
while (Remaining) {
if (this->Get(CurrentElement + Remaining - 1) == WantUnset) {
// Has an intersecting range
break;
}
--Remaining;
}
if (Remaining) {
// If we found at least one Element hole then track that
if (Remaining != ElementCount) {
FoundHole = true;
}
// Didn't find a slab range
CurrentElement += Remaining;
}
else {
// We have a slab range
return BitsetScanResults {CurrentElement, FoundHole};
}
}
return BitsetScanResults {~0ULL, FoundHole};
}
// This very explicitly doesn't let you take an address
// Is only a getter
bool operator[](size_t Element) const {
+116
View File
@@ -0,0 +1,116 @@
#include <array>
#include <cstdint>
#include <fcntl.h>
#include <limits.h>
#include <linux/magic.h>
#include <string>
#include <sys/stat.h>
#include <sys/vfs.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#define BACKEND_OFF 0
#define BACKEND_GPUVIS 1
#ifdef ENABLE_FEXCORE_PROFILER
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
namespace FEXCore::Profiler {
ProfilerBlock::ProfilerBlock(std::string_view const Format)
: DurationBegin {GetTime()}
, Format {Format} {
}
ProfilerBlock::~ProfilerBlock() {
auto Duration = GetTime() - DurationBegin;
TraceObject(Format, Duration);
}
}
namespace GPUVis {
// ftrace FD for writing trace data.
// Needs to be a raw FD since we hold this open for the entire application execution.
static int TraceFD {-1};
// Need to search the paths to find the real trace path
static std::array<char const*, 2> TraceFSDirectories {
"/sys/kernel/tracing",
"/sys/kernel/debug/tracing",
};
static bool IsTraceFS(char const* Path) {
struct statfs stat;
if (statfs(Path, &stat)) {
return false;
}
return stat.f_type == TRACEFS_MAGIC;
}
void Init() {
for (auto Path : TraceFSDirectories) {
if (IsTraceFS(Path)) {
std::string FilePath = fmt::format("{}/trace_marker", Path);
TraceFD = open(FilePath.c_str(), O_WRONLY | O_CLOEXEC);
if (TraceFD != -1) {
// Opened TraceFD, early exit
break;
}
}
}
}
void Shutdown() {
if (TraceFD != -1) {
close(TraceFD);
TraceFD = -1;
}
}
void TraceObject(std::string_view const Format, uint64_t Duration) {
if (TraceFD != -1) {
// Print the duration as something that began negative duration ago
std::string Event = fmt::format("{} (lduration=-{})\n", Format, Duration);
write(TraceFD, Event.c_str(), Event.size());
}
}
void TraceObject(std::string_view const Format) {
if (TraceFD != -1) {
std::string Event = fmt::format("{}\n", Format);
write(TraceFD, Format.data(), Format.size());
}
}
}
#else
#error Unknown profiler backend
#endif
#endif
namespace FEXCore::Profiler {
#ifdef ENABLE_FEXCORE_PROFILER
void Init() {
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
GPUVis::Init();
#endif
}
void Shutdown() {
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
GPUVis::Shutdown();
#endif
}
void TraceObject(std::string_view const Format, uint64_t Duration) {
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
GPUVis::TraceObject(Format, Duration);
#endif
}
void TraceObject(std::string_view const Format) {
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
GPUVis::TraceObject(Format);
#endif
}
#endif
}
+10 -3
View File
@@ -28,9 +28,16 @@ namespace FEXCore::Core {
uint64_t rip; ///< Current core's RIP. May not be entirely accurate while JIT is active
uint64_t gregs[16];
uint16_t es, cs, ss, ds;
uint64_t gs;
uint64_t fs;
// Raw segment register indexes
uint16_t es_idx, cs_idx, ss_idx, ds_idx;
uint16_t gs_idx, fs_idx;
uint16_t _pad[2];
// Segment registers holding base addresses
uint32_t es_cached, cs_cached, ss_cached, ds_cached;
uint64_t gs_cached;
uint64_t fs_cached;
uint64_t _pad2[1];
XMMRegs xmm;
uint8_t flags[48];
uint64_t mm[8][2];
+55
View File
@@ -0,0 +1,55 @@
#pragma once
#include <cstdint>
#include <string_view>
#include <time.h>
#include <FEXCore/Utils/CompilerDefs.h>
namespace FEXCore::Profiler {
#ifdef ENABLE_FEXCORE_PROFILER
FEX_DEFAULT_VISIBILITY void Init();
FEX_DEFAULT_VISIBILITY void Shutdown();
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format);
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format, uint64_t Duration);
static inline uint64_t GetTime() {
// We want the time in the least amount of overhead possible
// clock_gettime will do a VDSO call with the least amount of overhead
struct timespec ts;
clock_gettime(CLOCK_MONOTONIC, &ts);
return ts.tv_sec * 1'000'000'000ULL + ts.tv_nsec;
}
// A class that follows scoping rules to generate a profile duration block
class ProfilerBlock final {
public:
ProfilerBlock(std::string_view const Format);
~ProfilerBlock();
private:
uint64_t DurationBegin;
std::string_view const Format;
};
#define UniqueScopeName2(name, line) name ## line
#define UniqueScopeName(name, line) UniqueScopeName2(name, line)
// Declare an instantaneous profiler event.
#define FEXCORE_PROFILE_INSTANT(name) FEXCore::Profiler::TraceObject(name)
// Declare a scoped profile block variable with a fixed name.
#define FEXCORE_PROFILE_SCOPED(name) \
FEXCore::Profiler::ProfilerBlock UniqueScopeName(ScopedBlock_, __LINE__) (name)
#else
[[maybe_unused]] static void Init() {}
[[maybe_unused]] static void Shutdown() {}
[[maybe_unused]] static void TraceObject(std::string_view const Format) {}
[[maybe_unused]] static void TraceObject(std::string_view const, uint64_t) {}
#define FEXCORE_PROFILE_INSTANT(...) do {} while(0)
#define FEXCORE_PROFILE_SCOPED(...) do {} while(0)
#endif
}
+1 -1
+4 -4
View File
@@ -148,16 +148,16 @@ def HandleFunctionDeclCursor(Arch, Cursor):
elif (Child.kind == CursorKind.PARM_DECL):
# This gives us a parameter type
Function.Params.append(Child.type.spelling)
elif (Child.kind == CursorKind.UNEXPOSED_ATTR):
# Whatever you are we don't care about you
return Arch
elif (Child.kind == CursorKind.ASM_LABEL_ATTR):
# Whatever you are we don't care about you
return Arch
elif (Child.kind == CursorKind.WARN_UNUSED_RESULT_ATTR):
# Whatever you are we don't care about you
return Arch
elif (Child.kind == CursorKind.VISIBILITY_ATTR):
elif (Child.kind == CursorKind.VISIBILITY_ATTR or
Child.kind == CursorKind.UNEXPOSED_ATTR or
Child.kind == CursorKind.CONST_ATTR or
Child.kind == CursorKind.PURE_ATTR):
pass
else:
logging.critical ("\tUnhandled FunctionDeclCursor {0}-{1}-{2}".format(Child.kind, Child.type.spelling, Child.spelling))
+13 -7
View File
@@ -4,7 +4,7 @@ import subprocess
import os.path
from os import path
# Args: <Known Failures file> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <Test Harness Executable> <Args>...
# Args: <Known Failures file> <Known Failures Type File> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <Test Harness Executable> <Args>...
if (len(sys.argv) < 7):
sys.exit()
@@ -12,19 +12,25 @@ if (len(sys.argv) < 7):
known_failures = {}
disabled_tests = {}
known_failures_file = sys.argv[1]
disabled_tests_file = sys.argv[2]
disabled_tests_type_file = sys.argv[3]
disabled_tests_runner_file = sys.argv[4]
known_failures_type_file = sys.argv[2]
disabled_tests_file = sys.argv[3]
disabled_tests_type_file = sys.argv[4]
disabled_tests_runner_file = sys.argv[5]
current_test = sys.argv[5]
runner = sys.argv[6]
args_start_index = 7
current_test = sys.argv[6]
runner = sys.argv[7]
args_start_index = 8
# Open the known failures file and add it to a dictionary
with open(known_failures_file) as kff:
for line in kff:
known_failures[line.strip()] = 1
if path.exists(known_failures_type_file):
with open(known_failures_type_file) as dtf:
for line in dtf:
known_failures[line.strip()] = 1
with open(disabled_tests_file) as dtf:
for line in dtf:
disabled_tests[line.strip()] = 1
+104 -29
View File
@@ -14,9 +14,9 @@
#include <cstring>
#include <filesystem>
#include <fstream>
#include <list>
#include <random>
#include <string>
#include <vector>
#include <FEXCore/Core/CodeLoader.h>
#include <FEXCore/Core/CoreState.h>
@@ -353,7 +353,36 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
//
// This is still technically a memory leak if the stack grows, but since the primary thread's stack only gets destroyed on process close, this is
// fine.
StackPointer = reinterpret_cast<uintptr_t>(Mapper(nullptr, StackSize(), PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN, -1, 0));
// Stacks need to be allocated at the hint location just like on a real x86 system.
// These are 128MB regions on both x86-64 and x86.
//
// These are required to be in the correct location taking up the appropriate 128MB of space, otherwise the wine preloader crashes FEX.
// This is due to the wine-preloader hardcoding addresses [0x7FFFFE000000 - 0x7FFFFFFF0000) as a top-down
// allocation region. They use mmap with MAP_FIXED, ignoring any previously mapped area at that location and overwriting it.
// Wine-preloader is expecting to allocate 32MB out of the total 128MB stack space in this case. Leaving 96MB for the application.
//
// If FEX doesn't allocate the stack in this region (nullptr mmap hint) then later allocations that FEX does will /eventually/
// end up inside of this address space that wine allocates. This usually ends up being a JIT CodeBuffer, which zeroes the memory and faults with a
// SIGILL.
//
// On the upside, this more accurately emulates how the kernel allocates stack space for the application when hinting at the location.
//
void* StackPointerBase{};
uint64_t StackHint = Is64BitMode() ? STACK_HINT_64 : STACK_HINT_32;
// Allocate the base of the full 128MB stack range.
StackPointerBase = Mapper(reinterpret_cast<void*>(StackHint), FULL_STACK_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN | MAP_NORESERVE, -1, 0);
if (StackPointerBase == reinterpret_cast<void*>(~0ULL)) {
LogMan::Msg::EFmt("Allocating stack failed");
return false;
}
// Allocate with permissions the 8MB of regular stack size.
StackPointer = reinterpret_cast<uintptr_t>(Mapper(
reinterpret_cast<void*>(reinterpret_cast<uint64_t>(StackPointerBase) + FULL_STACK_SIZE - StackSize()),
StackSize(), PROT_READ | PROT_WRITE, MAP_FIXED | MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN, -1, 0));
if (StackPointer == ~0ULL) {
LogMan::Msg::EFmt("Allocating stack failed");
@@ -499,19 +528,16 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
AuxVariables.emplace_back(auxv_t{14, getauxval(AT_EGID)}); // AT_EGID
AuxVariables.emplace_back(auxv_t{17, getauxval(AT_CLKTCK)}); // AT_CLKTIK
AuxVariables.emplace_back(auxv_t{6, 0x1000}); // AT_PAGESIZE
AuxVariables.emplace_back(auxv_t{25, ~0ULL}); // AT_RANDOM
AuxRandom = &AuxVariables.emplace_back(auxv_t{25, ~0ULL}); // AT_RANDOM
AuxVariables.emplace_back(auxv_t{23, 0}); // AT_SECURE
AuxVariables.emplace_back(auxv_t{8, 0}); // AT_FLAGS
AuxVariables.emplace_back(auxv_t{5, MainElf.phdrs.size()}); // AT_PHNUM
AuxVariables.emplace_back(auxv_t{16, HWCap}); // AT_HWCAP
AuxVariables.emplace_back(auxv_t{26, HWCap2}); // AT_HWCAP2
AuxPlatform = &AuxVariables.emplace_back(auxv_t{24, ~0ULL}); // AT_PLATFORM
if (Is64BitMode()) {
AuxVariables.emplace_back(auxv_t{4, 0x38}); // AT_PHENT
// On x86 this is the value returned from CPUID 01h EDX
AuxVariables.emplace_back(auxv_t{16, 0}); // AT_HWCAP
//AuxVariables.emplace_back(auxv_t{24, ~0ULL}); // AT_PLATFORM
// On x86 only allows userspace to check for monitor and fs/gs base writing in CPL3
//AuxVariables.emplace_back(auxv_t{26, 0}); // AT_HWCAP2
// we don't support vsyscall so we don't set those
//AuxVariables.emplace_back(auxv_t{32, 0}); // AT_SYSINFO - Entry point to syscall
@@ -550,10 +576,11 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
uint64_t EnvpOffset,
const std::vector<std::string> &Args,
const std::vector<std::string> &EnvironmentVariables,
const std::vector<auxv_t> &AuxVariables,
const std::list<auxv_t> &AuxVariables,
uint64_t *AuxTabBase,
uint64_t *AuxTabSize,
PointerType RandomNumberOffset
PointerType RandomNumberOffset,
PointerType PlatformNameOffset
) {
// Pointer list offsets
PointerType *ArgumentPointers = reinterpret_cast<PointerType*>(StackPointer + PointerSize);
@@ -607,20 +634,10 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
// Last envp needs to be nullptr
EnvpPointers[EnvironmentVariables.size()] = 0;
for (size_t i = 0; i < AuxVariables.size(); ++i) {
if (AuxVariables[i].key == 25) {
// Random value is always 128bits
AuxType Random{25, static_cast<PointerType>(StackPointer + RandomNumberOffset)};
uint64_t *RandomLoc = reinterpret_cast<uint64_t*>(StackPointer + RandomNumberOffset);
RandomLoc[0] = 0xDEAD;
RandomLoc[1] = 0xDEAD2;
AuxVPointers[i].key = Random.key;
AuxVPointers[i].val = Random.val;
}
else {
AuxVPointers[i].key = AuxVariables[i].key;
AuxVPointers[i].val = AuxVariables[i].val;
}
for (size_t i = 0; auto const &Variable : AuxVariables) {
AuxVPointers[i].key = Variable.key;
AuxVPointers[i].val = Variable.val;
++i;
}
*AuxTabBase = reinterpret_cast<uint64_t>(AuxVPointers);
@@ -656,12 +673,44 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
TotalArgumentMemSize += EnvironmentBackingSize;
// Random number location
uint32_t RandomNumberLocation = TotalArgumentMemSize;
uint64_t RandomNumberLocation = TotalArgumentMemSize;
TotalArgumentMemSize += 16;
uint64_t PlatformNameLocation = TotalArgumentMemSize;
TotalArgumentMemSize += platform_string_max_size;
// Offset the stack by how much memory we need
StackPointer -= TotalArgumentMemSize;
// Setup our AUXP values that need memory now that the stack is setup
AuxPlatform->val = StackPointer + PlatformNameLocation;
char *PlatformLoc = reinterpret_cast<char*>(AuxPlatform->val);
memset(PlatformLoc, 0, platform_string_max_size);
if (Is64BitMode()) {
strncpy(PlatformLoc, platform_name_x86_64.data(), platform_string_max_size);
}
else {
strncpy(PlatformLoc, platform_name_i686.data(), platform_string_max_size);
}
// Random value is always 128bits
AuxRandom->val = StackPointer + RandomNumberLocation;
uint64_t *RandomLoc = reinterpret_cast<uint64_t*>(AuxRandom->val);
uint64_t *HostRandom = reinterpret_cast<uint64_t*>(getauxval(AT_RANDOM));
if (HostRandom) {
// Pass through the host's random values
RandomLoc[0] = HostRandom[0];
RandomLoc[1] = HostRandom[1];
}
else {
// Nothing provided from the kernel, generate our own random values.
std::random_device rd;
std::uniform_int_distribution<uint64_t> d(0);
RandomLoc[0] = d(rd);
RandomLoc[1] = d(rd);
}
// Stack setup
// [0, 8): Argument Count
// [8, 16): Argument Pointer 0
@@ -687,7 +736,8 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
AuxVariables,
&AuxTabBase,
&AuxTabSize,
RandomNumberLocation
RandomNumberLocation,
PlatformNameLocation
);
}
else {
@@ -701,7 +751,8 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
AuxVariables,
&AuxTabBase,
&AuxTabSize,
RandomNumberLocation
RandomNumberLocation,
PlatformNameLocation
);
}
}
@@ -734,19 +785,43 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
VDSOBase = Base;
}
void CalculateHWCaps(FEXCore::Context::Context *ctx) {
// HWCAP is just CPUID function 0x1, the EDX result
auto res_1 = FEXCore::Context::RunCPUIDFunction(ctx, 1, 0);
HWCap = res_1.edx;
// HWCAP2 is as follows:
// Bits:
// 0 - MONITOR/MWAIT available in CPL3
// 1 - FSGSBASE instructions available in CPL3
HWCap2 = 0;
}
constexpr static uint64_t BRK_SIZE = 8 * 1024 * 1024;
constexpr static uint64_t STACK_SIZE = 8 * 1024 * 1024;
constexpr static uint64_t FULL_STACK_SIZE = 128 * 1024 * 1024;
constexpr static uint64_t STACK_HINT_32 = 0xFFFFE000 - FULL_STACK_SIZE;
constexpr static uint64_t STACK_HINT_64 = 0x7FFFFFFFF000 - FULL_STACK_SIZE;
std::vector<std::string> Args;
std::vector<std::string> EnvironmentVariables;
std::vector<char const*> LoaderArgs;
std::vector<auxv_t> AuxVariables;
std::list<auxv_t> AuxVariables;
uint64_t AuxTabBase, AuxTabSize;
uint64_t ArgumentBackingSize{};
uint64_t EnvironmentBackingSize{};
uint64_t BaseOffset{};
void* VDSOBase{};
uint64_t HWCap{};
uint64_t HWCap2{};
auxv_t *AuxRandom{};
auxv_t *AuxPlatform{};
static constexpr std::string_view platform_name_x86_64 = "x86_64";
static constexpr std::string_view platform_name_i686 = "i686";
static constexpr size_t platform_string_max_size = std::max(platform_name_x86_64.size(), platform_name_i686.size());
FEX_CONFIG_OPT(AdditionalArguments, ADDITIONALARGUMENTS);
};
+4
View File
@@ -24,6 +24,7 @@ $end_info$
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Telemetry.h>
#include <FEXCore/Utils/Threads.h>
#include <FEXCore/Utils/Profiler.h>
#include <atomic>
#include <cerrno>
@@ -283,6 +284,7 @@ int main(int argc, char **argv, char **const envp) {
}
}
FEXCore::Profiler::Init();
FEXCore::Telemetry::Initialize();
RootFSRedirect(&Program.first, LDPath());
@@ -393,6 +395,7 @@ int main(int argc, char **argv, char **const envp) {
// Load VDSO in to memory prior to mapping our ELFs.
void* VDSOBase = FEX::VDSO::LoadVDSOThunks(Loader.Is64BitMode(), Mapper);
Loader.SetVDSOBase(VDSOBase);
Loader.CalculateHWCaps(CTX);
if (!Loader.MapMemory(Mapper, Unmapper)) {
// failed to map
@@ -502,6 +505,7 @@ int main(int argc, char **argv, char **const envp) {
FEXCore::Allocator::ReclaimMemoryRegion(Base48Bit);
// Allocator is now original system allocator
FEXCore::Telemetry::Shutdown(Program.second);
FEXCore::Profiler::Shutdown();
if (ShutdownReason == FEXCore::Context::ExitReason::EXIT_SHUTDOWN) {
return ProgramStatus;
}
+6 -6
View File
@@ -127,13 +127,13 @@ namespace FEX::HarnessHelper {
// GS
if (MatchMask & 1) {
CheckGPRs("GS", State1.gs, State2.gs);
CheckGPRs("GS", State1.gs_cached, State2.gs_cached);
}
MatchMask >>= 1;
// FS
if (MatchMask & 1) {
CheckGPRs("FS", State1.fs, State2.fs);
CheckGPRs("FS", State1.fs_cached, State2.fs_cached);
}
MatchMask >>= 1;
@@ -233,8 +233,8 @@ namespace FEX::HarnessHelper {
offsetof(FEXCore::Core::CPUState, xmm.avx.data[13][0]),
offsetof(FEXCore::Core::CPUState, xmm.avx.data[14][0]),
offsetof(FEXCore::Core::CPUState, xmm.avx.data[15][0]),
offsetof(FEXCore::Core::CPUState, gs),
offsetof(FEXCore::Core::CPUState, fs),
offsetof(FEXCore::Core::CPUState, gs_cached),
offsetof(FEXCore::Core::CPUState, fs_cached),
offsetof(FEXCore::Core::CPUState, flags),
offsetof(FEXCore::Core::CPUState, mm[0][0]),
offsetof(FEXCore::Core::CPUState, mm[1][0]),
@@ -280,8 +280,8 @@ namespace FEX::HarnessHelper {
offsetof(FEXCore::Core::CPUState, xmm.sse.data[13][0]),
offsetof(FEXCore::Core::CPUState, xmm.sse.data[14][0]),
offsetof(FEXCore::Core::CPUState, xmm.sse.data[15][0]),
offsetof(FEXCore::Core::CPUState, gs),
offsetof(FEXCore::Core::CPUState, fs),
offsetof(FEXCore::Core::CPUState, gs_cached),
offsetof(FEXCore::Core::CPUState, fs_cached),
offsetof(FEXCore::Core::CPUState, flags),
offsetof(FEXCore::Core::CPUState, mm[0][0]),
offsetof(FEXCore::Core::CPUState, mm[1][0]),
+43 -44
View File
@@ -230,7 +230,7 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
auto LoadThunksDB = [this, ThunkGuestPath](bool *LoadedThunkDatabase, json_t const* ThunksDB) {
// If a thunks DB property exists then we pull in data from the thunks database
// Load the initial thunks database
if (LoadedThunkDatabase) {
if (!*LoadedThunkDatabase) {
LoadThunkDatabase(true);
LoadThunkDatabase(false);
*LoadedThunkDatabase = true;
@@ -239,60 +239,25 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
// Now load this property
for (json_t const* Item = json_getChild(ThunksDB); Item != nullptr; Item = json_getSibling(Item)) {
const char *LibraryName = json_getName(Item);
int64_t LibraryEnabled = json_getInteger(Item);
if (LibraryEnabled != 0) {
// If the library is enabled then find it in the DB
// Enable the overlay and all the dependencies in one go
auto DBObject = ThunkDB.find(LibraryName);
if (DBObject != ThunkDB.end() &&
DBObject->second.Enabled == false) {
auto ThunkPath = ThunkGuestPath / DBObject->second.LibraryName;
if (std::filesystem::exists(ThunkPath)) {
for (auto Overlay : DBObject->second.Overlays) {
// Direct full path in guest RootFS to our overlay file
ThunkOverlays.emplace(Overlay, ThunkPath);
}
}
DBObject->second.Enabled = true;
// Now walk the dependencies and set them up as well
// Make sure to enable each one as we go to remove circular dependencies
std::function<void(std::unordered_set<std::string> &Depends)> InsertDependencies
= [this, &ThunkGuestPath, &InsertDependencies](std::unordered_set<std::string> &Depends) -> void {
for (auto &Depend : Depends) {
auto DBDepend = ThunkDB.find(Depend);
if (DBDepend != ThunkDB.end() &&
DBDepend->second.Enabled == false) {
auto ThunkPath = ThunkGuestPath / DBDepend->second.LibraryName;
if (std::filesystem::exists(ThunkPath)) {
for (auto Overlay : DBDepend->second.Overlays) {
// Direct full path in guest RootFS to our overlay file
ThunkOverlays.emplace(Overlay, ThunkPath);
}
}
// Enabled, now walk this dependencies
DBDepend->second.Enabled = true;
InsertDependencies(DBDepend->second.Depends);
}
}
};
InsertDependencies(DBObject->second.Depends);
}
bool LibraryEnabled = json_getInteger(Item) != 0;
// If the library is enabled then find it in the DB
// Enable the overlay and all the dependencies in one go
auto DBObject = ThunkDB.find(LibraryName);
if (DBObject != ThunkDB.end()) {
DBObject->second.Enabled = LibraryEnabled;
}
}
};
// We try to load ThunksDB from {FEX global config, FEX user config, AppConfig Global, AppConfig Local, Defined ThunksConfig option}
// We try to load ThunksDB from {FEX global config, FEX user config, Defined ThunksConfig option, AppConfig Global, AppConfig Local}
// This doesn't support the classic thunks interface.
std::vector<std::string> ConfigPaths {
FEXCore::Config::GetConfigFileLocation(true),
FEXCore::Config::GetConfigFileLocation(false),
ThunkConfigFile,
FEXCore::Config::GetApplicationConfig(AppConfigName(), true),
FEXCore::Config::GetApplicationConfig(AppConfigName(), false),
ThunkConfigFile,
};
for (const auto &Path : ConfigPaths) {
@@ -313,6 +278,40 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
}
}
// Now that we loaded the thunks object, walk through and ensure dependencies are enabled as well.
for (auto const &DBObject : ThunkDB) {
if (!DBObject.second.Enabled) {
continue;
}
// Now walk the dependencies and set them up as well
// Make sure to enable each one as we go to remove circular dependencies
std::function<void(const std::unordered_set<std::string> &Depends, bool AlreadyEnabled)> InsertDependencies
= [this, &ThunkGuestPath, &InsertDependencies](const std::unordered_set<std::string> &Depends, bool AlreadyEnabled) -> void {
for (auto const &Depend : Depends) {
auto DBDepend = ThunkDB.find(Depend);
if (DBDepend != ThunkDB.end() &&
(DBDepend->second.Enabled == false || AlreadyEnabled)) {
auto ThunkPath = ThunkGuestPath / DBDepend->second.LibraryName;
if (std::filesystem::exists(ThunkPath)) {
for (const auto& Overlay : DBDepend->second.Overlays) {
// Direct full path in guest RootFS to our overlay file
ThunkOverlays.emplace(Overlay, ThunkPath);
}
}
// Enabled, now walk this dependencies
DBDepend->second.Enabled = true;
InsertDependencies(DBDepend->second.Depends, false);
}
}
};
InsertDependencies({DBObject.first}, true);
InsertDependencies(DBObject.second.Depends, false);
}
// Now clear the thunk database since we're loaded
ThunkDB.clear();
@@ -455,7 +455,7 @@ namespace FEX::HLE {
// Ignore a non-canonical address
return -EPERM;
}
Frame->State.gs = addr;
Frame->State.gs_cached = addr;
Result = 0;
break;
case 0x1002: // ARCH_SET_FS
@@ -463,15 +463,15 @@ namespace FEX::HLE {
// Ignore a non-canonical address
return -EPERM;
}
Frame->State.fs = addr;
Frame->State.fs_cached = addr;
Result = 0;
break;
case 0x1003: // ARCH_GET_FS
*reinterpret_cast<uint64_t*>(addr) = Frame->State.fs;
*reinterpret_cast<uint64_t*>(addr) = Frame->State.fs_cached;
Result = 0;
break;
case 0x1004: // ARCH_GET_GS
*reinterpret_cast<uint64_t*>(addr) = Frame->State.gs;
*reinterpret_cast<uint64_t*>(addr) = Frame->State.gs_cached;
Result = 0;
break;
case 0x3001: // ARCH_CET_STATUS
+23
View File
@@ -68,6 +68,29 @@ namespace FEX::HLE::x32 {
// Now we need to update the thread's GDT to handle this change
auto GDT = &Frame->State.gdt[u_info->entry_number];
GDT->base = u_info->base_addr;
// With the segment register optimization we need to check all of the segment registers and update.
const auto GetEntry = [](auto value) {
return value >> 3;
};
if (GetEntry(Frame->State.cs_idx) == u_info->entry_number) {
Frame->State.cs_cached = GDT->base;
}
if (GetEntry(Frame->State.ds_idx) == u_info->entry_number) {
Frame->State.ds_cached = GDT->base;
}
if (GetEntry(Frame->State.es_idx) == u_info->entry_number) {
Frame->State.es_cached = GDT->base;
}
if (GetEntry(Frame->State.fs_idx) == u_info->entry_number) {
Frame->State.fs_cached = GDT->base;
}
if (GetEntry(Frame->State.gs_idx) == u_info->entry_number) {
Frame->State.gs_cached = GDT->base;
}
if (GetEntry(Frame->State.ss_idx) == u_info->entry_number) {
Frame->State.ss_cached = GDT->base;
}
return 0;
}
+2 -2
View File
@@ -36,7 +36,7 @@ namespace FEX::HLE::x64 {
Result = -1;
}
} else {
Result = reinterpret_cast<uint64_t>(FEXCore::Allocator::mmap(reinterpret_cast<void*>(addr), length, prot, flags, fd, offset));
Result = reinterpret_cast<uint64_t>(::mmap(reinterpret_cast<void*>(addr), length, prot, flags, fd, offset));
}
if (Result != -1) {
@@ -56,7 +56,7 @@ namespace FEX::HLE::x64 {
Result = -1;
}
} else {
Result = FEXCore::Allocator::munmap(addr, length);
Result = ::munmap(addr, length);
}
if (Result != -1) {
+1 -1
View File
@@ -24,7 +24,7 @@ $end_info$
namespace FEX::HLE::x64 {
uint64_t SetThreadArea(FEXCore::Core::CpuStateFrame *Frame, void *tls) {
Frame->State.fs = reinterpret_cast<uint64_t>(tls);
Frame->State.fs_cached = reinterpret_cast<uint64_t>(tls);
return 0;
}
+3
View File
@@ -431,6 +431,9 @@ namespace ProcessPipe {
// Invalid
case FEXServerClient::PacketType::TYPE_ERROR:
default:
// Something sent us an invalid packet. To ensure we don't spin infinitely, consume all the data.
LogMan::Msg::EFmt("[FEXServer] InvalidPacket size received 0x{:x} bytes", CurrentRead - CurrentOffset);
CurrentOffset = CurrentRead;
break;
}
}
+21
View File
@@ -103,6 +103,11 @@ function(add_guest_lib NAME SONAME)
## Make signed overflow well defined 2's complement overflow
target_compile_options(${NAME}-guest PRIVATE -fwrapv)
if (CMAKE_CURRENT_SOURCE_DIR STREQUAL CMAKE_SOURCE_DIR)
## Compile for SSE2
## Compile with fpmath=sse to remove x87 usage
target_compile_options(${NAME}-guest PRIVATE -msse2 -mfpmath=sse)
endif()
if (BITNESS EQUAL 32)
# Makes the GOT/PLT lookups slightly less painful
@@ -254,4 +259,20 @@ target_link_options(VDSO-guest PRIVATE "-nostdlib" "LINKER:--no-undefined" "LINK
if (BITNESS EQUAL 32)
# 32-bit entrypoint points to __kernel_vsyscall and needs to exist
target_link_options(VDSO-guest PRIVATE "LINKER:-e,__kernel_vsyscall")
# 32-bit VDSO needs to have PIC disabled.
# Otherwise GCC/Clang generates GOT prologues on the functions that corrupt vsyscall.
# Correct:
# 00000350 <__kernel_vsyscall>:
# 350: cd 80 int 0x80
# 352: c3 ret
# 353: 0f 0b ud2
# Incorrect:
# 0000032a <__kernel_vsyscall>:
# 32a: e8 0b 00 00 00 call 33a <__x86.get_pc_thunk.ax>
# 32f: 05 79 03 00 00 add eax,0x379
# 334: cd 80 int 0x80
# 336: c3 ret
# 337: 90 nop
# 338: 0f 0b ud2
target_compile_options(VDSO-guest PRIVATE "-fno-pic")
endif()
+30 -12
View File
@@ -101,23 +101,41 @@ inline bool IsLibLoaded(const char *libname) {
template<auto Thunk, typename Result, typename... Args>
inline Result CallHostFunction(Args... args) {
#ifndef _M_ARM_64
uintptr_t host_addr;
asm volatile("mov %%r11, %0" : "=r" (host_addr));
// This magic incantation of using a register variable with an empty asm block is necessary for correct operation!
// If we only use inline asm that sets a variable then the compiler will reorder the function
// prologue to be BEFORE our inline asm. Which makes sense in hindsight, but for anything with 8+ arguments this
// will clobber our r11 register we save the data that is inside of it.
// First we need to declare the r11 register variable
register uintptr_t host_addr asm ("r11");
// We then create an empty *volatile* asm block saying that it is assigning the register variable.
// Yes, it is already set coming in to this function due to custom ABI.
// This gets both GCC and Clang to understand that the variable is set, seemingly at the start of the function.
// So its own internal live-range tracking extends its begining range to the start of the function.
//
// To verify this in the future, search for `mov r11` in binaryninja, and ensure that all uses inside of `CallHostFunction`
// don't have intersecting ranges.
//
// Note that this issue is more likely to occur when clang is used to compile thunks, since its optimizer is more aggressive at using R11.
// This magic incantation also works in that instance so this is about the best we can do without adding a new attribute to clang for modifying the
// ABI.
asm volatile("" : "=r" (host_addr));
#else
uintptr_t host_addr = 0;
uintptr_t host_addr = 0;
#endif
PackedArguments<Result, Args..., uintptr_t> packed_args = {
args...,
host_addr
// Return value not explicitly initialized since an initializer would fail to compile for the void case
};
PackedArguments<Result, Args..., uintptr_t> packed_args = {
args...,
host_addr
// Return value not explicitly initialized since an initializer would fail to compile for the void case
};
Thunk(reinterpret_cast<void*>(&packed_args));
Thunk(reinterpret_cast<void*>(&packed_args));
if constexpr (!std::is_void_v<Result>) {
return packed_args.rv;
}
if constexpr (!std::is_void_v<Result>) {
return packed_args.rv;
}
}
// Convenience wrapper that returns the function pointer to a CallHostFunction
+1
View File
@@ -4,6 +4,7 @@
template<auto>
struct fex_gen_config {
unsigned version = 1;
};
template<> struct fex_gen_config<eglBindAPI> {};
+1
View File
@@ -13,6 +13,7 @@
template<auto>
struct fex_gen_config {
unsigned version = 1;
};
template<> struct fex_gen_config<glXGetProcAddress> : fexgen::custom_guest_entrypoint, fexgen::returns_guest_pointer {};
+3 -1
View File
@@ -5,6 +5,8 @@ desc: Handles callbacks and varargs
$end_info$
*/
extern "C" {
#define XUTIL_DEFINE_FUNCTIONS
#include <X11/Xlib.h>
#include <X11/Xutil.h>
#include <X11/Xresource.h>
@@ -16,6 +18,7 @@ $end_info$
#include <X11/Xlibint.h>
#undef min
#undef max
}
#include <cstdint>
#include <stdio.h>
@@ -28,7 +31,6 @@ $end_info$
#include "thunkgen_guest_libX11.inl"
// Custom implementations //
#include <vector>
+25 -12
View File
@@ -5,9 +5,14 @@ desc: Handles callbacks and varargs
$end_info$
*/
#include <atomic>
#include <cstdlib>
#include <stdio.h>
extern "C" {
#define XUTIL_DEFINE_FUNCTIONS
#include <X11/Xproto.h>
#include <X11/XKBlib.h>
#include <X11/Xlib.h>
#include <X11/Xlibint.h>
#undef min
@@ -19,6 +24,7 @@ $end_info$
#include <X11/Xproto.h>
#include <X11/extensions/XKBstr.h>
}
#include "common/Host.h"
#include <dlfcn.h>
@@ -328,25 +334,32 @@ Status fexfn_impl_libX11__XReply(Display*, xReply*, int, Bool);
static int (*ACTUAL_XInitDisplayLock_fn)(Display*) = nullptr;
static int (*INTERNAL_XInitDisplayLock_fn)(Display*) = nullptr;
static std::atomic<bool> Initialized{};
static int _XInitDisplayLock(Display* display) {
auto ret = ACTUAL_XInitDisplayLock_fn(display);
INTERNAL_XInitDisplayLock_fn(display);
return ret;
auto ret = ACTUAL_XInitDisplayLock_fn(display);
INTERNAL_XInitDisplayLock_fn(display);
return ret;
}
Status fexfn_impl_libX11_XInitThreadsInternal(uintptr_t GuestTarget, uintptr_t GuestUnpacker) {
auto ret = fexldr_ptr_libX11_XInitThreads();
auto _XInitDisplayLock_fn = (int(**)(Display*))dlsym(fexldr_ptr_libX11_so, "_XInitDisplayLock_fn");
ACTUAL_XInitDisplayLock_fn = std::exchange(*_XInitDisplayLock_fn, _XInitDisplayLock);
MakeHostTrampolineForGuestFunctionAt(GuestTarget, GuestUnpacker, &INTERNAL_XInitDisplayLock_fn);
return ret;
bool Expected = false;
if (!Initialized.compare_exchange_strong(Expected, true)) {
// If already initialized then this is a no-op.
return 1;
}
auto ret = fexldr_ptr_libX11_XInitThreads();
auto _XInitDisplayLock_fn = (int(**)(Display*))dlsym(fexldr_ptr_libX11_so, "_XInitDisplayLock_fn");
ACTUAL_XInitDisplayLock_fn = std::exchange(*_XInitDisplayLock_fn, _XInitDisplayLock);
MakeHostTrampolineForGuestFunctionAt(GuestTarget, GuestUnpacker, &INTERNAL_XInitDisplayLock_fn);
return ret;
}
Status fexfn_impl_libX11__XReply(Display* display, xReply* reply, int extra, Bool discard) {
for(auto handler = display->async_handlers; handler; handler = handler->next) {
FinalizeHostTrampolineForGuestFunction(handler->handler);
}
return fexldr_ptr_libX11__XReply(display, reply, extra, discard);
for(auto handler = display->async_handlers; handler; handler = handler->next) {
FinalizeHostTrampolineForGuestFunction(handler->handler);
}
return fexldr_ptr_libX11__XReply(display, reply, extra, discard);
}
EXPORTS(libX11)
File diff suppressed because it is too large. Load diff
+541 -394
View File
@@ -20,423 +20,570 @@ template<auto>
struct fex_gen_config : fexgen::generate_guest_symtable, fexgen::indirect_guest_calls {
};
template<> struct fex_gen_config<vkAcquireNextImage2KHR> {};
template<> struct fex_gen_config<vkAcquireNextImageKHR> {};
template<> struct fex_gen_config<vkAcquirePerformanceConfigurationINTEL> {};
template<> struct fex_gen_config<vkAcquireProfilingLockKHR> {};
template<> struct fex_gen_config<vkAcquireXlibDisplayEXT> {};
template<> struct fex_gen_config<vkAllocateCommandBuffers> {};
template<> struct fex_gen_config<vkAllocateDescriptorSets> {};
template<> struct fex_gen_config<vkCreateInstance> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkDestroyInstance> {};
template<> struct fex_gen_config<vkEnumeratePhysicalDevices> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFeatures> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFormatProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceImageFormatProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceQueueFamilyProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceMemoryProperties> {};
// Manually implemented
// template<> struct fex_gen_config<vkGetInstanceProcAddr> {};
// template<> struct fex_gen_config<vkGetDeviceProcAddr> {};
template<> struct fex_gen_config<vkCreateDevice> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkDestroyDevice> {};
template<> struct fex_gen_config<vkEnumerateInstanceExtensionProperties> {};
template<> struct fex_gen_config<vkEnumerateDeviceExtensionProperties> {};
template<> struct fex_gen_config<vkEnumerateInstanceLayerProperties> {};
template<> struct fex_gen_config<vkEnumerateDeviceLayerProperties> {};
template<> struct fex_gen_config<vkGetDeviceQueue> {};
template<> struct fex_gen_config<vkQueueSubmit> {};
template<> struct fex_gen_config<vkQueueWaitIdle> {};
template<> struct fex_gen_config<vkDeviceWaitIdle> {};
template<> struct fex_gen_config<vkAllocateMemory> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkBeginCommandBuffer> {};
template<> struct fex_gen_config<vkBindAccelerationStructureMemoryNV> {};
template<> struct fex_gen_config<vkFreeMemory> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkMapMemory> {};
template<> struct fex_gen_config<vkUnmapMemory> {};
template<> struct fex_gen_config<vkFlushMappedMemoryRanges> {};
template<> struct fex_gen_config<vkInvalidateMappedMemoryRanges> {};
template<> struct fex_gen_config<vkGetDeviceMemoryCommitment> {};
template<> struct fex_gen_config<vkBindBufferMemory> {};
template<> struct fex_gen_config<vkBindBufferMemory2> {};
template<> struct fex_gen_config<vkBindBufferMemory2KHR> {};
template<> struct fex_gen_config<vkBindImageMemory> {};
template<> struct fex_gen_config<vkBindImageMemory2> {};
template<> struct fex_gen_config<vkBindImageMemory2KHR> {};
template<> struct fex_gen_config<vkBuildAccelerationStructuresKHR> {};
template<> struct fex_gen_config<vkCmdBeginConditionalRenderingEXT> {};
template<> struct fex_gen_config<vkCmdBeginDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkCmdBeginQuery> {};
template<> struct fex_gen_config<vkCmdBeginQueryIndexedEXT> {};
template<> struct fex_gen_config<vkCmdBeginRenderPass> {};
template<> struct fex_gen_config<vkCmdBeginRenderPass2> {};
template<> struct fex_gen_config<vkCmdBeginRenderPass2KHR> {};
template<> struct fex_gen_config<vkCmdBeginTransformFeedbackEXT> {};
template<> struct fex_gen_config<vkGetBufferMemoryRequirements> {};
template<> struct fex_gen_config<vkGetImageMemoryRequirements> {};
template<> struct fex_gen_config<vkGetImageSparseMemoryRequirements> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSparseImageFormatProperties> {};
template<> struct fex_gen_config<vkQueueBindSparse> {};
template<> struct fex_gen_config<vkCreateFence> {};
template<> struct fex_gen_config<vkDestroyFence> {};
template<> struct fex_gen_config<vkResetFences> {};
template<> struct fex_gen_config<vkGetFenceStatus> {};
template<> struct fex_gen_config<vkWaitForFences> {};
template<> struct fex_gen_config<vkCreateSemaphore> {};
template<> struct fex_gen_config<vkDestroySemaphore> {};
template<> struct fex_gen_config<vkCreateEvent> {};
template<> struct fex_gen_config<vkDestroyEvent> {};
template<> struct fex_gen_config<vkGetEventStatus> {};
template<> struct fex_gen_config<vkSetEvent> {};
template<> struct fex_gen_config<vkResetEvent> {};
template<> struct fex_gen_config<vkCreateQueryPool> {};
template<> struct fex_gen_config<vkDestroyQueryPool> {};
template<> struct fex_gen_config<vkGetQueryPoolResults> {};
template<> struct fex_gen_config<vkCreateBuffer> {};
template<> struct fex_gen_config<vkDestroyBuffer> {};
template<> struct fex_gen_config<vkCreateBufferView> {};
template<> struct fex_gen_config<vkDestroyBufferView> {};
template<> struct fex_gen_config<vkCreateImage> {};
template<> struct fex_gen_config<vkDestroyImage> {};
template<> struct fex_gen_config<vkGetImageSubresourceLayout> {};
template<> struct fex_gen_config<vkCreateImageView> {};
template<> struct fex_gen_config<vkDestroyImageView> {};
template<> struct fex_gen_config<vkCreateShaderModule> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkDestroyShaderModule> {};
template<> struct fex_gen_config<vkCreatePipelineCache> {};
template<> struct fex_gen_config<vkDestroyPipelineCache> {};
template<> struct fex_gen_config<vkGetPipelineCacheData> {};
template<> struct fex_gen_config<vkMergePipelineCaches> {};
template<> struct fex_gen_config<vkCreateGraphicsPipelines> {};
template<> struct fex_gen_config<vkCreateComputePipelines> {};
template<> struct fex_gen_config<vkDestroyPipeline> {};
template<> struct fex_gen_config<vkCreatePipelineLayout> {};
template<> struct fex_gen_config<vkDestroyPipelineLayout> {};
template<> struct fex_gen_config<vkCreateSampler> {};
template<> struct fex_gen_config<vkDestroySampler> {};
template<> struct fex_gen_config<vkCreateDescriptorSetLayout> {};
template<> struct fex_gen_config<vkDestroyDescriptorSetLayout> {};
template<> struct fex_gen_config<vkCreateDescriptorPool> {};
template<> struct fex_gen_config<vkDestroyDescriptorPool> {};
template<> struct fex_gen_config<vkResetDescriptorPool> {};
template<> struct fex_gen_config<vkAllocateDescriptorSets> {};
template<> struct fex_gen_config<vkFreeDescriptorSets> {};
template<> struct fex_gen_config<vkUpdateDescriptorSets> {};
template<> struct fex_gen_config<vkCreateFramebuffer> {};
template<> struct fex_gen_config<vkDestroyFramebuffer> {};
template<> struct fex_gen_config<vkCreateRenderPass> {};
template<> struct fex_gen_config<vkDestroyRenderPass> {};
template<> struct fex_gen_config<vkGetRenderAreaGranularity> {};
template<> struct fex_gen_config<vkCreateCommandPool> {};
template<> struct fex_gen_config<vkDestroyCommandPool> {};
template<> struct fex_gen_config<vkResetCommandPool> {};
template<> struct fex_gen_config<vkAllocateCommandBuffers> {};
template<> struct fex_gen_config<vkFreeCommandBuffers> {};
template<> struct fex_gen_config<vkBeginCommandBuffer> {};
template<> struct fex_gen_config<vkEndCommandBuffer> {};
template<> struct fex_gen_config<vkResetCommandBuffer> {};
template<> struct fex_gen_config<vkCmdBindPipeline> {};
template<> struct fex_gen_config<vkCmdSetViewport> {};
template<> struct fex_gen_config<vkCmdSetScissor> {};
template<> struct fex_gen_config<vkCmdSetLineWidth> {};
template<> struct fex_gen_config<vkCmdSetDepthBias> {};
template<> struct fex_gen_config<vkCmdSetBlendConstants> {};
template<> struct fex_gen_config<vkCmdSetDepthBounds> {};
template<> struct fex_gen_config<vkCmdSetStencilCompareMask> {};
template<> struct fex_gen_config<vkCmdSetStencilWriteMask> {};
template<> struct fex_gen_config<vkCmdSetStencilReference> {};
template<> struct fex_gen_config<vkCmdBindDescriptorSets> {};
template<> struct fex_gen_config<vkCmdBindIndexBuffer> {};
template<> struct fex_gen_config<vkCmdBindPipeline> {};
template<> struct fex_gen_config<vkCmdBindPipelineShaderGroupNV> {};
template<> struct fex_gen_config<vkCmdBindShadingRateImageNV> {};
template<> struct fex_gen_config<vkCmdBindTransformFeedbackBuffersEXT> {};
template<> struct fex_gen_config<vkCmdBindVertexBuffers> {};
template<> struct fex_gen_config<vkCmdBindVertexBuffers2EXT> {};
template<> struct fex_gen_config<vkCmdDraw> {};
template<> struct fex_gen_config<vkCmdDrawIndexed> {};
template<> struct fex_gen_config<vkCmdDrawIndirect> {};
template<> struct fex_gen_config<vkCmdDrawIndexedIndirect> {};
template<> struct fex_gen_config<vkCmdDispatch> {};
template<> struct fex_gen_config<vkCmdDispatchIndirect> {};
template<> struct fex_gen_config<vkCmdCopyBuffer> {};
template<> struct fex_gen_config<vkCmdCopyImage> {};
template<> struct fex_gen_config<vkCmdBlitImage> {};
template<> struct fex_gen_config<vkCmdBlitImage2KHR> {};
template<> struct fex_gen_config<vkCmdBuildAccelerationStructureNV> {};
template<> struct fex_gen_config<vkCmdBuildAccelerationStructuresIndirectKHR> {};
template<> struct fex_gen_config<vkCmdBuildAccelerationStructuresKHR> {};
template<> struct fex_gen_config<vkCmdClearAttachments> {};
template<> struct fex_gen_config<vkCmdCopyBufferToImage> {};
template<> struct fex_gen_config<vkCmdCopyImageToBuffer> {};
template<> struct fex_gen_config<vkCmdUpdateBuffer> {};
template<> struct fex_gen_config<vkCmdFillBuffer> {};
template<> struct fex_gen_config<vkCmdClearColorImage> {};
template<> struct fex_gen_config<vkCmdClearDepthStencilImage> {};
template<> struct fex_gen_config<vkCmdCopyAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkCmdCopyAccelerationStructureNV> {};
template<> struct fex_gen_config<vkCmdCopyAccelerationStructureToMemoryKHR> {};
template<> struct fex_gen_config<vkCmdCopyBuffer> {};
template<> struct fex_gen_config<vkCmdCopyBuffer2KHR> {};
template<> struct fex_gen_config<vkCmdCopyBufferToImage> {};
template<> struct fex_gen_config<vkCmdCopyBufferToImage2KHR> {};
template<> struct fex_gen_config<vkCmdCopyImage> {};
template<> struct fex_gen_config<vkCmdCopyImage2KHR> {};
template<> struct fex_gen_config<vkCmdCopyImageToBuffer> {};
template<> struct fex_gen_config<vkCmdCopyImageToBuffer2KHR> {};
template<> struct fex_gen_config<vkCmdCopyMemoryToAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkCmdClearAttachments> {};
template<> struct fex_gen_config<vkCmdResolveImage> {};
template<> struct fex_gen_config<vkCmdSetEvent> {};
template<> struct fex_gen_config<vkCmdResetEvent> {};
template<> struct fex_gen_config<vkCmdWaitEvents> {};
template<> struct fex_gen_config<vkCmdPipelineBarrier> {};
template<> struct fex_gen_config<vkCmdBeginQuery> {};
template<> struct fex_gen_config<vkCmdEndQuery> {};
template<> struct fex_gen_config<vkCmdResetQueryPool> {};
template<> struct fex_gen_config<vkCmdWriteTimestamp> {};
template<> struct fex_gen_config<vkCmdCopyQueryPoolResults> {};
template<> struct fex_gen_config<vkCmdPushConstants> {};
template<> struct fex_gen_config<vkCmdBeginRenderPass> {};
template<> struct fex_gen_config<vkCmdNextSubpass> {};
template<> struct fex_gen_config<vkCmdEndRenderPass> {};
template<> struct fex_gen_config<vkCmdExecuteCommands> {};
template<> struct fex_gen_config<vkEnumerateInstanceVersion> {};
template<> struct fex_gen_config<vkBindBufferMemory2> {};
template<> struct fex_gen_config<vkBindImageMemory2> {};
template<> struct fex_gen_config<vkGetDeviceGroupPeerMemoryFeatures> {};
template<> struct fex_gen_config<vkCmdSetDeviceMask> {};
template<> struct fex_gen_config<vkCmdDispatchBase> {};
template<> struct fex_gen_config<vkEnumeratePhysicalDeviceGroups> {};
template<> struct fex_gen_config<vkGetImageMemoryRequirements2> {};
template<> struct fex_gen_config<vkGetBufferMemoryRequirements2> {};
template<> struct fex_gen_config<vkGetImageSparseMemoryRequirements2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFeatures2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFormatProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceImageFormatProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceQueueFamilyProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceMemoryProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSparseImageFormatProperties2> {};
template<> struct fex_gen_config<vkTrimCommandPool> {};
template<> struct fex_gen_config<vkGetDeviceQueue2> {};
template<> struct fex_gen_config<vkCreateSamplerYcbcrConversion> {};
template<> struct fex_gen_config<vkDestroySamplerYcbcrConversion> {};
template<> struct fex_gen_config<vkCreateDescriptorUpdateTemplate> {};
template<> struct fex_gen_config<vkDestroyDescriptorUpdateTemplate> {};
template<> struct fex_gen_config<vkUpdateDescriptorSetWithTemplate> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalBufferProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalFenceProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalSemaphoreProperties> {};
template<> struct fex_gen_config<vkGetDescriptorSetLayoutSupport> {};
template<> struct fex_gen_config<vkCmdDrawIndirectCount> {};
template<> struct fex_gen_config<vkCmdDrawIndexedIndirectCount> {};
template<> struct fex_gen_config<vkCreateRenderPass2> {};
template<> struct fex_gen_config<vkCmdBeginRenderPass2> {};
template<> struct fex_gen_config<vkCmdNextSubpass2> {};
template<> struct fex_gen_config<vkCmdEndRenderPass2> {};
template<> struct fex_gen_config<vkResetQueryPool> {};
template<> struct fex_gen_config<vkGetSemaphoreCounterValue> {};
template<> struct fex_gen_config<vkWaitSemaphores> {};
template<> struct fex_gen_config<vkSignalSemaphore> {};
template<> struct fex_gen_config<vkGetBufferDeviceAddress> {};
template<> struct fex_gen_config<vkGetBufferOpaqueCaptureAddress> {};
template<> struct fex_gen_config<vkGetDeviceMemoryOpaqueCaptureAddress> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceToolProperties> {};
template<> struct fex_gen_config<vkCreatePrivateDataSlot> {};
template<> struct fex_gen_config<vkDestroyPrivateDataSlot> {};
template<> struct fex_gen_config<vkSetPrivateData> {};
template<> struct fex_gen_config<vkGetPrivateData> {};
template<> struct fex_gen_config<vkCmdSetEvent2> {};
template<> struct fex_gen_config<vkCmdResetEvent2> {};
template<> struct fex_gen_config<vkCmdWaitEvents2> {};
template<> struct fex_gen_config<vkCmdPipelineBarrier2> {};
template<> struct fex_gen_config<vkCmdWriteTimestamp2> {};
template<> struct fex_gen_config<vkQueueSubmit2> {};
template<> struct fex_gen_config<vkCmdCopyBuffer2> {};
template<> struct fex_gen_config<vkCmdCopyImage2> {};
template<> struct fex_gen_config<vkCmdCopyBufferToImage2> {};
template<> struct fex_gen_config<vkCmdCopyImageToBuffer2> {};
template<> struct fex_gen_config<vkCmdBlitImage2> {};
template<> struct fex_gen_config<vkCmdResolveImage2> {};
template<> struct fex_gen_config<vkCmdBeginRendering> {};
template<> struct fex_gen_config<vkCmdEndRendering> {};
template<> struct fex_gen_config<vkCmdSetCullMode> {};
template<> struct fex_gen_config<vkCmdSetFrontFace> {};
template<> struct fex_gen_config<vkCmdSetPrimitiveTopology> {};
template<> struct fex_gen_config<vkCmdSetViewportWithCount> {};
template<> struct fex_gen_config<vkCmdSetScissorWithCount> {};
template<> struct fex_gen_config<vkCmdBindVertexBuffers2> {};
template<> struct fex_gen_config<vkCmdSetDepthTestEnable> {};
template<> struct fex_gen_config<vkCmdSetDepthWriteEnable> {};
template<> struct fex_gen_config<vkCmdSetDepthCompareOp> {};
template<> struct fex_gen_config<vkCmdSetDepthBoundsTestEnable> {};
template<> struct fex_gen_config<vkCmdSetStencilTestEnable> {};
template<> struct fex_gen_config<vkCmdSetStencilOp> {};
template<> struct fex_gen_config<vkCmdSetRasterizerDiscardEnable> {};
template<> struct fex_gen_config<vkCmdSetDepthBiasEnable> {};
template<> struct fex_gen_config<vkCmdSetPrimitiveRestartEnable> {};
template<> struct fex_gen_config<vkGetDeviceBufferMemoryRequirements> {};
template<> struct fex_gen_config<vkGetDeviceImageMemoryRequirements> {};
template<> struct fex_gen_config<vkGetDeviceImageSparseMemoryRequirements> {};
template<> struct fex_gen_config<vkDestroySurfaceKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceSupportKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceCapabilitiesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceFormatsKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfacePresentModesKHR> {};
template<> struct fex_gen_config<vkCreateSwapchainKHR> {};
template<> struct fex_gen_config<vkDestroySwapchainKHR> {};
template<> struct fex_gen_config<vkGetSwapchainImagesKHR> {};
template<> struct fex_gen_config<vkAcquireNextImageKHR> {};
template<> struct fex_gen_config<vkQueuePresentKHR> {};
template<> struct fex_gen_config<vkGetDeviceGroupPresentCapabilitiesKHR> {};
template<> struct fex_gen_config<vkGetDeviceGroupSurfacePresentModesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDevicePresentRectanglesKHR> {};
template<> struct fex_gen_config<vkAcquireNextImage2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceDisplayPropertiesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceDisplayPlanePropertiesKHR> {};
template<> struct fex_gen_config<vkGetDisplayPlaneSupportedDisplaysKHR> {};
template<> struct fex_gen_config<vkGetDisplayModePropertiesKHR> {};
template<> struct fex_gen_config<vkCreateDisplayModeKHR> {};
template<> struct fex_gen_config<vkGetDisplayPlaneCapabilitiesKHR> {};
template<> struct fex_gen_config<vkCreateDisplayPlaneSurfaceKHR> {};
template<> struct fex_gen_config<vkCreateSharedSwapchainsKHR> {};
template<> struct fex_gen_config<vkCmdBeginRenderingKHR> {};
template<> struct fex_gen_config<vkCmdEndRenderingKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFeatures2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFormatProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceImageFormatProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceQueueFamilyProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceMemoryProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSparseImageFormatProperties2KHR> {};
template<> struct fex_gen_config<vkGetDeviceGroupPeerMemoryFeaturesKHR> {};
template<> struct fex_gen_config<vkCmdSetDeviceMaskKHR> {};
template<> struct fex_gen_config<vkCmdDispatchBaseKHR> {};
template<> struct fex_gen_config<vkTrimCommandPoolKHR> {};
template<> struct fex_gen_config<vkEnumeratePhysicalDeviceGroupsKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalBufferPropertiesKHR> {};
template<> struct fex_gen_config<vkGetMemoryFdKHR> {};
template<> struct fex_gen_config<vkGetMemoryFdPropertiesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalSemaphorePropertiesKHR> {};
template<> struct fex_gen_config<vkImportSemaphoreFdKHR> {};
template<> struct fex_gen_config<vkGetSemaphoreFdKHR> {};
template<> struct fex_gen_config<vkCmdPushDescriptorSetKHR> {};
template<> struct fex_gen_config<vkCmdPushDescriptorSetWithTemplateKHR> {};
template<> struct fex_gen_config<vkCreateDescriptorUpdateTemplateKHR> {};
template<> struct fex_gen_config<vkDestroyDescriptorUpdateTemplateKHR> {};
template<> struct fex_gen_config<vkUpdateDescriptorSetWithTemplateKHR> {};
template<> struct fex_gen_config<vkCreateRenderPass2KHR> {};
template<> struct fex_gen_config<vkCmdBeginRenderPass2KHR> {};
template<> struct fex_gen_config<vkCmdNextSubpass2KHR> {};
template<> struct fex_gen_config<vkCmdEndRenderPass2KHR> {};
template<> struct fex_gen_config<vkGetSwapchainStatusKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalFencePropertiesKHR> {};
template<> struct fex_gen_config<vkImportFenceFdKHR> {};
template<> struct fex_gen_config<vkGetFenceFdKHR> {};
template<> struct fex_gen_config<vkEnumeratePhysicalDeviceQueueFamilyPerformanceQueryCountersKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceQueueFamilyPerformanceQueryPassesKHR> {};
template<> struct fex_gen_config<vkAcquireProfilingLockKHR> {};
template<> struct fex_gen_config<vkReleaseProfilingLockKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceCapabilities2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceFormats2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceDisplayProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceDisplayPlaneProperties2KHR> {};
template<> struct fex_gen_config<vkGetDisplayModeProperties2KHR> {};
template<> struct fex_gen_config<vkGetDisplayPlaneCapabilities2KHR> {};
template<> struct fex_gen_config<vkGetImageMemoryRequirements2KHR> {};
template<> struct fex_gen_config<vkGetBufferMemoryRequirements2KHR> {};
template<> struct fex_gen_config<vkGetImageSparseMemoryRequirements2KHR> {};
template<> struct fex_gen_config<vkCreateSamplerYcbcrConversionKHR> {};
template<> struct fex_gen_config<vkDestroySamplerYcbcrConversionKHR> {};
template<> struct fex_gen_config<vkBindBufferMemory2KHR> {};
template<> struct fex_gen_config<vkBindImageMemory2KHR> {};
template<> struct fex_gen_config<vkGetDescriptorSetLayoutSupportKHR> {};
template<> struct fex_gen_config<vkCmdDrawIndirectCountKHR> {};
template<> struct fex_gen_config<vkCmdDrawIndexedIndirectCountKHR> {};
template<> struct fex_gen_config<vkGetSemaphoreCounterValueKHR> {};
template<> struct fex_gen_config<vkWaitSemaphoresKHR> {};
template<> struct fex_gen_config<vkSignalSemaphoreKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFragmentShadingRatesKHR> {};
template<> struct fex_gen_config<vkCmdSetFragmentShadingRateKHR> {};
template<> struct fex_gen_config<vkWaitForPresentKHR> {};
template<> struct fex_gen_config<vkGetBufferDeviceAddressKHR> {};
template<> struct fex_gen_config<vkGetBufferOpaqueCaptureAddressKHR> {};
template<> struct fex_gen_config<vkGetDeviceMemoryOpaqueCaptureAddressKHR> {};
template<> struct fex_gen_config<vkCreateDeferredOperationKHR> {};
template<> struct fex_gen_config<vkDestroyDeferredOperationKHR> {};
template<> struct fex_gen_config<vkGetDeferredOperationMaxConcurrencyKHR> {};
template<> struct fex_gen_config<vkGetDeferredOperationResultKHR> {};
template<> struct fex_gen_config<vkDeferredOperationJoinKHR> {};
template<> struct fex_gen_config<vkGetPipelineExecutablePropertiesKHR> {};
template<> struct fex_gen_config<vkGetPipelineExecutableStatisticsKHR> {};
template<> struct fex_gen_config<vkGetPipelineExecutableInternalRepresentationsKHR> {};
template<> struct fex_gen_config<vkCmdSetEvent2KHR> {};
template<> struct fex_gen_config<vkCmdResetEvent2KHR> {};
template<> struct fex_gen_config<vkCmdWaitEvents2KHR> {};
template<> struct fex_gen_config<vkCmdPipelineBarrier2KHR> {};
template<> struct fex_gen_config<vkCmdWriteTimestamp2KHR> {};
template<> struct fex_gen_config<vkQueueSubmit2KHR> {};
template<> struct fex_gen_config<vkCmdWriteBufferMarker2AMD> {};
template<> struct fex_gen_config<vkGetQueueCheckpointData2NV> {};
template<> struct fex_gen_config<vkCmdCopyBuffer2KHR> {};
template<> struct fex_gen_config<vkCmdCopyImage2KHR> {};
template<> struct fex_gen_config<vkCmdCopyBufferToImage2KHR> {};
template<> struct fex_gen_config<vkCmdCopyImageToBuffer2KHR> {};
template<> struct fex_gen_config<vkCmdBlitImage2KHR> {};
template<> struct fex_gen_config<vkCmdResolveImage2KHR> {};
template<> struct fex_gen_config<vkCmdTraceRaysIndirect2KHR> {};
template<> struct fex_gen_config<vkGetDeviceBufferMemoryRequirementsKHR> {};
template<> struct fex_gen_config<vkGetDeviceImageMemoryRequirementsKHR> {};
template<> struct fex_gen_config<vkGetDeviceImageSparseMemoryRequirementsKHR> {};
template<> struct fex_gen_config<vkCreateDebugReportCallbackEXT> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkDestroyDebugReportCallbackEXT> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkDebugReportMessageEXT> {};
template<> struct fex_gen_config<vkDebugMarkerSetObjectTagEXT> {};
template<> struct fex_gen_config<vkDebugMarkerSetObjectNameEXT> {};
template<> struct fex_gen_config<vkCmdDebugMarkerBeginEXT> {};
template<> struct fex_gen_config<vkCmdDebugMarkerEndEXT> {};
template<> struct fex_gen_config<vkCmdDebugMarkerInsertEXT> {};
template<> struct fex_gen_config<vkCmdDispatch> {};
template<> struct fex_gen_config<vkCmdDispatchBase> {};
template<> struct fex_gen_config<vkCmdDispatchBaseKHR> {};
template<> struct fex_gen_config<vkCmdDispatchIndirect> {};
template<> struct fex_gen_config<vkCmdDraw> {};
template<> struct fex_gen_config<vkCmdDrawIndexed> {};
template<> struct fex_gen_config<vkCmdDrawIndexedIndirect> {};
template<> struct fex_gen_config<vkCmdDrawIndexedIndirectCount> {};
template<> struct fex_gen_config<vkCmdDrawIndexedIndirectCountAMD> {};
template<> struct fex_gen_config<vkCmdDrawIndexedIndirectCountKHR> {};
template<> struct fex_gen_config<vkCmdDrawIndirect> {};
template<> struct fex_gen_config<vkCmdDrawIndirectByteCountEXT> {};
template<> struct fex_gen_config<vkCmdDrawIndirectCount> {};
template<> struct fex_gen_config<vkCmdDrawIndirectCountAMD> {};
template<> struct fex_gen_config<vkCmdDrawIndirectCountKHR> {};
template<> struct fex_gen_config<vkCmdDrawMeshTasksIndirectCountNV> {};
template<> struct fex_gen_config<vkCmdDrawMeshTasksIndirectNV> {};
template<> struct fex_gen_config<vkCmdDrawMeshTasksNV> {};
template<> struct fex_gen_config<vkCmdEndConditionalRenderingEXT> {};
template<> struct fex_gen_config<vkCmdEndDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkCmdEndQuery> {};
template<> struct fex_gen_config<vkCmdEndQueryIndexedEXT> {};
template<> struct fex_gen_config<vkCmdEndRenderPass> {};
template<> struct fex_gen_config<vkCmdEndRenderPass2> {};
template<> struct fex_gen_config<vkCmdEndRenderPass2KHR> {};
template<> struct fex_gen_config<vkCmdBindTransformFeedbackBuffersEXT> {};
template<> struct fex_gen_config<vkCmdBeginTransformFeedbackEXT> {};
template<> struct fex_gen_config<vkCmdEndTransformFeedbackEXT> {};
template<> struct fex_gen_config<vkCmdExecuteCommands> {};
template<> struct fex_gen_config<vkCmdExecuteGeneratedCommandsNV> {};
template<> struct fex_gen_config<vkCmdFillBuffer> {};
template<> struct fex_gen_config<vkCmdBeginQueryIndexedEXT> {};
template<> struct fex_gen_config<vkCmdEndQueryIndexedEXT> {};
template<> struct fex_gen_config<vkCmdDrawIndirectByteCountEXT> {};
template<> struct fex_gen_config<vkCreateCuModuleNVX> {};
template<> struct fex_gen_config<vkCreateCuFunctionNVX> {};
template<> struct fex_gen_config<vkDestroyCuModuleNVX> {};
template<> struct fex_gen_config<vkDestroyCuFunctionNVX> {};
template<> struct fex_gen_config<vkCmdCuLaunchKernelNVX> {};
template<> struct fex_gen_config<vkGetImageViewHandleNVX> {};
template<> struct fex_gen_config<vkGetImageViewAddressNVX> {};
template<> struct fex_gen_config<vkCmdDrawIndirectCountAMD> {};
template<> struct fex_gen_config<vkCmdDrawIndexedIndirectCountAMD> {};
template<> struct fex_gen_config<vkGetShaderInfoAMD> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalImageFormatPropertiesNV> {};
template<> struct fex_gen_config<vkCmdBeginConditionalRenderingEXT> {};
template<> struct fex_gen_config<vkCmdEndConditionalRenderingEXT> {};
template<> struct fex_gen_config<vkCmdSetViewportWScalingNV> {};
template<> struct fex_gen_config<vkReleaseDisplayEXT> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceCapabilities2EXT> {};
template<> struct fex_gen_config<vkDisplayPowerControlEXT> {};
template<> struct fex_gen_config<vkRegisterDeviceEventEXT> {};
template<> struct fex_gen_config<vkRegisterDisplayEventEXT> {};
template<> struct fex_gen_config<vkGetSwapchainCounterEXT> {};
template<> struct fex_gen_config<vkGetRefreshCycleDurationGOOGLE> {};
template<> struct fex_gen_config<vkGetPastPresentationTimingGOOGLE> {};
template<> struct fex_gen_config<vkCmdSetDiscardRectangleEXT> {};
template<> struct fex_gen_config<vkSetHdrMetadataEXT> {};
template<> struct fex_gen_config<vkSetDebugUtilsObjectNameEXT> {};
template<> struct fex_gen_config<vkSetDebugUtilsObjectTagEXT> {};
template<> struct fex_gen_config<vkQueueBeginDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkQueueEndDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkQueueInsertDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkCmdBeginDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkCmdEndDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkCmdInsertDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkCmdNextSubpass> {};
template<> struct fex_gen_config<vkCmdNextSubpass2> {};
template<> struct fex_gen_config<vkCmdNextSubpass2KHR> {};
template<> struct fex_gen_config<vkCmdPipelineBarrier> {};
template<> struct fex_gen_config<vkCmdPreprocessGeneratedCommandsNV> {};
template<> struct fex_gen_config<vkCmdPushConstants> {};
template<> struct fex_gen_config<vkCmdPushDescriptorSetKHR> {};
template<> struct fex_gen_config<vkCmdPushDescriptorSetWithTemplateKHR> {};
template<> struct fex_gen_config<vkCmdResetEvent> {};
template<> struct fex_gen_config<vkCmdResetQueryPool> {};
template<> struct fex_gen_config<vkCmdResolveImage> {};
template<> struct fex_gen_config<vkCmdResolveImage2KHR> {};
template<> struct fex_gen_config<vkCmdSetBlendConstants> {};
template<> struct fex_gen_config<vkCmdSetCheckpointNV> {};
template<> struct fex_gen_config<vkCreateDebugUtilsMessengerEXT> {};
template<> struct fex_gen_config<vkDestroyDebugUtilsMessengerEXT> {};
template<> struct fex_gen_config<vkSubmitDebugUtilsMessageEXT> {};
template<> struct fex_gen_config<vkCmdSetSampleLocationsEXT> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceMultisamplePropertiesEXT> {};
template<> struct fex_gen_config<vkGetImageDrmFormatModifierPropertiesEXT> {};
template<> struct fex_gen_config<vkCreateValidationCacheEXT> {};
template<> struct fex_gen_config<vkDestroyValidationCacheEXT> {};
template<> struct fex_gen_config<vkMergeValidationCachesEXT> {};
template<> struct fex_gen_config<vkGetValidationCacheDataEXT> {};
template<> struct fex_gen_config<vkCmdBindShadingRateImageNV> {};
template<> struct fex_gen_config<vkCmdSetViewportShadingRatePaletteNV> {};
template<> struct fex_gen_config<vkCmdSetCoarseSampleOrderNV> {};
template<> struct fex_gen_config<vkCreateAccelerationStructureNV> {};
template<> struct fex_gen_config<vkDestroyAccelerationStructureNV> {};
template<> struct fex_gen_config<vkGetAccelerationStructureMemoryRequirementsNV> {};
template<> struct fex_gen_config<vkBindAccelerationStructureMemoryNV> {};
template<> struct fex_gen_config<vkCmdBuildAccelerationStructureNV> {};
template<> struct fex_gen_config<vkCmdCopyAccelerationStructureNV> {};
template<> struct fex_gen_config<vkCmdTraceRaysNV> {};
template<> struct fex_gen_config<vkCreateRayTracingPipelinesNV> {};
template<> struct fex_gen_config<vkGetRayTracingShaderGroupHandlesKHR> {};
template<> struct fex_gen_config<vkGetRayTracingShaderGroupHandlesNV> {};
template<> struct fex_gen_config<vkGetAccelerationStructureHandleNV> {};
template<> struct fex_gen_config<vkCmdWriteAccelerationStructuresPropertiesNV> {};
template<> struct fex_gen_config<vkCompileDeferredNV> {};
template<> struct fex_gen_config<vkGetMemoryHostPointerPropertiesEXT> {};
template<> struct fex_gen_config<vkCmdWriteBufferMarkerAMD> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceCalibrateableTimeDomainsEXT> {};
template<> struct fex_gen_config<vkGetCalibratedTimestampsEXT> {};
template<> struct fex_gen_config<vkCmdDrawMeshTasksNV> {};
template<> struct fex_gen_config<vkCmdDrawMeshTasksIndirectNV> {};
template<> struct fex_gen_config<vkCmdDrawMeshTasksIndirectCountNV> {};
template<> struct fex_gen_config<vkCmdSetExclusiveScissorNV> {};
template<> struct fex_gen_config<vkCmdSetCheckpointNV> {};
template<> struct fex_gen_config<vkGetQueueCheckpointDataNV> {};
template<> struct fex_gen_config<vkInitializePerformanceApiINTEL> {};
template<> struct fex_gen_config<vkUninitializePerformanceApiINTEL> {};
template<> struct fex_gen_config<vkCmdSetPerformanceMarkerINTEL> {};
template<> struct fex_gen_config<vkCmdSetPerformanceStreamMarkerINTEL> {};
template<> struct fex_gen_config<vkCmdSetPerformanceOverrideINTEL> {};
template<> struct fex_gen_config<vkAcquirePerformanceConfigurationINTEL> {};
template<> struct fex_gen_config<vkReleasePerformanceConfigurationINTEL> {};
template<> struct fex_gen_config<vkQueueSetPerformanceConfigurationINTEL> {};
template<> struct fex_gen_config<vkGetPerformanceParameterINTEL> {};
template<> struct fex_gen_config<vkSetLocalDimmingAMD> {};
template<> struct fex_gen_config<vkGetBufferDeviceAddressEXT> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceToolPropertiesEXT> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceCooperativeMatrixPropertiesNV> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSupportedFramebufferMixedSamplesCombinationsNV> {};
template<> struct fex_gen_config<vkCreateHeadlessSurfaceEXT> {};
template<> struct fex_gen_config<vkCmdSetLineStippleEXT> {};
template<> struct fex_gen_config<vkResetQueryPoolEXT> {};
template<> struct fex_gen_config<vkCmdSetCullModeEXT> {};
template<> struct fex_gen_config<vkCmdSetDepthBias> {};
template<> struct fex_gen_config<vkCmdSetDepthBounds> {};
template<> struct fex_gen_config<vkCmdSetDepthBoundsTestEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetDepthCompareOpEXT> {};
template<> struct fex_gen_config<vkCmdSetFrontFaceEXT> {};
template<> struct fex_gen_config<vkCmdSetPrimitiveTopologyEXT> {};
template<> struct fex_gen_config<vkCmdSetViewportWithCountEXT> {};
template<> struct fex_gen_config<vkCmdSetScissorWithCountEXT> {};
template<> struct fex_gen_config<vkCmdBindVertexBuffers2EXT> {};
template<> struct fex_gen_config<vkCmdSetDepthTestEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetDepthWriteEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetDeviceMask> {};
template<> struct fex_gen_config<vkCmdSetDeviceMaskKHR> {};
template<> struct fex_gen_config<vkCmdSetDiscardRectangleEXT> {};
template<> struct fex_gen_config<vkCmdSetEvent> {};
template<> struct fex_gen_config<vkCmdSetExclusiveScissorNV> {};
template<> struct fex_gen_config<vkCmdSetFragmentShadingRateEnumNV> {};
template<> struct fex_gen_config<vkCmdSetFragmentShadingRateKHR> {};
template<> struct fex_gen_config<vkCmdSetFrontFaceEXT> {};
template<> struct fex_gen_config<vkCmdSetLineStippleEXT> {};
template<> struct fex_gen_config<vkCmdSetLineWidth> {};
template<> struct fex_gen_config<vkCmdSetPerformanceMarkerINTEL> {};
template<> struct fex_gen_config<vkCmdSetPerformanceOverrideINTEL> {};
template<> struct fex_gen_config<vkCmdSetPerformanceStreamMarkerINTEL> {};
template<> struct fex_gen_config<vkCmdSetPrimitiveTopologyEXT> {};
template<> struct fex_gen_config<vkCmdSetRayTracingPipelineStackSizeKHR> {};
template<> struct fex_gen_config<vkCmdSetSampleLocationsEXT> {};
template<> struct fex_gen_config<vkCmdSetScissor> {};
template<> struct fex_gen_config<vkCmdSetScissorWithCountEXT> {};
template<> struct fex_gen_config<vkCmdSetStencilCompareMask> {};
template<> struct fex_gen_config<vkCmdSetStencilOpEXT> {};
template<> struct fex_gen_config<vkCmdSetStencilReference> {};
template<> struct fex_gen_config<vkCmdSetDepthCompareOpEXT> {};
template<> struct fex_gen_config<vkCmdSetDepthBoundsTestEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetStencilTestEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetStencilWriteMask> {};
template<> struct fex_gen_config<vkCmdSetViewport> {};
template<> struct fex_gen_config<vkCmdSetViewportShadingRatePaletteNV> {};
template<> struct fex_gen_config<vkCmdSetViewportWithCountEXT> {};
template<> struct fex_gen_config<vkCmdSetViewportWScalingNV> {};
template<> struct fex_gen_config<vkCmdTraceRaysIndirectKHR> {};
template<> struct fex_gen_config<vkCmdTraceRaysKHR> {};
template<> struct fex_gen_config<vkCmdTraceRaysNV> {};
template<> struct fex_gen_config<vkCmdUpdateBuffer> {};
template<> struct fex_gen_config<vkCmdWaitEvents> {};
template<> struct fex_gen_config<vkCmdWriteAccelerationStructuresPropertiesKHR> {};
template<> struct fex_gen_config<vkCmdWriteAccelerationStructuresPropertiesNV> {};
template<> struct fex_gen_config<vkCmdWriteBufferMarkerAMD> {};
template<> struct fex_gen_config<vkCmdWriteTimestamp> {};
template<> struct fex_gen_config<vkCompileDeferredNV> {};
template<> struct fex_gen_config<vkCmdSetStencilOpEXT> {};
template<> struct fex_gen_config<vkGetGeneratedCommandsMemoryRequirementsNV> {};
template<> struct fex_gen_config<vkCmdPreprocessGeneratedCommandsNV> {};
template<> struct fex_gen_config<vkCmdExecuteGeneratedCommandsNV> {};
template<> struct fex_gen_config<vkCmdBindPipelineShaderGroupNV> {};
template<> struct fex_gen_config<vkCreateIndirectCommandsLayoutNV> {};
template<> struct fex_gen_config<vkDestroyIndirectCommandsLayoutNV> {};
template<> struct fex_gen_config<vkAcquireDrmDisplayEXT> {};
template<> struct fex_gen_config<vkGetDrmDisplayEXT> {};
template<> struct fex_gen_config<vkCreatePrivateDataSlotEXT> {};
template<> struct fex_gen_config<vkDestroyPrivateDataSlotEXT> {};
template<> struct fex_gen_config<vkSetPrivateDataEXT> {};
template<> struct fex_gen_config<vkGetPrivateDataEXT> {};
template<> struct fex_gen_config<vkCmdSetFragmentShadingRateEnumNV> {};
template<> struct fex_gen_config<vkGetImageSubresourceLayout2EXT> {};
template<> struct fex_gen_config<vkGetDeviceFaultInfoEXT> {};
template<> struct fex_gen_config<vkAcquireWinrtDisplayNV> {};
template<> struct fex_gen_config<vkGetWinrtDisplayNV> {};
template<> struct fex_gen_config<vkCmdSetVertexInputEXT> {};
template<> struct fex_gen_config<vkGetDeviceSubpassShadingMaxWorkgroupSizeHUAWEI> {};
template<> struct fex_gen_config<vkCmdSubpassShadingHUAWEI> {};
template<> struct fex_gen_config<vkCmdBindInvocationMaskHUAWEI> {};
template<> struct fex_gen_config<vkGetMemoryRemoteAddressNV> {};
template<> struct fex_gen_config<vkGetPipelinePropertiesEXT> {};
template<> struct fex_gen_config<vkCmdSetPatchControlPointsEXT> {};
template<> struct fex_gen_config<vkCmdSetRasterizerDiscardEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetDepthBiasEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetLogicOpEXT> {};
template<> struct fex_gen_config<vkCmdSetPrimitiveRestartEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetColorWriteEnableEXT> {};
template<> struct fex_gen_config<vkCmdDrawMultiEXT> {};
template<> struct fex_gen_config<vkCmdDrawMultiIndexedEXT> {};
template<> struct fex_gen_config<vkCreateMicromapEXT> {};
template<> struct fex_gen_config<vkDestroyMicromapEXT> {};
template<> struct fex_gen_config<vkCmdBuildMicromapsEXT> {};
template<> struct fex_gen_config<vkBuildMicromapsEXT> {};
template<> struct fex_gen_config<vkCopyMicromapEXT> {};
template<> struct fex_gen_config<vkCopyMicromapToMemoryEXT> {};
template<> struct fex_gen_config<vkCopyMemoryToMicromapEXT> {};
template<> struct fex_gen_config<vkWriteMicromapsPropertiesEXT> {};
template<> struct fex_gen_config<vkCmdCopyMicromapEXT> {};
template<> struct fex_gen_config<vkCmdCopyMicromapToMemoryEXT> {};
template<> struct fex_gen_config<vkCmdCopyMemoryToMicromapEXT> {};
template<> struct fex_gen_config<vkCmdWriteMicromapsPropertiesEXT> {};
template<> struct fex_gen_config<vkGetDeviceMicromapCompatibilityEXT> {};
template<> struct fex_gen_config<vkGetMicromapBuildSizesEXT> {};
template<> struct fex_gen_config<vkSetDeviceMemoryPriorityEXT> {};
template<> struct fex_gen_config<vkGetDescriptorSetLayoutHostMappingInfoVALVE> {};
template<> struct fex_gen_config<vkGetDescriptorSetHostMappingVALVE> {};
template<> struct fex_gen_config<vkCmdSetTessellationDomainOriginEXT> {};
template<> struct fex_gen_config<vkCmdSetDepthClampEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetPolygonModeEXT> {};
template<> struct fex_gen_config<vkCmdSetRasterizationSamplesEXT> {};
template<> struct fex_gen_config<vkCmdSetSampleMaskEXT> {};
template<> struct fex_gen_config<vkCmdSetAlphaToCoverageEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetAlphaToOneEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetLogicOpEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetColorBlendEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetColorBlendEquationEXT> {};
template<> struct fex_gen_config<vkCmdSetColorWriteMaskEXT> {};
template<> struct fex_gen_config<vkCmdSetRasterizationStreamEXT> {};
template<> struct fex_gen_config<vkCmdSetConservativeRasterizationModeEXT> {};
template<> struct fex_gen_config<vkCmdSetExtraPrimitiveOverestimationSizeEXT> {};
template<> struct fex_gen_config<vkCmdSetDepthClipEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetSampleLocationsEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetColorBlendAdvancedEXT> {};
template<> struct fex_gen_config<vkCmdSetProvokingVertexModeEXT> {};
template<> struct fex_gen_config<vkCmdSetLineRasterizationModeEXT> {};
template<> struct fex_gen_config<vkCmdSetLineStippleEnableEXT> {};
template<> struct fex_gen_config<vkCmdSetDepthClipNegativeOneToOneEXT> {};
template<> struct fex_gen_config<vkCmdSetViewportWScalingEnableNV> {};
template<> struct fex_gen_config<vkCmdSetViewportSwizzleNV> {};
template<> struct fex_gen_config<vkCmdSetCoverageToColorEnableNV> {};
template<> struct fex_gen_config<vkCmdSetCoverageToColorLocationNV> {};
template<> struct fex_gen_config<vkCmdSetCoverageModulationModeNV> {};
template<> struct fex_gen_config<vkCmdSetCoverageModulationTableEnableNV> {};
template<> struct fex_gen_config<vkCmdSetCoverageModulationTableNV> {};
template<> struct fex_gen_config<vkCmdSetShadingRateImageEnableNV> {};
template<> struct fex_gen_config<vkCmdSetRepresentativeFragmentTestEnableNV> {};
template<> struct fex_gen_config<vkCmdSetCoverageReductionModeNV> {};
template<> struct fex_gen_config<vkGetShaderModuleIdentifierEXT> {};
template<> struct fex_gen_config<vkGetShaderModuleCreateInfoIdentifierEXT> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceOpticalFlowImageFormatsNV> {};
template<> struct fex_gen_config<vkCreateOpticalFlowSessionNV> {};
template<> struct fex_gen_config<vkDestroyOpticalFlowSessionNV> {};
template<> struct fex_gen_config<vkBindOpticalFlowSessionImageNV> {};
template<> struct fex_gen_config<vkCmdOpticalFlowExecuteNV> {};
template<> struct fex_gen_config<vkGetFramebufferTilePropertiesQCOM> {};
template<> struct fex_gen_config<vkGetDynamicRenderingTilePropertiesQCOM> {};
template<> struct fex_gen_config<vkCreateAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkDestroyAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkCmdBuildAccelerationStructuresKHR> {};
template<> struct fex_gen_config<vkCmdBuildAccelerationStructuresIndirectKHR> {};
template<> struct fex_gen_config<vkBuildAccelerationStructuresKHR> {};
template<> struct fex_gen_config<vkCopyAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkCopyAccelerationStructureToMemoryKHR> {};
template<> struct fex_gen_config<vkCopyMemoryToAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkCreateAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkCreateAccelerationStructureNV> {};
template<> struct fex_gen_config<vkCreateBuffer> {};
template<> struct fex_gen_config<vkCreateBufferView> {};
template<> struct fex_gen_config<vkCreateCommandPool> {};
template<> struct fex_gen_config<vkCreateComputePipelines> {};
template<> struct fex_gen_config<vkCreateDebugReportCallbackEXT> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkCreateDebugUtilsMessengerEXT> {};
template<> struct fex_gen_config<vkCreateDeferredOperationKHR> {};
template<> struct fex_gen_config<vkCreateDescriptorPool> {};
template<> struct fex_gen_config<vkCreateDescriptorSetLayout> {};
template<> struct fex_gen_config<vkCreateDescriptorUpdateTemplate> {};
template<> struct fex_gen_config<vkCreateDescriptorUpdateTemplateKHR> {};
template<> struct fex_gen_config<vkCreateDevice> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkCreateDisplayModeKHR> {};
template<> struct fex_gen_config<vkCreateDisplayPlaneSurfaceKHR> {};
template<> struct fex_gen_config<vkCreateEvent> {};
template<> struct fex_gen_config<vkCreateFence> {};
template<> struct fex_gen_config<vkCreateFramebuffer> {};
template<> struct fex_gen_config<vkCreateGraphicsPipelines> {};
template<> struct fex_gen_config<vkCreateHeadlessSurfaceEXT> {};
template<> struct fex_gen_config<vkCreateImage> {};
template<> struct fex_gen_config<vkCreateImageView> {};
template<> struct fex_gen_config<vkCreateIndirectCommandsLayoutNV> {};
template<> struct fex_gen_config<vkCreateInstance> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkCreatePipelineCache> {};
template<> struct fex_gen_config<vkCreatePipelineLayout> {};
template<> struct fex_gen_config<vkCreatePrivateDataSlotEXT> {};
template<> struct fex_gen_config<vkCreateQueryPool> {};
template<> struct fex_gen_config<vkCreateRayTracingPipelinesKHR> {};
template<> struct fex_gen_config<vkCreateRayTracingPipelinesNV> {};
template<> struct fex_gen_config<vkCreateRenderPass> {};
template<> struct fex_gen_config<vkCreateRenderPass2> {};
template<> struct fex_gen_config<vkCreateRenderPass2KHR> {};
template<> struct fex_gen_config<vkCreateSampler> {};
template<> struct fex_gen_config<vkCreateSamplerYcbcrConversion> {};
template<> struct fex_gen_config<vkCreateSamplerYcbcrConversionKHR> {};
template<> struct fex_gen_config<vkCreateSemaphore> {};
template<> struct fex_gen_config<vkCreateShaderModule> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkCreateSharedSwapchainsKHR> {};
template<> struct fex_gen_config<vkCreateSwapchainKHR> {};
template<> struct fex_gen_config<vkCreateValidationCacheEXT> {};
template<> struct fex_gen_config<vkCreateWaylandSurfaceKHR> {};
template<> struct fex_gen_config<vkCreateXcbSurfaceKHR> {};
template<> struct fex_gen_config<vkCreateXlibSurfaceKHR> {};
template<> struct fex_gen_config<vkDebugMarkerSetObjectNameEXT> {};
template<> struct fex_gen_config<vkDebugMarkerSetObjectTagEXT> {};
template<> struct fex_gen_config<vkDebugReportMessageEXT> {};
template<> struct fex_gen_config<vkDeferredOperationJoinKHR> {};
template<> struct fex_gen_config<vkDestroyAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkDestroyAccelerationStructureNV> {};
template<> struct fex_gen_config<vkDestroyBuffer> {};
template<> struct fex_gen_config<vkDestroyBufferView> {};
template<> struct fex_gen_config<vkDestroyCommandPool> {};
template<> struct fex_gen_config<vkDestroyDebugReportCallbackEXT> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkDestroyDebugUtilsMessengerEXT> {};
template<> struct fex_gen_config<vkDestroyDeferredOperationKHR> {};
template<> struct fex_gen_config<vkDestroyDescriptorPool> {};
template<> struct fex_gen_config<vkDestroyDescriptorSetLayout> {};
template<> struct fex_gen_config<vkDestroyDescriptorUpdateTemplate> {};
template<> struct fex_gen_config<vkDestroyDescriptorUpdateTemplateKHR> {};
template<> struct fex_gen_config<vkDestroyDevice> {};
template<> struct fex_gen_config<vkDestroyEvent> {};
template<> struct fex_gen_config<vkDestroyFence> {};
template<> struct fex_gen_config<vkDestroyFramebuffer> {};
template<> struct fex_gen_config<vkDestroyImage> {};
template<> struct fex_gen_config<vkDestroyImageView> {};
template<> struct fex_gen_config<vkDestroyIndirectCommandsLayoutNV> {};
template<> struct fex_gen_config<vkDestroyInstance> {};
template<> struct fex_gen_config<vkDestroyPipeline> {};
template<> struct fex_gen_config<vkDestroyPipelineCache> {};
template<> struct fex_gen_config<vkDestroyPipelineLayout> {};
template<> struct fex_gen_config<vkDestroyPrivateDataSlotEXT> {};
template<> struct fex_gen_config<vkDestroyQueryPool> {};
template<> struct fex_gen_config<vkDestroyRenderPass> {};
template<> struct fex_gen_config<vkDestroySampler> {};
template<> struct fex_gen_config<vkDestroySamplerYcbcrConversion> {};
template<> struct fex_gen_config<vkDestroySamplerYcbcrConversionKHR> {};
template<> struct fex_gen_config<vkDestroySemaphore> {};
template<> struct fex_gen_config<vkDestroyShaderModule> {};
template<> struct fex_gen_config<vkDestroySurfaceKHR> {};
template<> struct fex_gen_config<vkDestroySwapchainKHR> {};
template<> struct fex_gen_config<vkDestroyValidationCacheEXT> {};
template<> struct fex_gen_config<vkDeviceWaitIdle> {};
template<> struct fex_gen_config<vkDisplayPowerControlEXT> {};
template<> struct fex_gen_config<vkEndCommandBuffer> {};
template<> struct fex_gen_config<vkEnumerateDeviceExtensionProperties> {};
template<> struct fex_gen_config<vkEnumerateDeviceLayerProperties> {};
template<> struct fex_gen_config<vkEnumerateInstanceExtensionProperties> {};
template<> struct fex_gen_config<vkEnumerateInstanceLayerProperties> {};
template<> struct fex_gen_config<vkEnumerateInstanceVersion> {};
template<> struct fex_gen_config<vkEnumeratePhysicalDeviceGroups> {};
template<> struct fex_gen_config<vkEnumeratePhysicalDeviceGroupsKHR> {};
template<> struct fex_gen_config<vkEnumeratePhysicalDeviceQueueFamilyPerformanceQueryCountersKHR> {};
template<> struct fex_gen_config<vkEnumeratePhysicalDevices> {};
template<> struct fex_gen_config<vkFlushMappedMemoryRanges> {};
template<> struct fex_gen_config<vkFreeCommandBuffers> {};
template<> struct fex_gen_config<vkFreeDescriptorSets> {};
template<> struct fex_gen_config<vkFreeMemory> : fexgen::custom_host_impl {};
template<> struct fex_gen_config<vkGetAccelerationStructureBuildSizesKHR> {};
template<> struct fex_gen_config<vkGetAccelerationStructureDeviceAddressKHR> {};
template<> struct fex_gen_config<vkGetAccelerationStructureHandleNV> {};
template<> struct fex_gen_config<vkGetAccelerationStructureMemoryRequirementsNV> {};
template<> struct fex_gen_config<vkGetBufferDeviceAddress> {};
template<> struct fex_gen_config<vkGetBufferDeviceAddressEXT> {};
template<> struct fex_gen_config<vkGetBufferDeviceAddressKHR> {};
template<> struct fex_gen_config<vkGetBufferMemoryRequirements> {};
template<> struct fex_gen_config<vkGetBufferMemoryRequirements2> {};
template<> struct fex_gen_config<vkGetBufferMemoryRequirements2KHR> {};
template<> struct fex_gen_config<vkGetBufferOpaqueCaptureAddress> {};
template<> struct fex_gen_config<vkGetBufferOpaqueCaptureAddressKHR> {};
template<> struct fex_gen_config<vkGetCalibratedTimestampsEXT> {};
template<> struct fex_gen_config<vkGetDeferredOperationMaxConcurrencyKHR> {};
template<> struct fex_gen_config<vkGetDeferredOperationResultKHR> {};
template<> struct fex_gen_config<vkGetDescriptorSetLayoutSupport> {};
template<> struct fex_gen_config<vkGetDescriptorSetLayoutSupportKHR> {};
template<> struct fex_gen_config<vkGetDeviceAccelerationStructureCompatibilityKHR> {};
template<> struct fex_gen_config<vkGetDeviceGroupPeerMemoryFeatures> {};
template<> struct fex_gen_config<vkGetDeviceGroupPeerMemoryFeaturesKHR> {};
template<> struct fex_gen_config<vkGetDeviceGroupPresentCapabilitiesKHR> {};
template<> struct fex_gen_config<vkGetDeviceGroupSurfacePresentModesKHR> {};
template<> struct fex_gen_config<vkGetDeviceMemoryCommitment> {};
template<> struct fex_gen_config<vkGetDeviceMemoryOpaqueCaptureAddress> {};
template<> struct fex_gen_config<vkGetDeviceMemoryOpaqueCaptureAddressKHR> {};
template<> struct fex_gen_config<vkGetDeviceQueue> {};
template<> struct fex_gen_config<vkGetDeviceQueue2> {};
template<> struct fex_gen_config<vkGetDisplayModeProperties2KHR> {};
template<> struct fex_gen_config<vkGetDisplayModePropertiesKHR> {};
template<> struct fex_gen_config<vkGetDisplayPlaneCapabilities2KHR> {};
template<> struct fex_gen_config<vkGetDisplayPlaneCapabilitiesKHR> {};
template<> struct fex_gen_config<vkGetDisplayPlaneSupportedDisplaysKHR> {};
template<> struct fex_gen_config<vkGetEventStatus> {};
template<> struct fex_gen_config<vkGetFenceFdKHR> {};
template<> struct fex_gen_config<vkGetFenceStatus> {};
template<> struct fex_gen_config<vkGetGeneratedCommandsMemoryRequirementsNV> {};
template<> struct fex_gen_config<vkGetImageDrmFormatModifierPropertiesEXT> {};
template<> struct fex_gen_config<vkGetImageMemoryRequirements> {};
template<> struct fex_gen_config<vkGetImageMemoryRequirements2> {};
template<> struct fex_gen_config<vkGetImageMemoryRequirements2KHR> {};
template<> struct fex_gen_config<vkGetImageSparseMemoryRequirements> {};
template<> struct fex_gen_config<vkGetImageSparseMemoryRequirements2> {};
template<> struct fex_gen_config<vkGetImageSparseMemoryRequirements2KHR> {};
template<> struct fex_gen_config<vkGetImageSubresourceLayout> {};
template<> struct fex_gen_config<vkGetImageViewAddressNVX> {};
template<> struct fex_gen_config<vkGetImageViewHandleNVX> {};
template<> struct fex_gen_config<vkGetMemoryFdKHR> {};
template<> struct fex_gen_config<vkGetMemoryFdPropertiesKHR> {};
template<> struct fex_gen_config<vkGetMemoryHostPointerPropertiesEXT> {};
template<> struct fex_gen_config<vkGetPastPresentationTimingGOOGLE> {};
template<> struct fex_gen_config<vkGetPerformanceParameterINTEL> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceCalibrateableTimeDomainsEXT> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceCooperativeMatrixPropertiesNV> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceDisplayPlaneProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceDisplayPlanePropertiesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceDisplayProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceDisplayPropertiesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalBufferProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalBufferPropertiesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalFenceProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalFencePropertiesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalImageFormatPropertiesNV> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalSemaphoreProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceExternalSemaphorePropertiesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFeatures> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFeatures2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFeatures2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFormatProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFormatProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFormatProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceFragmentShadingRatesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceImageFormatProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceImageFormatProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceImageFormatProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceMemoryProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceMemoryProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceMemoryProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceMultisamplePropertiesEXT> {};
template<> struct fex_gen_config<vkGetPhysicalDevicePresentRectanglesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceQueueFamilyPerformanceQueryPassesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceQueueFamilyProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceQueueFamilyProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceQueueFamilyProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSparseImageFormatProperties> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSparseImageFormatProperties2> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSparseImageFormatProperties2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSupportedFramebufferMixedSamplesCombinationsNV> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceCapabilities2EXT> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceCapabilities2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceCapabilitiesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceFormats2KHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceFormatsKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfacePresentModesKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceSurfaceSupportKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceToolPropertiesEXT> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceWaylandPresentationSupportKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceXcbPresentationSupportKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceXlibPresentationSupportKHR> {};
template<> struct fex_gen_config<vkGetPipelineCacheData> {};
template<> struct fex_gen_config<vkGetPipelineExecutableInternalRepresentationsKHR> {};
template<> struct fex_gen_config<vkGetPipelineExecutablePropertiesKHR> {};
template<> struct fex_gen_config<vkGetPipelineExecutableStatisticsKHR> {};
template<> struct fex_gen_config<vkGetPrivateDataEXT> {};
template<> struct fex_gen_config<vkGetQueryPoolResults> {};
template<> struct fex_gen_config<vkGetQueueCheckpointDataNV> {};
template<> struct fex_gen_config<vkGetRandROutputDisplayEXT> {};
template<> struct fex_gen_config<vkGetRayTracingCaptureReplayShaderGroupHandlesKHR> {};
template<> struct fex_gen_config<vkGetRayTracingShaderGroupHandlesKHR> {};
template<> struct fex_gen_config<vkGetRayTracingShaderGroupHandlesNV> {};
template<> struct fex_gen_config<vkGetRayTracingShaderGroupStackSizeKHR> {};
template<> struct fex_gen_config<vkGetRefreshCycleDurationGOOGLE> {};
template<> struct fex_gen_config<vkGetRenderAreaGranularity> {};
template<> struct fex_gen_config<vkGetSemaphoreCounterValue> {};
template<> struct fex_gen_config<vkGetSemaphoreCounterValueKHR> {};
template<> struct fex_gen_config<vkGetSemaphoreFdKHR> {};
template<> struct fex_gen_config<vkGetShaderInfoAMD> {};
template<> struct fex_gen_config<vkGetSwapchainCounterEXT> {};
template<> struct fex_gen_config<vkGetSwapchainImagesKHR> {};
template<> struct fex_gen_config<vkGetSwapchainStatusKHR> {};
template<> struct fex_gen_config<vkGetValidationCacheDataEXT> {};
template<> struct fex_gen_config<vkImportFenceFdKHR> {};
template<> struct fex_gen_config<vkImportSemaphoreFdKHR> {};
template<> struct fex_gen_config<vkInitializePerformanceApiINTEL> {};
template<> struct fex_gen_config<vkInvalidateMappedMemoryRanges> {};
template<> struct fex_gen_config<vkMapMemory> {};
template<> struct fex_gen_config<vkMergePipelineCaches> {};
template<> struct fex_gen_config<vkMergeValidationCachesEXT> {};
template<> struct fex_gen_config<vkQueueBeginDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkQueueBindSparse> {};
template<> struct fex_gen_config<vkQueueEndDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkQueueInsertDebugUtilsLabelEXT> {};
template<> struct fex_gen_config<vkQueuePresentKHR> {};
template<> struct fex_gen_config<vkQueueSetPerformanceConfigurationINTEL> {};
template<> struct fex_gen_config<vkQueueSubmit> {};
template<> struct fex_gen_config<vkQueueWaitIdle> {};
template<> struct fex_gen_config<vkRegisterDeviceEventEXT> {};
template<> struct fex_gen_config<vkRegisterDisplayEventEXT> {};
template<> struct fex_gen_config<vkReleaseDisplayEXT> {};
template<> struct fex_gen_config<vkReleasePerformanceConfigurationINTEL> {};
template<> struct fex_gen_config<vkReleaseProfilingLockKHR> {};
template<> struct fex_gen_config<vkResetCommandBuffer> {};
template<> struct fex_gen_config<vkResetCommandPool> {};
template<> struct fex_gen_config<vkResetDescriptorPool> {};
template<> struct fex_gen_config<vkResetEvent> {};
template<> struct fex_gen_config<vkResetFences> {};
template<> struct fex_gen_config<vkResetQueryPool> {};
template<> struct fex_gen_config<vkResetQueryPoolEXT> {};
template<> struct fex_gen_config<vkSetDebugUtilsObjectNameEXT> {};
template<> struct fex_gen_config<vkSetDebugUtilsObjectTagEXT> {};
template<> struct fex_gen_config<vkSetEvent> {};
template<> struct fex_gen_config<vkSetHdrMetadataEXT> {};
template<> struct fex_gen_config<vkSetLocalDimmingAMD> {};
template<> struct fex_gen_config<vkSetPrivateDataEXT> {};
template<> struct fex_gen_config<vkSignalSemaphore> {};
template<> struct fex_gen_config<vkSignalSemaphoreKHR> {};
template<> struct fex_gen_config<vkSubmitDebugUtilsMessageEXT> {};
template<> struct fex_gen_config<vkTrimCommandPool> {};
template<> struct fex_gen_config<vkTrimCommandPoolKHR> {};
template<> struct fex_gen_config<vkUninitializePerformanceApiINTEL> {};
template<> struct fex_gen_config<vkUnmapMemory> {};
template<> struct fex_gen_config<vkUpdateDescriptorSets> {};
template<> struct fex_gen_config<vkUpdateDescriptorSetWithTemplate> {};
template<> struct fex_gen_config<vkUpdateDescriptorSetWithTemplateKHR> {};
template<> struct fex_gen_config<vkWaitForFences> {};
template<> struct fex_gen_config<vkWaitSemaphores> {};
template<> struct fex_gen_config<vkWaitSemaphoresKHR> {};
template<> struct fex_gen_config<vkWriteAccelerationStructuresPropertiesKHR> {};
template<> struct fex_gen_config<vkCmdCopyAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkCmdCopyAccelerationStructureToMemoryKHR> {};
template<> struct fex_gen_config<vkCmdCopyMemoryToAccelerationStructureKHR> {};
template<> struct fex_gen_config<vkGetAccelerationStructureDeviceAddressKHR> {};
template<> struct fex_gen_config<vkCmdWriteAccelerationStructuresPropertiesKHR> {};
template<> struct fex_gen_config<vkGetDeviceAccelerationStructureCompatibilityKHR> {};
template<> struct fex_gen_config<vkGetAccelerationStructureBuildSizesKHR> {};
template<> struct fex_gen_config<vkCmdTraceRaysKHR> {};
template<> struct fex_gen_config<vkCreateRayTracingPipelinesKHR> {};
template<> struct fex_gen_config<vkGetRayTracingCaptureReplayShaderGroupHandlesKHR> {};
template<> struct fex_gen_config<vkCmdTraceRaysIndirectKHR> {};
template<> struct fex_gen_config<vkGetRayTracingShaderGroupStackSizeKHR> {};
template<> struct fex_gen_config<vkCmdSetRayTracingPipelineStackSizeKHR> {};
template<> struct fex_gen_config<vkCmdDrawMeshTasksEXT> {};
template<> struct fex_gen_config<vkCmdDrawMeshTasksIndirectEXT> {};
template<> struct fex_gen_config<vkCmdDrawMeshTasksIndirectCountEXT> {};
// vulkan_xlib_xrandr.h
template<> struct fex_gen_config<vkAcquireXlibDisplayEXT> {};
template<> struct fex_gen_config<vkGetRandROutputDisplayEXT> {};
// vulkan_wayland.h
template<> struct fex_gen_config<vkCreateWaylandSurfaceKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceWaylandPresentationSupportKHR> {};
// vulkan_xcb.h
template<> struct fex_gen_config<vkCreateXcbSurfaceKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceXcbPresentationSupportKHR> {};
// vulkan_xlib.h
template<> struct fex_gen_config<vkCreateXlibSurfaceKHR> {};
template<> struct fex_gen_config<vkGetPhysicalDeviceXlibPresentationSupportKHR> {};
} // namespace internal
@@ -3,7 +3,9 @@
#include <xcb/dri2.h>
template<auto>
struct fex_gen_config;
struct fex_gen_config {
unsigned version = 0;
};
void FEX_xcb_dri2_init_extension(xcb_connection_t*, xcb_extension_t*);
size_t FEX_usable_size(void*);
@@ -3,7 +3,9 @@
#include <xcb/dri3.h>
template<auto>
struct fex_gen_config;
struct fex_gen_config {
unsigned version = 0;
};
void FEX_xcb_dri3_init_extension(xcb_connection_t*, xcb_extension_t*);
size_t FEX_usable_size(void*);
@@ -3,7 +3,9 @@
#include <xcb/glx.h>
template<auto>
struct fex_gen_config;
struct fex_gen_config {
unsigned version = 0;
};
void FEX_xcb_glx_init_extension(xcb_connection_t*, xcb_extension_t*);
size_t FEX_usable_size(void*);
@@ -3,7 +3,9 @@
#include <xcb/present.h>
template<auto>
struct fex_gen_config;
struct fex_gen_config {
unsigned version = 0;
};
void FEX_xcb_present_init_extension(xcb_connection_t*, xcb_extension_t*);
size_t FEX_usable_size(void*);
@@ -3,7 +3,9 @@
#include <xcb/randr.h>
template<auto>
struct fex_gen_config;
struct fex_gen_config {
unsigned version = 0;
};
void FEX_xcb_randr_init_extension(xcb_connection_t*, xcb_extension_t*);
size_t FEX_usable_size(void*);
@@ -3,7 +3,9 @@
#include <xcb/shm.h>
template<auto>
struct fex_gen_config;
struct fex_gen_config {
unsigned version = 0;
};
void FEX_xcb_shm_init_extension(xcb_connection_t*, xcb_extension_t*);
size_t FEX_usable_size(void*);
@@ -3,7 +3,9 @@
#include <xcb/xfixes.h>
template<auto>
struct fex_gen_config;
struct fex_gen_config {
unsigned version = 0;
};
void FEX_xcb_xfixes_init_extension(xcb_connection_t*, xcb_extension_t*);
size_t FEX_usable_size(void*);
+3 -1
View File
@@ -7,7 +7,9 @@
#include "WorkEventData.h"
template<auto>
struct fex_gen_config;
struct fex_gen_config {
unsigned version = 1;
};
void FEX_xcb_init_extension(xcb_connection_t*, xcb_extension_t*);
size_t FEX_usable_size(void*);
+1 -1
View File
@@ -1,4 +1,4 @@
# FEX-2210
# FEX-2211
## External/FEXCore
See [FEXCore/Readme.md](../External/FEXCore/Readme.md) for more details
+1
View File
@@ -84,6 +84,7 @@ foreach(ASM_SRC ${ASM_SOURCES})
add_test(NAME ${TEST_NAME}
COMMAND "python3" "${CMAKE_SOURCE_DIR}/Scripts/testharness_runner.py"
"${CMAKE_SOURCE_DIR}/unittests/32Bit_ASM/Known_Failures"
"${CMAKE_SOURCE_DIR}/unittests/32Bit_ASM/Known_Failures_${TEST_TYPE}"
"${CMAKE_SOURCE_DIR}/unittests/32Bit_ASM/Disabled_Tests"
"${CMAKE_SOURCE_DIR}/unittests/32Bit_ASM/Disabled_Tests_${TEST_TYPE}"
"${CMAKE_SOURCE_DIR}/unittests/32Bit_ASM/Disabled_Tests_${CPU_CLASS}"
@@ -0,0 +1,15 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0x12345637"
},
"Mode": "32BIT"
}
%endif
mov eax, 0x1234561f
daa
daa
daa
daa
hlt
@@ -0,0 +1,15 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0x12345607"
},
"Mode": "32BIT"
}
%endif
mov eax, 0x1234561f
das
das
das
das
hlt
@@ -0,0 +1,15 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0x12345a07"
},
"Mode": "32BIT"
}
%endif
mov eax, 0x1234561f
aaa
aaa
aaa
aaa
hlt
@@ -0,0 +1,15 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0x12345107"
},
"Mode": "32BIT"
}
%endif
mov eax, 0x1234561f
aas
aas
aas
aas
hlt
@@ -0,0 +1,15 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0x2"
},
"Mode": "32BIT"
}
%endif
mov eax, 0x1234
aam
aam 0xc
aam 0x1f
aam 0xff
hlt
@@ -0,0 +1,15 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "0xe8"
},
"Mode": "32BIT"
}
%endif
mov eax, 0x1234
aad
aad 0x3
aad 0x1f
aad 0xae
hlt
+1
View File
@@ -92,6 +92,7 @@ foreach(ASM_SRC ${ASM_SOURCES})
add_test(NAME ${TEST_NAME}
COMMAND "python3" "${CMAKE_SOURCE_DIR}/Scripts/testharness_runner.py"
"${CMAKE_SOURCE_DIR}/unittests/ASM/Known_Failures"
"${CMAKE_SOURCE_DIR}/unittests/ASM/Known_Failures_${TEST_TYPE}"
"${CMAKE_SOURCE_DIR}/unittests/ASM/Disabled_Tests"
"${CMAKE_SOURCE_DIR}/unittests/ASM/Disabled_Tests_${TEST_TYPE}"
"${CMAKE_SOURCE_DIR}/unittests/ASM/Disabled_Tests_${CPU_CLASS}"
-12
View File
@@ -70,15 +70,3 @@ Test_H0F3A/66_09.asm
Test_H0F3A/66_0A.asm
Test_H0F3A/66_0B.asm
Test_OpSize/66_5B.asm
# Simulator has a bug in narrowing or widening operations
Test_H0F38/66_03.asm
Test_H0F38/66_04.asm
Test_H0F38/66_07.asm
Test_H0F38/66_2B.asm
Test_H0F38/XX_03.asm
Test_H0F38/XX_04.asm
Test_H0F38/XX_07.asm
Test_OpSize/66_63.asm
Test_OpSize/66_67.asm
Test_OpSize/66_6B.asm
+2 -2
View File
@@ -3,7 +3,7 @@
"RegData": {
"XMM0": ["0x0", "0x0"],
"XMM1": ["0xFE02FE02FE02FE02", "0xFE02FE02FE02FE02"],
"XMM2": ["0x7F7F7F7F7F7F7F7F", "0x7F7F7F7F7F7F7F7F"],
"XMM2": ["0x7E027E027E027E02", "0x7E027E027E027E02"],
"XMM3": ["0x7FFF7FFF7FFF7FFF", "0x7FFF7FFF7FFF7FFF"],
"XMM4": ["0x057306BC07B808B8", "0xBC53BC0EBAE5BA2E"],
"XMM5": ["0xA473A5BCA6B8A7B8", "0x0553070E07E5092E"]
@@ -45,7 +45,7 @@ pmaddubsw xmm1, [rdx + 8 * 2]
; 127
movaps xmm2, [rdx + 8 * 4]
pmaddubsw mm2, [rdx + 8 * 4]
pmaddubsw xmm2, [rdx + 8 * 4]
; 255 and 127
movaps xmm3, [rdx + 8 * 2]
+3
View File
@@ -0,0 +1,3 @@
# FPREM is incorrect
Test_X87/D9_F5_2.asm
Test_X87/D9_F5_3.asm
+3
View File
@@ -0,0 +1,3 @@
# FPREM is incorrect
Test_X87/D9_F5_2.asm
Test_X87/D9_F5_3.asm
+152
View File
@@ -0,0 +1,152 @@
%ifdef CONFIG
{
"RegData": {
"R15": "0x0000000001060004"
}
}
%endif
%macro cfmerge 0
; Get CF
lahf
shr rax, 8
and rax, 1
; Merge in to results
shl r15, 1
or r15, rax
%endmacro
stc
cfmerge
; 8-bit
; Shift 1 past size - Bit Set
mov rbx, 0x800
rol bl, 9
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x000
rol bl, 9
cfmerge
; Shift size - Bit Set
mov rbx, 0x80
rol bl, 8
cfmerge
; Shift size - Bit unset
mov rbx, 0x8000
rol bl, 8
cfmerge
; 8-bit - wrapped
; Shift 1 past size - Bit Set
mov rbx, 0x01
rol bl, 9
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0xFFF2
rol bl, 9
cfmerge
; Shift size - Bit Set
mov rbx, 0xFF
rol bl, 8
cfmerge
; Shift size - Bit unset
mov rbx, 0xFF00
rol bl, 8
cfmerge
; 16-bit
; Shift 1 past size - Bit Set
mov rbx, 0x80000
rol bx, 17
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x00000
rol bx, 17
cfmerge
; Shift size - Bit Set
mov rbx, 0x8000
rol bx, 16
cfmerge
; Shift size - Bit unset
mov rbx, 0x80000
rol bx, 16
cfmerge
; 32-bit
; Shift 1 past size - Bit Set
mov rbx, 0x800000000
rol ebx, 33
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x000000000
rol ebx, 33
cfmerge
; Shift size - Bit Set
mov rbx, 0x80000000
rol ebx, 32
cfmerge
; Shift size - Bit unset
mov rbx, 0x800000000
rol ebx, 32
cfmerge
; 32-bit - Wrapping
; Shift 1 past size - Bit Set
mov rbx, 0x02
rol ebx, 33
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x01
rol ebx, 33
cfmerge
; Shift size - Bit Set
mov rbx, 0x1
rol ebx, 32
cfmerge
; Shift size - Bit unset
mov rbx, 0x02
rol ebx, 32
cfmerge
; 64-bit
; Shift 1 past size - Bit Set
mov rbx, 0x02
rol rbx, 65
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x8000000000000000
rol rbx, 65
cfmerge
; Shift size - Bit Set
mov rbx, 0x1
rol rbx, 64
cfmerge
; Shift size - Bit unset
mov rbx, 0x02
rol rbx, 64
cfmerge
hlt
+72
View File
@@ -0,0 +1,72 @@
%ifdef CONFIG
{
"RegData": {
"R15": "0x00000000000000aa"
}
}
%endif
%macro clearof 0
mov r12, 0
ror r12, 1
%endmacro
%macro ofmerge 0
mov r14, 0
mov r13, 1
cmovo r14, r13
or r15, r14
shl r15, 1
%endmacro
mov r15, 0
mov r14, 1
; 1 bit rotate
; rol OF = XOR of LSB and MSB after rotate
clearof
mov rax, 0
mov rcx, 1
rol rax, cl
ofmerge
clearof
mov rax, 0x8000000000000000
mov rcx, 1
rol rax, cl
ofmerge
clearof
mov rax, 0xC000000000000000
mov rcx, 1
rol rax, cl
ofmerge
clearof
mov rax, 0x4000000000000000
mov rcx, 1
rol rax, cl
ofmerge
clearof
mov rax, 0
rol rax, 1
ofmerge
clearof
mov rax, 0x8000000000000000
rol rax, 1
ofmerge
clearof
mov rax, 0xC000000000000000
rol rax, 1
ofmerge
clearof
mov rax, 0x4000000000000000
rol rax, 1
ofmerge
hlt
+152
View File
@@ -0,0 +1,152 @@
%ifdef CONFIG
{
"RegData": {
"R15": "0x00000000012a2040"
}
}
%endif
%macro cfmerge 0
; Get CF
lahf
shr rax, 8
and rax, 1
; Merge in to results
shl r15, 1
or r15, rax
%endmacro
stc
cfmerge
; 8-bit
; Shift 1 past size - Bit Set
mov rbx, 0x800
ror bl, 9
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x000
ror bl, 9
cfmerge
; Shift size - Bit Set
mov rbx, 0x80
ror bl, 8
cfmerge
; Shift size - Bit unset
mov rbx, 0x8000
ror bl, 8
cfmerge
; 8-bit - wrapped
; Shift 1 past size - Bit Set
mov rbx, 0x01
ror bl, 9
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0xFFF2
ror bl, 9
cfmerge
; Shift size - Bit Set
mov rbx, 0xFF
ror bl, 8
cfmerge
; Shift size - Bit unset
mov rbx, 0xFF00
ror bl, 8
cfmerge
; 16-bit
; Shift 1 past size - Bit Set
mov rbx, 0x80000
ror bx, 17
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x00000
ror bx, 17
cfmerge
; Shift size - Bit Set
mov rbx, 0x8000
ror bx, 16
cfmerge
; Shift size - Bit unset
mov rbx, 0x80000
ror bx, 16
cfmerge
; 32-bit
; Shift 1 past size - Bit Set
mov rbx, 0x800000000
ror ebx, 33
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x000000000
ror ebx, 33
cfmerge
; Shift size - Bit Set
mov rbx, 0x80000000
ror ebx, 32
cfmerge
; Shift size - Bit unset
mov rbx, 0x800000000
ror ebx, 32
cfmerge
; 32-bit - Wrapping
; Shift 1 past size - Bit Set
mov rbx, 0x02
ror ebx, 33
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x01
ror ebx, 33
cfmerge
; Shift size - Bit Set
mov rbx, 0x1
ror ebx, 32
cfmerge
; Shift size - Bit unset
mov rbx, 0x02
ror ebx, 32
cfmerge
; 64-bit
; Shift 1 past size - Bit Set
mov rbx, 0x02
ror rbx, 65
cfmerge
; Shift 1 past size - Bit unset
mov rbx, 0x8000000000000000
ror rbx, 65
cfmerge
; Shift size - Bit Set
mov rbx, 0x1
ror rbx, 64
cfmerge
; Shift size - Bit unset
mov rbx, 0x02
ror rbx, 64
cfmerge
hlt
+72
View File
@@ -0,0 +1,72 @@
%ifdef CONFIG
{
"RegData": {
"R15": "0x00000000000000cc"
}
}
%endif
%macro clearof 0
mov r12, 0
ror r12, 1
%endmacro
%macro ofmerge 0
mov r14, 0
mov r13, 1
cmovo r14, r13
or r15, r14
shl r15, 1
%endmacro
mov r15, 0
mov r14, 1
; 1 bit rotate
; ror OF = XOR or two most significant bits of result
clearof
mov rax, 0
mov rcx, 1
ror rax, cl
ofmerge
clearof
mov rax, 1
mov rcx, 1
ror rax, cl
ofmerge
clearof
mov rax, 0x8000000000000000
mov rcx, 1
ror rax, cl
ofmerge
clearof
mov rax, 0x8000000000000001
mov rcx, 1
ror rax, cl
ofmerge
clearof
mov rax, 0
ror rax, 1
ofmerge
clearof
mov rax, 1
ror rax, 1
ofmerge
clearof
mov rax, 0x8000000000000000
ror rax, 1
ofmerge
clearof
mov rax, 0x8000000000000001
ror rax, 1
ofmerge
hlt
+43
View File
@@ -0,0 +1,43 @@
%ifdef CONFIG
{
"RegData": {
"R14": "0xffffffffffffffff",
"R13": "0xfffffffffffffffe",
"R12": "0xfffffffffffffffc",
"R11": "0xfffffffffffffff8",
"R10": "0xfffffffffffffff0",
"R9": "0xffffffffffffffe0",
"R8": "0xffffffffffffffc0",
"RBP": "0xffffffffffffff80",
"RSP": "0xffffffffffffff00",
"RDI": "0xffffffffffffff00",
"RSI": "0xffffffffffffff00"
}
}
%endif
mov r14, -1
mov r13, -1
mov r12, -1
mov r11, -1
mov r10, -1
mov r9, -1
mov r8, -1
mov rbp, -1
mov rsp, -1
mov rdi, -1
mov rsi, -1
shl r14b, 0
shl r13b, 1
shl r12b, 2
shl r11b, 3
shl r10b, 4
shl r9b, 5
shl r8b, 6
shl bpl, 7
shl spl, 8
shl dil, 9
shl sil, 10
hlt
Loaded 100 of 106 files, more files were not shown because too many files have changed in this diff. Show more